Compare commits
241
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
92615f474b | ||
|
|
82d71b7702 | ||
|
|
47bfae07d2 | ||
|
|
35ed8a2fcb | ||
|
|
b1fcf15937 | ||
|
|
ed9414ebda | ||
|
|
5fea326988 | ||
|
|
dc5d2c8caa | ||
|
|
c30b381eb2 | ||
|
|
79334d4840 | ||
|
|
f49e781102 | ||
|
|
aab54d4825 | ||
|
|
a09c485fc0 | ||
|
|
8c1d22a658 | ||
|
|
6a56260768 | ||
|
|
97bc190fc2 | ||
|
|
17dd05ec9d | ||
|
|
7bf4f12584 | ||
|
|
2066623986 | ||
|
|
dc0bff764d | ||
|
|
dfe8d7c28d | ||
|
|
fad44669d9 | ||
|
|
29ad93d145 | ||
|
|
f2dbf30e81 | ||
|
|
2b4e43042a | ||
|
|
0f9390aab4 | ||
|
|
d7ad2838ec | ||
|
|
c6d68dbc7b | ||
|
|
c83a10d7c2 | ||
|
|
ca22c326a4 | ||
|
|
3bbe6df6c7 | ||
|
|
71031c812e | ||
|
|
e43ddd3cbe | ||
|
|
a64ba08e27 | ||
|
|
26f54851d1 | ||
|
|
f02a2dc030 | ||
|
|
bb8c3a537b | ||
|
|
04ae3532cc | ||
|
|
7bb5ff4719 | ||
|
|
5b7ceefa9a | ||
|
|
4a2fae8495 | ||
|
|
461e1dac78 | ||
|
|
9c69bfcd80 | ||
|
|
6010f4295b | ||
|
|
9b8e315b49 | ||
|
|
9a01543477 | ||
|
|
59aab06fe1 | ||
|
|
4f06d30e07 | ||
|
|
211890f361 | ||
|
|
1c88b87ec5 | ||
|
|
b993ad1c64 | ||
|
|
76f293eb28 | ||
|
|
d0006e9f71 | ||
|
|
54559aebc3 | ||
|
|
715863799f | ||
|
|
6da68fffb8 | ||
|
|
daf7ed4c2b | ||
|
|
8598537a35 | ||
|
|
53ce1b1a5e | ||
|
|
220361ad94 | ||
|
|
433f66add8 | ||
|
|
6e6ca94338 | ||
|
|
d5d121a21b | ||
|
|
1ca2b50406 | ||
|
|
9bc021e9c0 | ||
|
|
2f4dec8323 | ||
|
|
3a9d634c17 | ||
|
|
a81db75402 | ||
|
|
930dc24632 | ||
|
|
2068bae341 | ||
|
|
7af40fb5ff | ||
|
|
619f679077 | ||
|
|
9517834913 | ||
|
|
824c42f7e3 | ||
|
|
578c44b685 | ||
|
|
3b68d15593 | ||
|
|
41622c5985 | ||
|
|
a4c73766f4 | ||
|
|
9f686253eb | ||
|
|
b2e28428a4 | ||
|
|
95e4aae287 | ||
|
|
301c78de20 | ||
|
|
dac40ab9b3 | ||
|
|
ccde9e8f11 | ||
|
|
1948d3dc21 | ||
|
|
c74b8da400 | ||
|
|
714190e02a | ||
|
|
069a9af7e6 | ||
|
|
5deb66c7f6 | ||
|
|
1cbbde0089 | ||
|
|
0a78da39e5 | ||
|
|
870843f999 | ||
|
|
6862049ef6 | ||
|
|
9b5289940c | ||
|
|
64e6d7b7df | ||
|
|
06c476b37d | ||
|
|
42657b3b65 | ||
|
|
35714258f0 | ||
|
|
4e269f3a7a | ||
|
|
87c30484aa | ||
|
|
36fe4785ec | ||
|
|
1232789b41 | ||
|
|
2976c21ee6 | ||
|
|
2602605c83 | ||
|
|
ae31e1e852 | ||
|
|
572a3cf8b8 | ||
|
|
22f9547fd8 | ||
|
|
2e4ed38c51 | ||
|
|
a7a283f449 | ||
|
|
e42756b27f | ||
|
|
6f74552617 | ||
|
|
f7ef719bd6 | ||
|
|
38506a753f | ||
|
|
0987f93c67 | ||
|
|
103d0df289 | ||
|
|
b179610e7f | ||
|
|
cad5e44703 | ||
|
|
4b436ea7d0 | ||
|
|
8ad3641fd3 | ||
|
|
7e18dccbd9 | ||
|
|
205207abb0 | ||
|
|
0041542fe2 | ||
|
|
a87a7d1da2 | ||
|
|
bc54effbd0 | ||
|
|
c4d089f931 | ||
|
|
9504fa8bbd | ||
|
|
18bca47977 | ||
|
|
1f144705e2 | ||
|
|
e1d844bfed | ||
|
|
06e95254f0 | ||
|
|
b6a8989c77 | ||
|
|
3acfb039a9 | ||
|
|
a58b5d6e69 | ||
|
|
5e935dffb4 | ||
|
|
5c5c1fdf77 | ||
|
|
baf3a474df | ||
|
|
0ae05cb9bc | ||
|
|
82464f4054 | ||
|
|
2a5d6571ec | ||
|
|
c33c69b3f3 | ||
|
|
37c3e5dc39 | ||
|
|
784369cc25 | ||
|
|
2baf726ee6 | ||
|
|
67b4889984 | ||
|
|
e0b87a0ae5 | ||
|
|
34173e079c | ||
|
|
d2eaca4949 | ||
|
|
fd558ce5d8 | ||
|
|
ae1161524d | ||
|
|
d456a763fa | ||
|
|
cc56aeeacd | ||
|
|
44fe8d2eed | ||
|
|
d03d982e3b | ||
|
|
657b5bc1b3 | ||
|
|
b6ca778cef | ||
|
|
73f82a2305 | ||
|
|
cc7dc8ac14 | ||
|
|
e151759212 | ||
|
|
60df5087e9 | ||
|
|
78e3befbbb | ||
|
|
d7e69fbe77 | ||
|
|
4ff4d2acc9 | ||
|
|
c9aa09f341 | ||
|
|
b4afc8cefd | ||
|
|
0088ecaf00 | ||
|
|
e0536d344f | ||
|
|
467e35504c | ||
|
|
bf5c72e3ba | ||
|
|
5d59c57c98 | ||
|
|
a54b16676a | ||
|
|
2d0d8a682b | ||
|
|
eb35c75514 | ||
|
|
89657a06c4 | ||
|
|
d542b08ced | ||
|
|
499b87c482 | ||
|
|
ba3ea3012c | ||
|
|
3428fb4190 | ||
|
|
ef14622ba0 | ||
|
|
24c52abf6b | ||
|
|
a3f8f67c93 | ||
|
|
0b60fd6557 | ||
|
|
fc8fe329d2 | ||
|
|
e593444eea | ||
|
|
6d0015cabc | ||
|
|
67cd2da561 | ||
|
|
18d6583e83 | ||
|
|
fe259e6d38 | ||
|
|
f80e3b33b0 | ||
|
|
dc99c15ffa | ||
|
|
e9f6d68bd7 | ||
|
|
3a0d9e24ea | ||
|
|
5adc328a2b | ||
|
|
6a636c58e7 | ||
|
|
9301739910 | ||
|
|
b3859f6dad | ||
|
|
dc0a05e5e9 | ||
|
|
9cca5f3dd2 | ||
|
|
d2fe0110a0 | ||
|
|
edd5f813b2 | ||
|
|
9f759150b8 | ||
|
|
347464a057 | ||
|
|
1301a57de4 | ||
|
|
3b2b4e1dca | ||
|
|
b9ba43a5bf | ||
|
|
a1e5a4af8c | ||
|
|
188e83c4d6 | ||
|
|
db5ed6042b | ||
|
|
a942afe6c4 | ||
|
|
0254a99336 | ||
|
|
15c75d2225 | ||
|
|
8b34f9da0a | ||
|
|
2d95e0fcc6 | ||
|
|
8ba1c5b87c | ||
|
|
df58b5fb90 | ||
|
|
ecda200180 | ||
|
|
b70d5f3efa | ||
|
|
fa6ba8a162 | ||
|
|
a20975688d | ||
|
|
25bc2a3291 | ||
|
|
f1e4809930 | ||
|
|
0752b5d242 | ||
|
|
c040bd4674 | ||
|
|
f0c9ffb25e | ||
|
|
04d9df559e | ||
|
|
79256f9093 | ||
|
|
1c455b6ec0 | ||
|
|
c3f282ba44 | ||
|
|
5eb89f8830 | ||
|
|
a6c15afec1 | ||
|
|
dc1d0e045f | ||
|
|
6d4a0d12ec | ||
|
|
6868b345ee | ||
|
|
64b6eb5d54 | ||
|
|
f21f81f9b5 | ||
|
|
badc4e636b | ||
|
|
da6a864463 | ||
|
|
b7a63a5579 | ||
|
|
08061b7b8a | ||
|
|
5494696227 | ||
|
|
c1d2bad901 | ||
|
|
243f52dc79 |
+338
-29
@@ -23,6 +23,7 @@ import json
|
||||
import os
|
||||
import uuid
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
from control_plane_db import (
|
||||
@@ -53,6 +54,8 @@ OUTCOME_CANDIDATE_SET_DRIFT = "candidate_set_drift"
|
||||
SKIP_CLAIMED_BY_OTHER_SESSION = "claimed_by_other_session"
|
||||
# #776: controller-supplied pre-rank exclusion.
|
||||
SKIP_EXCLUDED_BY_CONTROLLER = "excluded_by_controller"
|
||||
# #844: epic / child-only implementation container (pre-rank).
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER = "epic_or_child_only_container"
|
||||
|
||||
# Ownership verdicts for a live claim on a candidate (#765).
|
||||
OWNERSHIP_OWN = "own"
|
||||
@@ -130,6 +133,71 @@ ROLE_ACTIONS: dict[str, tuple[tuple[str, ...], tuple[str, ...]]] = {
|
||||
}
|
||||
|
||||
|
||||
# Body phrases that prove an issue is an implementation container, not a
|
||||
# unit of direct author work (#844 / #854). Matched case-insensitively against
|
||||
# the issue body. Title alone is never sufficient (ordinary issues may mention
|
||||
# "epic", "roadmap", "vision", or "umbrella" incidentally).
|
||||
#
|
||||
# #854 extends the #844 marker set so product-vision (#652), phased-roadmap
|
||||
# (#653), and umbrella (#655) coordination records — which do not use the word
|
||||
# "epic" — are classified with the same semantic exclusion as epic containers.
|
||||
_CHILD_ONLY_BODY_MARKERS: tuple[str, ...] = (
|
||||
# Epic / child-only (#844, live #631)
|
||||
"implementation is delivered via child issues only",
|
||||
"implementation is delivered through child issues only",
|
||||
"implementation is delivered via child issues",
|
||||
"implementation is delivered through child issues",
|
||||
"do not implement product features in this epic",
|
||||
"do not implement product features in this epic issue itself",
|
||||
"no product feature implementation is claimed complete solely on this epic",
|
||||
"implementable child issues remain independently eligible",
|
||||
"owns the product roadmap and linkage",
|
||||
"this epic owns the product roadmap",
|
||||
"coordination container",
|
||||
"child-only container",
|
||||
"implementation is delegated to child",
|
||||
# Vision / roadmap / umbrella coordination (#854, live #652/#653/#655).
|
||||
# Prefer authoritative non-implementation / child-only scope language over
|
||||
# bare words like "roadmap" so ordinary implementable issues that mention
|
||||
# a parent vision or roadmap stay eligible.
|
||||
"do not implement features on this issue",
|
||||
"implementing features on this roadmap issue",
|
||||
"implementation is via linked children only",
|
||||
"no product feature claimed complete on this issue alone",
|
||||
"phased delivery roadmap and epic sequencing",
|
||||
"this issue is the enduring source of truth",
|
||||
"enduring source of truth for the",
|
||||
"canonical product vision — enduring source of truth",
|
||||
"canonical product vision - enduring source of truth",
|
||||
"state: vision-active",
|
||||
"state: roadmap-active",
|
||||
)
|
||||
|
||||
# Explicit epic / umbrella / vision / roadmap labels (structured evidence
|
||||
# preferred over title). Tracker alone is *not* included — ordinary issues
|
||||
# may carry a tracker label without being non-implementable containers.
|
||||
_EPIC_LABELS: frozenset[str] = frozenset(
|
||||
{
|
||||
"type:epic",
|
||||
"epic",
|
||||
"kind:epic",
|
||||
"scope:epic",
|
||||
"type:umbrella",
|
||||
"umbrella",
|
||||
"kind:umbrella",
|
||||
"scope:umbrella",
|
||||
"type:vision",
|
||||
"vision",
|
||||
"kind:vision",
|
||||
"scope:vision",
|
||||
"type:roadmap",
|
||||
"roadmap",
|
||||
"kind:roadmap",
|
||||
"scope:roadmap",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class WorkCandidate:
|
||||
"""One assignable Gitea issue or PR presented to the allocator."""
|
||||
@@ -139,6 +207,7 @@ class WorkCandidate:
|
||||
state: str = "open"
|
||||
labels: tuple[str, ...] = ()
|
||||
title: str = ""
|
||||
body: str = ""
|
||||
priority: int = 0
|
||||
head_sha: str | None = None
|
||||
# Routing signals (callers derive from Gitea / review feedback).
|
||||
@@ -158,6 +227,7 @@ class WorkCandidate:
|
||||
self.labels = tuple(
|
||||
str(x).strip().lower() for x in (self.labels or ()) if str(x).strip()
|
||||
)
|
||||
self.body = str(self.body or "")
|
||||
if self.kind not in WORK_KINDS:
|
||||
raise InvalidWorkKindError(
|
||||
f"candidate kind '{self.kind}' is not assignable; only "
|
||||
@@ -171,6 +241,7 @@ class WorkCandidate:
|
||||
"state": self.state,
|
||||
"labels": list(self.labels),
|
||||
"title": self.title,
|
||||
"body": self.body,
|
||||
"priority": self.priority,
|
||||
"head_sha": self.head_sha,
|
||||
"request_changes_current_head": self.request_changes_current_head,
|
||||
@@ -184,6 +255,83 @@ class WorkCandidate:
|
||||
}
|
||||
|
||||
|
||||
def _title_container_prefix(title_l: str) -> str | None:
|
||||
"""Return a coordination-title prefix token if *title_l* uses one (#854).
|
||||
|
||||
Title prefixes alone never exclude; they only corroborate body/label
|
||||
evidence. Ordinary issues may say "roadmap" or "vision" mid-title.
|
||||
"""
|
||||
for prefix, token in (
|
||||
("epic:", "title_epic_prefix"),
|
||||
("epic ", "title_epic_prefix"),
|
||||
("umbrella:", "title_umbrella_prefix"),
|
||||
("umbrella ", "title_umbrella_prefix"),
|
||||
("roadmap:", "title_roadmap_prefix"),
|
||||
("roadmap ", "title_roadmap_prefix"),
|
||||
("product vision:", "title_vision_prefix"),
|
||||
("product vision ", "title_vision_prefix"),
|
||||
("vision:", "title_vision_prefix"),
|
||||
("vision ", "title_vision_prefix"),
|
||||
):
|
||||
if title_l.startswith(prefix):
|
||||
return token
|
||||
return None
|
||||
|
||||
|
||||
def classify_epic_or_child_only_container(
|
||||
c: WorkCandidate,
|
||||
) -> tuple[bool, str | None]:
|
||||
"""Return whether *c* is a non-implementable coordination container (#844/#854).
|
||||
|
||||
Exclusion uses structured evidence first (labels, body scope language).
|
||||
A bare title containing the words "epic", "roadmap", "vision", or
|
||||
"umbrella" is **not** enough — ordinary implementable issues may mention
|
||||
those terms incidentally. Explicit title prefixes (``Epic:``, ``Roadmap:``,
|
||||
``Product vision:``, ``Umbrella:``) only count when the body also proves
|
||||
child-only / no-direct-implementation scope (or a container label is
|
||||
present).
|
||||
|
||||
Covers epic, product-vision, phased-roadmap, umbrella, and child-only
|
||||
records so the allocator never assigns coordination containers as direct
|
||||
author work.
|
||||
|
||||
PRs are never classified as containers here (they already have a head).
|
||||
"""
|
||||
if c.kind != "issue":
|
||||
return False, None
|
||||
|
||||
labels = set(c.labels)
|
||||
epic_label = sorted(labels & _EPIC_LABELS)
|
||||
body_l = (c.body or "").lower()
|
||||
title = (c.title or "").strip()
|
||||
title_l = title.lower()
|
||||
|
||||
body_hits = [m for m in _CHILD_ONLY_BODY_MARKERS if m in body_l]
|
||||
title_prefix = _title_container_prefix(title_l)
|
||||
|
||||
if epic_label:
|
||||
detail = f"label={epic_label[0]}"
|
||||
if body_hits:
|
||||
detail = f"{detail}; body_marker={body_hits[0]!r}"
|
||||
if title_prefix:
|
||||
detail = f"{title_prefix}; {detail}"
|
||||
return True, detail
|
||||
|
||||
if body_hits:
|
||||
# Body proves child-only / vision / roadmap / umbrella scope. Title
|
||||
# prefixes are corroborating but not required — containers without the
|
||||
# title word still exclude.
|
||||
detail = f"body_marker={body_hits[0]!r}"
|
||||
if title_prefix:
|
||||
detail = f"{title_prefix}; {detail}"
|
||||
return True, detail
|
||||
|
||||
# Title-only coordination prefix without body scope evidence is
|
||||
# insufficient (#844/#854 AC: eligibility does not rely solely on a title
|
||||
# word). Incidental mid-title mentions without markers stay eligible.
|
||||
return False, None
|
||||
|
||||
|
||||
@dataclass
|
||||
class SkipRecord:
|
||||
kind: str
|
||||
@@ -591,6 +739,46 @@ def normalize_exclude_issue_numbers(
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def _claim_expires_at(claim: Any) -> datetime | None:
|
||||
"""Parse a claim's ``expires_at``, or ``None`` when it is absent/malformed."""
|
||||
if not isinstance(claim, Mapping):
|
||||
return None
|
||||
text = str(claim.get("expires_at") or "").strip()
|
||||
if not text:
|
||||
return None
|
||||
if text.endswith("Z"):
|
||||
text = text[:-1] + "+00:00"
|
||||
try:
|
||||
parsed = datetime.fromisoformat(text)
|
||||
except ValueError:
|
||||
return None
|
||||
if parsed.tzinfo is None:
|
||||
parsed = parsed.replace(tzinfo=timezone.utc)
|
||||
return parsed.astimezone(timezone.utc)
|
||||
|
||||
|
||||
def _drop_expired_claims(
|
||||
claims: Mapping[tuple[str, int], dict[str, Any]],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
) -> dict[tuple[str, int], dict[str, Any]]:
|
||||
"""Claims minus those whose lease has already expired (#643).
|
||||
|
||||
The read-only mirror of ``expire_stale_leases``: the sweep marks such rows
|
||||
``expired`` so they stop being returned as claims, and this reaches the same
|
||||
view without writing. A claim with no parseable ``expires_at`` is **kept** —
|
||||
an unreadable expiry is not evidence that work is free.
|
||||
"""
|
||||
moment = now or datetime.now(timezone.utc)
|
||||
kept: dict[tuple[str, int], dict[str, Any]] = {}
|
||||
for key, claim in (claims or {}).items():
|
||||
expires_at = _claim_expires_at(claim)
|
||||
if expires_at is not None and expires_at <= moment:
|
||||
continue
|
||||
kept[key] = claim
|
||||
return kept
|
||||
|
||||
|
||||
def candidate_set_fingerprint(
|
||||
candidates: Sequence[WorkCandidate],
|
||||
*,
|
||||
@@ -679,12 +867,22 @@ def allocate_next_work(
|
||||
exclude_issue_numbers: Sequence[int] | None = None,
|
||||
expected_candidate_set_fingerprint: str | None = None,
|
||||
allocation_mode: str | None = None,
|
||||
side_effect_free: bool = False,
|
||||
) -> dict[str, Any]:
|
||||
"""Select and optionally reserve the next work unit via control-plane DB.
|
||||
|
||||
*apply=False* (default): dry-run selection only — no lease/assignment.
|
||||
*apply=True*: atomic ``assign_and_lease`` for the selected candidate.
|
||||
|
||||
*side_effect_free* (#643): a dry run that writes **nothing** to the
|
||||
control-plane DB. A plain ``apply=False`` still registered a session row and
|
||||
swept stale leases globally, so a caller advertising a read-only preview was
|
||||
mutating on every call. Under this flag both writes are suppressed and stale
|
||||
leases are instead filtered out of the claim map in memory, which yields the
|
||||
same selection the sweep would have produced without persisting anything.
|
||||
Incompatible with *apply* — the combination fails closed rather than
|
||||
silently reserving.
|
||||
|
||||
*allocation_mode* (#840): ``cross_role`` (default for controller) inspects
|
||||
the complete queue and returns one authoritative selection naming the
|
||||
required downstream role/profile/action. ``role_scoped`` keeps prior
|
||||
@@ -738,40 +936,57 @@ def allocate_next_work(
|
||||
"allocation_mode": (allocation_mode or "").strip() or None,
|
||||
}
|
||||
|
||||
session_id = (session_id or "").strip() or f"alloc-{uuid.uuid4().hex[:12]}"
|
||||
try:
|
||||
db.upsert_session(
|
||||
session_id=session_id,
|
||||
role=role_norm,
|
||||
profile=profile_name,
|
||||
pid=os.getpid(),
|
||||
controller_instance_id=controller_instance_id,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — surface structured
|
||||
# A side-effect-free run may never reserve: reserving is a write, and the
|
||||
# flag is the caller's assertion that this call writes nothing (#643).
|
||||
if side_effect_free and apply:
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"apply": True,
|
||||
"reasons": [
|
||||
f"failed to register session in control-plane DB: {exc} "
|
||||
"(fail closed, #613)"
|
||||
"side_effect_free is incompatible with apply=True; an "
|
||||
"assignment is a write (fail closed, #643)"
|
||||
],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
# Expire stale leases globally before selection.
|
||||
try:
|
||||
db.expire_stale_leases()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [f"lease expiry failed: {exc} (fail closed)"],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
session_id = (session_id or "").strip() or f"alloc-{uuid.uuid4().hex[:12]}"
|
||||
if not side_effect_free:
|
||||
try:
|
||||
db.upsert_session(
|
||||
session_id=session_id,
|
||||
role=role_norm,
|
||||
profile=profile_name,
|
||||
pid=os.getpid(),
|
||||
controller_instance_id=controller_instance_id,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — surface structured
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [
|
||||
f"failed to register session in control-plane DB: {exc} "
|
||||
"(fail closed, #613)"
|
||||
],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
# Expire stale leases globally before selection.
|
||||
try:
|
||||
db.expire_stale_leases()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [f"lease expiry failed: {exc} (fail closed)"],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
terminal = None
|
||||
try:
|
||||
@@ -806,6 +1021,12 @@ def allocate_next_work(
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
if side_effect_free:
|
||||
# ``list_active_claims`` filters on status alone, so without the
|
||||
# global sweep an already-expired lease would still read as a live
|
||||
# claim and the preview would report work as taken that is free.
|
||||
# Drop those in memory: same view the sweep produces, no write.
|
||||
claims = _drop_expired_claims(claims)
|
||||
|
||||
try:
|
||||
exclude_nums = normalize_exclude_issue_numbers(exclude_issue_numbers)
|
||||
@@ -850,7 +1071,8 @@ def allocate_next_work(
|
||||
ownership_defects: list[dict[str, Any]] = []
|
||||
controller_excluded: list[dict[str, Any]] = []
|
||||
|
||||
# #776 AC2: remove excluded numbers *before* ranking / selection / lease.
|
||||
# #776 AC2 + #844: remove excluded numbers *and* epic/child-only containers
|
||||
# *before* ranking / selection / lease so they never receive assignments.
|
||||
rankable: list[WorkCandidate] = []
|
||||
for c in candidates:
|
||||
if int(c.number) in exclude_set:
|
||||
@@ -929,6 +1151,23 @@ def allocate_next_work(
|
||||
},
|
||||
}
|
||||
continue
|
||||
# #844: epics / child-only containers are never direct implement targets.
|
||||
is_container, container_detail = classify_epic_or_child_only_container(c)
|
||||
if is_container:
|
||||
detail = container_detail or "epic or child-only container"
|
||||
reason = (
|
||||
f"{c.kind}#{c.number} {SKIP_EPIC_OR_CHILD_ONLY_CONTAINER}: "
|
||||
f"{detail}; implementation is delegated to child issues"
|
||||
)
|
||||
skipped.append(
|
||||
SkipRecord(
|
||||
c.kind,
|
||||
c.number,
|
||||
reason,
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
)
|
||||
)
|
||||
continue
|
||||
rankable.append(c)
|
||||
|
||||
ordered = sort_candidates(rankable)
|
||||
@@ -1115,6 +1354,12 @@ def allocate_next_work(
|
||||
"reasons": [
|
||||
"dry-run only (apply=false); no assignment/lease created — "
|
||||
"call again with apply=true to reserve via control-plane DB"
|
||||
+ (
|
||||
"; after apply, the required-role worker consumes via "
|
||||
"gitea_adopt_workflow_lease (#843)"
|
||||
if mode == ALLOCATION_MODE_CROSS_ROLE and expected_role != role_norm
|
||||
else ""
|
||||
)
|
||||
],
|
||||
"skipped": [s.as_dict() for s in skipped],
|
||||
"terminal_pr": terminal_pr,
|
||||
@@ -1146,6 +1391,9 @@ def allocate_next_work(
|
||||
# Atomic reserve via #613 substrate.
|
||||
ttl = lease_ttl_seconds if lease_ttl_seconds is not None else None
|
||||
try:
|
||||
cross_role_handoff = (
|
||||
mode == ALLOCATION_MODE_CROSS_ROLE and lease_role != role_norm
|
||||
)
|
||||
kwargs: dict[str, Any] = {
|
||||
"session_id": session_id,
|
||||
"role": lease_role,
|
||||
@@ -1157,7 +1405,8 @@ def allocate_next_work(
|
||||
"expected_head_sha": selected.head_sha,
|
||||
"allowed_actions": allowed,
|
||||
"forbidden_actions": forbidden,
|
||||
"phase": "allocated",
|
||||
# #843: mark cross-role allocations as awaiting independent consume
|
||||
"phase": "awaiting_handoff" if cross_role_handoff else "allocated",
|
||||
}
|
||||
if ttl is not None:
|
||||
kwargs["lease_ttl_seconds"] = int(ttl)
|
||||
@@ -1237,7 +1486,52 @@ def allocate_next_work(
|
||||
"lease_role": lease_role,
|
||||
"source": "control_plane_db.assign_and_lease",
|
||||
}
|
||||
return {
|
||||
consume_allocation = None
|
||||
if cross_role_handoff and result.lease_id:
|
||||
# Durable handoff marker so independent required-role workers can
|
||||
# consume without sharing the controller session (#843).
|
||||
handoff_prov = {
|
||||
"cross_role_handoff": True,
|
||||
"handoff_status": "pending",
|
||||
"allocating_session_id": session_id,
|
||||
"allocating_role": role_norm,
|
||||
"required_role": expected_role,
|
||||
"required_profile": selection["required_profile"],
|
||||
"required_namespace": selection["required_namespace"],
|
||||
"assignment_id": result.assignment_id,
|
||||
"lease_id": result.lease_id,
|
||||
"allocation_mode": mode,
|
||||
"adopted_by_session_id": None,
|
||||
}
|
||||
try:
|
||||
db.attach_lease_provenance(result.lease_id, handoff_prov)
|
||||
except ControlPlaneError:
|
||||
# Still return assignment evidence; consume path may be unavailable
|
||||
handoff_prov["attach_failed"] = True
|
||||
consume_allocation = {
|
||||
"tool": "gitea_adopt_workflow_lease",
|
||||
"lease_id": result.lease_id,
|
||||
"assignment_id": result.assignment_id,
|
||||
"required_role": expected_role,
|
||||
"required_profile": selection["required_profile"],
|
||||
"required_namespace": selection["required_namespace"],
|
||||
"handoff_status": "pending",
|
||||
"controller_session_required": False,
|
||||
"instructions": (
|
||||
f"From an independent {expected_role} session "
|
||||
f"({selection['required_namespace']} / "
|
||||
f"{selection['required_profile']}), call "
|
||||
f"gitea_adopt_workflow_lease(lease_id={result.lease_id!r}) "
|
||||
"to consume this controller allocation. The allocating "
|
||||
"controller process does not need to remain alive. Wrong-role "
|
||||
"and second-adoption attempts fail closed."
|
||||
),
|
||||
}
|
||||
lease_proof["cross_role_handoff"] = True
|
||||
lease_proof["handoff_status"] = "pending"
|
||||
lease_proof["consume_tool"] = "gitea_adopt_workflow_lease"
|
||||
|
||||
out = {
|
||||
"success": True,
|
||||
"outcome": OUTCOME_ASSIGNED,
|
||||
"apply": True,
|
||||
@@ -1271,8 +1565,17 @@ def allocate_next_work(
|
||||
"lease_role": lease_role,
|
||||
"lease_proof": lease_proof,
|
||||
"selection_policy": SELECTION_POLICY,
|
||||
"cross_role_handoff": bool(cross_role_handoff),
|
||||
},
|
||||
"next_valid_command": _next_command(lease_role, selected),
|
||||
"next_valid_command": (
|
||||
(
|
||||
f"consume lease {result.lease_id} via gitea_adopt_workflow_lease "
|
||||
f"as {expected_role}, then "
|
||||
)
|
||||
+ _next_command(lease_role, selected)
|
||||
if cross_role_handoff
|
||||
else _next_command(lease_role, selected)
|
||||
),
|
||||
"substrate": "control_plane_db",
|
||||
"file_lock_only": False,
|
||||
"comment_lease_only": False,
|
||||
@@ -1287,9 +1590,14 @@ def allocate_next_work(
|
||||
"downstream_note": (
|
||||
"#612 incident bridge remains downstream of #600; "
|
||||
"allocator never assigns raw monitoring incidents; "
|
||||
"controller routes only under cross_role (#840)"
|
||||
"controller routes only under cross_role (#840); "
|
||||
"cross-role assignments are consumable by independent "
|
||||
"required-role workers via gitea_adopt_workflow_lease (#843)"
|
||||
),
|
||||
}
|
||||
if consume_allocation is not None:
|
||||
out["consume_allocation"] = consume_allocation
|
||||
return out
|
||||
|
||||
|
||||
def _next_command(role: str, c: WorkCandidate) -> str:
|
||||
@@ -1341,6 +1649,7 @@ def candidate_from_dict(data: dict[str, Any]) -> WorkCandidate:
|
||||
state=str(data.get("state") or "open"),
|
||||
labels=tuple(data.get("labels") or ()),
|
||||
title=str(data.get("title") or ""),
|
||||
body=str(data.get("body") or ""),
|
||||
priority=priority,
|
||||
head_sha=data.get("head_sha"),
|
||||
request_changes_current_head=bool(data.get("request_changes_current_head")),
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+80
-37
@@ -40,22 +40,88 @@ def _normalize_path(path: str) -> str:
|
||||
return (path or "").replace("\\", "/").rstrip("/")
|
||||
|
||||
|
||||
def get_canonical_branches_root(project_root: str | None = None) -> str:
|
||||
"""Return the absolute path of the canonical branches directory for *project_root*."""
|
||||
root = os.path.realpath(project_root) if project_root else os.path.realpath(os.getcwd())
|
||||
canonical_repo_root = resolve_canonical_repo_root(root, root)
|
||||
return os.path.realpath(os.path.join(canonical_repo_root, "branches"))
|
||||
|
||||
|
||||
def is_path_under_branches(path: str, project_root: str | None = None) -> bool:
|
||||
"""True when *path* resolves inside ``<project_root>/branches/``."""
|
||||
normalized = _normalize_path(path)
|
||||
if not normalized:
|
||||
"""True when *path* resolves inside a canonical ``branches/`` directory."""
|
||||
if not path or not str(path).strip():
|
||||
return False
|
||||
if "/branches/" in f"{normalized}/":
|
||||
return True
|
||||
if normalized.endswith("/branches"):
|
||||
return True
|
||||
if project_root:
|
||||
root = _normalize_path(os.path.realpath(project_root))
|
||||
real = _normalize_path(os.path.realpath(path))
|
||||
if real.startswith(f"{root}/"):
|
||||
rel = real[len(root) + 1 :]
|
||||
return rel == "branches" or rel.startswith("branches/")
|
||||
return False
|
||||
try:
|
||||
real_path = os.path.realpath(os.path.abspath(str(path).strip()))
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
branches_root = get_canonical_branches_root(project_root or real_path)
|
||||
try:
|
||||
common = os.path.commonpath([branches_root, real_path])
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
if common != branches_root:
|
||||
return False
|
||||
|
||||
rel = os.path.relpath(real_path, branches_root)
|
||||
return rel != "." and not rel.startswith("..")
|
||||
|
||||
|
||||
def resolve_canonical_repo_root(workspace_path: str, fallback_project_root: str) -> str:
|
||||
"""Return the stable repository root for *workspace_path* via git metadata (#460)."""
|
||||
p = (workspace_path or "").strip()
|
||||
if p:
|
||||
try:
|
||||
res = subprocess.run(
|
||||
["git", "-C", p, "rev-parse", "--git-common-dir"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
)
|
||||
common = _realpath_git_common_dir(p, res.stdout)
|
||||
if common.endswith(f"{os.sep}.git") or os.path.basename(common) == ".git":
|
||||
candidate_root = os.path.dirname(common)
|
||||
real_p = os.path.realpath(p)
|
||||
try:
|
||||
if os.path.commonpath([candidate_root, real_p]) == candidate_root:
|
||||
return candidate_root
|
||||
except Exception:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Fallback when git metadata is unavailable. Never string-split on
|
||||
# "/branches/" (review #531 F2 / #551): recover the repo root only via
|
||||
# resolved-path commonpath ancestry. Do **not** require on-disk isdir —
|
||||
# MCP may launch with project_root = branches/<wt> before that path
|
||||
# exists, and #274 path-shaped worktree-as-project-root must still resolve.
|
||||
fallback = os.path.realpath(fallback_project_root or workspace_path or ".")
|
||||
cur = fallback
|
||||
for _ in range(64):
|
||||
parent = os.path.dirname(cur)
|
||||
if parent == cur:
|
||||
break
|
||||
branches_dir = os.path.realpath(os.path.join(parent, "branches"))
|
||||
try:
|
||||
# Path-shaped: fallback is under parent/branches/ (commonpath).
|
||||
if os.path.commonpath([branches_dir, fallback]) == branches_dir:
|
||||
return parent
|
||||
except ValueError:
|
||||
pass
|
||||
# Fallback path itself is the branches directory.
|
||||
if os.path.basename(os.path.realpath(cur)) == "branches":
|
||||
try:
|
||||
if os.path.commonpath([os.path.realpath(cur), fallback]) == os.path.realpath(
|
||||
cur
|
||||
):
|
||||
return parent
|
||||
except ValueError:
|
||||
pass
|
||||
cur = parent
|
||||
|
||||
return fallback
|
||||
|
||||
|
||||
def resolve_mutation_workspace(
|
||||
@@ -87,29 +153,6 @@ def _realpath_git_common_dir(workspace_path: str, common_dir: str) -> str:
|
||||
return os.path.realpath(os.path.join(workspace_path, raw))
|
||||
|
||||
|
||||
def resolve_canonical_repo_root(workspace_path: str, fallback_project_root: str) -> str:
|
||||
"""Return the stable repository root for *workspace_path* via git metadata (#460)."""
|
||||
path = (workspace_path or "").strip()
|
||||
fallback = os.path.realpath(fallback_project_root)
|
||||
if not path:
|
||||
return fallback
|
||||
try:
|
||||
res = subprocess.run(
|
||||
["git", "-C", path, "rev-parse", "--git-common-dir"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
)
|
||||
common = _realpath_git_common_dir(path, res.stdout)
|
||||
except Exception:
|
||||
return fallback
|
||||
if common.endswith(f"{os.sep}.git"):
|
||||
return os.path.dirname(common)
|
||||
if os.path.basename(common) == ".git":
|
||||
return os.path.dirname(common)
|
||||
return fallback
|
||||
|
||||
|
||||
def resolve_author_mutation_context(
|
||||
worktree_path: str | None,
|
||||
process_project_root: str,
|
||||
|
||||
+97
-1
@@ -163,7 +163,19 @@ _TERMINAL_OWNERSHIP_STATUSES = frozenset(
|
||||
{"released", "abandoned", "done", "blocked", "terminal", "closed"}
|
||||
)
|
||||
_EXPIRED_STATUSES = frozenset({"expired"})
|
||||
_STALE_STATUSES = frozenset({"stale", "stale_dead_process", "stale_missing_worktree"})
|
||||
_STALE_STATUSES = frozenset(
|
||||
{
|
||||
"stale",
|
||||
"stale_dead_process",
|
||||
"stale_missing_worktree",
|
||||
# #790 Slice A heartbeat-lifecycle bands. Listed here so they are
|
||||
# *classified* rather than falling through to the unknown-status branch;
|
||||
# they still block unless the ownership record proves
|
||||
# ``reclaim_allowed is True``, so the O2 fail-closed rule is unchanged.
|
||||
"stale_missed_heartbeat",
|
||||
"stale_absolute_cap",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _norm_str(value: Any) -> str:
|
||||
@@ -525,6 +537,90 @@ def assess_ownership_record_activity(record: dict[str, Any]) -> dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
# Reviewer-lease reclaim is only reachable from a non-live (expired/stale) lease.
|
||||
_RECLAIMABLE_REVIEWER_STATUSES = _EXPIRED_STATUSES | _STALE_STATUSES
|
||||
|
||||
|
||||
def is_active_ownership_status(status: str | None) -> bool:
|
||||
"""True when *status* denotes live/active ownership of a branch (#855).
|
||||
|
||||
Used to decide whether a *competing* active claimant still uses a branch
|
||||
when weighing an expired reviewer lease for reclaim. Expired, stale,
|
||||
released, and terminal statuses are not active.
|
||||
"""
|
||||
return _norm_str(status).lower() in _ACTIVE_OWNERSHIP_STATUSES
|
||||
|
||||
|
||||
def assess_expired_reviewer_lease_reclaim(
|
||||
*,
|
||||
role: str,
|
||||
status: str,
|
||||
pr_merged: bool | None,
|
||||
owner_pid_alive: bool | None,
|
||||
competing_active_claimant: bool | None,
|
||||
) -> dict[str, Any]:
|
||||
"""Decide, explicitly and fail-closed, whether an expired reviewer lease
|
||||
may stop protecting an already-merged branch (#855 AC4).
|
||||
|
||||
An expired reviewer lease should not protect a merged branch forever once
|
||||
its work is done and no live claimant remains. Reclaim is permitted only
|
||||
when **every** condition below is provably satisfied; any unknown
|
||||
(``None``) or contrary value keeps the lease protective:
|
||||
|
||||
- the lease is a ``reviewer`` lease (author/merger/controller/reconciler
|
||||
leases are out of scope and always keep protecting);
|
||||
- its status is expired or stale (never an active/live lease);
|
||||
- the PR is proven merged (``pr_merged is True``);
|
||||
- the lease owner process is proven dead (``owner_pid_alive is False``);
|
||||
- no competing active claimant uses the branch
|
||||
(``competing_active_claimant is False``).
|
||||
|
||||
Returns a decision dict with ``reclaim_allowed`` and, when refused, the
|
||||
fail-closed ``reasons``. The reasons never contain secrets — only the
|
||||
role, the status, and which condition was unproven.
|
||||
"""
|
||||
reasons: list[str] = []
|
||||
normalized_role = _norm_str(role).lower()
|
||||
normalized_status = _norm_str(status).lower()
|
||||
|
||||
if normalized_role != "reviewer":
|
||||
reasons.append(
|
||||
f"lease role '{normalized_role or 'unknown'}' is not a reviewer "
|
||||
"lease; expired-reviewer reclaim does not apply"
|
||||
)
|
||||
if normalized_status not in _RECLAIMABLE_REVIEWER_STATUSES:
|
||||
reasons.append(
|
||||
f"lease status '{normalized_status or 'unknown'}' is not expired "
|
||||
"or stale; only a non-live reviewer lease may be reclaimed"
|
||||
)
|
||||
if pr_merged is not True:
|
||||
reasons.append(
|
||||
"PR merged state is not proven true; reclaim requires an "
|
||||
"already-merged PR (fail closed)"
|
||||
)
|
||||
if owner_pid_alive is not False:
|
||||
reasons.append(
|
||||
"lease owner process liveness is not proven dead; a live owner "
|
||||
"still protects the branch (fail closed)"
|
||||
)
|
||||
if competing_active_claimant is not False:
|
||||
reasons.append(
|
||||
"a competing active claimant may still use the branch; reclaim "
|
||||
"requires no other active ownership (fail closed)"
|
||||
)
|
||||
|
||||
allowed = not reasons
|
||||
return {
|
||||
"reclaim_allowed": allowed,
|
||||
"role": normalized_role,
|
||||
"status": normalized_status,
|
||||
"decision": (
|
||||
"reclaim_expired_reviewer_lease" if allowed else "keep_protecting"
|
||||
),
|
||||
"reasons": [] if allowed else reasons,
|
||||
}
|
||||
|
||||
|
||||
def assess_active_branch_ownership(
|
||||
*,
|
||||
remote: str,
|
||||
|
||||
@@ -46,6 +46,17 @@ _FIELD_RE = re.compile(
|
||||
)
|
||||
|
||||
|
||||
def is_known_cth_type(value: str | None) -> bool:
|
||||
"""True when *value* is a declared member of the :data:`CTH_TYPES` contract.
|
||||
|
||||
``CTH_TYPES`` is the single authority for what a CTH type may be. The
|
||||
heading a comment carries is free text, so a *read* path that turns a parsed
|
||||
type into something durable — a serialized field, a routing decision — must
|
||||
check membership here rather than trust the parse or keep a list of its own.
|
||||
"""
|
||||
return (value or "").strip() in CTH_TYPES
|
||||
|
||||
|
||||
def format_cth_body(
|
||||
*,
|
||||
cth_type: str,
|
||||
@@ -60,7 +71,7 @@ def format_cth_body(
|
||||
) -> str:
|
||||
"""Render a canonical CTH comment body."""
|
||||
normalized_type = (cth_type or "").strip()
|
||||
if normalized_type not in CTH_TYPES:
|
||||
if not is_known_cth_type(normalized_type):
|
||||
raise ValueError(
|
||||
f"unknown CTH type '{cth_type}'; expected one of {sorted(CTH_TYPES)}"
|
||||
)
|
||||
@@ -101,6 +112,12 @@ def parse_cth_comment(body: str) -> dict[str, Any] | None:
|
||||
fields[key] = match.group(2).strip()
|
||||
return {
|
||||
"cth_type": cth_type,
|
||||
# The heading capture is unconstrained free text, so the parse states
|
||||
# whether it satisfies the CTH_TYPES contract instead of leaving every
|
||||
# reader to decide (or forget). Parsing stays total — an unknown type is
|
||||
# still parsed and reported, never raised on — but a reader that turns
|
||||
# the type into a durable value can now tell the two apart.
|
||||
"cth_type_known": is_known_cth_type(cth_type),
|
||||
"fields": fields,
|
||||
"raw_body": text,
|
||||
}
|
||||
@@ -119,7 +136,7 @@ def assess_cth_comment(body: str) -> dict[str, Any]:
|
||||
}
|
||||
|
||||
cth_type = parsed.get("cth_type") or ""
|
||||
if cth_type not in CTH_TYPES:
|
||||
if not is_known_cth_type(cth_type):
|
||||
reasons.append(
|
||||
f"unknown CTH type '{cth_type}'; expected one of {sorted(CTH_TYPES)}"
|
||||
)
|
||||
|
||||
+1045
-4
File diff suppressed because it is too large
Load Diff
@@ -241,13 +241,19 @@ def bootstrap_permits_control_checkout(
|
||||
caller's ordinary block in force.
|
||||
|
||||
``assessment`` is server-derived only: it is produced by
|
||||
:func:`assess_create_issue_bootstrap` from inspected repository state. It is
|
||||
never accepted from an MCP tool argument, so no caller can assert
|
||||
eligibility it has not proven.
|
||||
:func:`assess_create_issue_bootstrap` or
|
||||
:func:`author_issue_bootstrap.assess_author_issue_bootstrap` from inspected
|
||||
repository state. It is never accepted from an MCP tool argument, so no
|
||||
caller can assert eligibility it has not proven.
|
||||
|
||||
#892: author issue worktree bootstrap uses the same predicate with
|
||||
``task_scope='author_issue_bootstrap'`` so a clean control checkout can
|
||||
create the first ``branches/`` worktree without the lock↔worktree cycle.
|
||||
"""
|
||||
if not isinstance(assessment, dict):
|
||||
return False
|
||||
if not is_create_issue_task(task):
|
||||
import author_issue_bootstrap
|
||||
if not is_create_issue_task(task) and not author_issue_bootstrap.is_author_issue_bootstrap_task(task):
|
||||
return False
|
||||
|
||||
# Positive proof: the assessment must affirmatively allow, with no
|
||||
@@ -263,9 +269,16 @@ def bootstrap_permits_control_checkout(
|
||||
if assessment.get("reasons"):
|
||||
return False
|
||||
|
||||
# Scope proof: only the create_issue bootstrap, only via the clean
|
||||
# canonical control checkout path.
|
||||
if assessment.get("task_scope") != "create_issue_only":
|
||||
# Scope proof: create_issue (#749) or author issue bootstrap (#850/#892),
|
||||
# only via the clean canonical control checkout path.
|
||||
task_scope = assessment.get("task_scope")
|
||||
if is_create_issue_task(task):
|
||||
if task_scope != "create_issue_only":
|
||||
return False
|
||||
elif author_issue_bootstrap.is_author_issue_bootstrap_task(task):
|
||||
if task_scope != "author_issue_bootstrap":
|
||||
return False
|
||||
else:
|
||||
return False
|
||||
if assessment.get("bootstrap_path") != "clean_canonical_control_checkout":
|
||||
return False
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -14,12 +14,13 @@
|
||||
|
||||
| Capability | Notes |
|
||||
|------------|--------|
|
||||
| Schema | `sessions`, `work_items`, `leases`, `assignments`, `terminal_locks`, `events`, `incident_links` |
|
||||
| Schema | `sessions`, `work_items`, `leases`, `assignments`, `terminal_locks`, `events`, `incident_links`, `session_checkpoints` |
|
||||
| Atomic assign+lease | `ControlPlaneDB.assign_and_lease` — one `BEGIN IMMEDIATE` transaction |
|
||||
| Mutation gate | `require_valid_assignment` — live lease + allowed action + non-terminal work + non-stale head |
|
||||
| Heartbeat / release / expire | Lease lifecycle helpers |
|
||||
| Terminal-lock index | Routing signal for #600 (terminal path first) |
|
||||
| `incident_links` | Provider-neutral link model for #612 — **not** assignable work; scope keys NULL-safe |
|
||||
| `session_checkpoints` | Durable session resume state for #660 — redacted at write, reconciled (never restored) on boot |
|
||||
|
||||
## Hard rules (enforced in code)
|
||||
|
||||
@@ -92,6 +93,44 @@ Module: `incident_bridge.py` · MCP tools: `gitea_observability_*`
|
||||
- Provider tokens never appear in issue bodies, links, or tool results.
|
||||
- Allocator sees bridge work only after a Gitea issue exists.
|
||||
|
||||
## Session checkpoints (#660)
|
||||
|
||||
Module: `control_plane_db.py` · Table: `session_checkpoints` · Parent **#655** · Soft-depends **#659** · Vision **#652** · Roadmap **#653**
|
||||
|
||||
Workflow state that lived only in MCP process memory or chat did not survive a restart, so session identity, stage, lease ownership, and the next valid action had to be reconstructed by hand. The `session_checkpoints` table makes that state durable.
|
||||
|
||||
### Schema version
|
||||
|
||||
Rows carry `checkpoint_schema_version` (currently **5**, tracking the module-level `SCHEMA_VERSION`), so a reader can tell which field set a record was written under. The row identity is
|
||||
`UNIQUE (remote, org, repo, session_id, work_kind, work_number)` — one current-state row per session per work unit, upserted rather than appended. A session-level checkpoint that is not bound to an issue or PR uses the `work_number = 0` sentinel with an empty `work_kind`.
|
||||
|
||||
| Group | Columns |
|
||||
|-------|---------|
|
||||
| Identity | `session_id`, `provider_identity`, `role`, `remote`/`org`/`repo` |
|
||||
| Work unit | `work_kind` (`issue`/`pr`/empty), `work_number`, `worktree_path`, `branch`, `head_sha` |
|
||||
| Ownership | `capabilities` (JSON), `lease_id`, `assignment_id` |
|
||||
| Progress | `workflow_stage`, `last_completed_action`, `current_operation`, `pending_mutation` (JSON) |
|
||||
| Recovery | `evidence` (JSON), `blocker`, `next_valid_action`, `recovery_instructions` |
|
||||
| Bookkeeping | `checkpoint_schema_version`, `status`, `created_at`, `updated_at` |
|
||||
|
||||
### API
|
||||
|
||||
| Method | Purpose |
|
||||
|--------|---------|
|
||||
| `write_session_checkpoint` | Upsert the current-state row; redacts, and optionally enforces drain completeness |
|
||||
| `get_session_checkpoint` | Read one checkpoint by session + work unit |
|
||||
| `list_session_checkpoints` | List checkpoints by session or by work unit |
|
||||
| `reconcile_session_checkpoint` | Pure diagnosis of a stored checkpoint against live head/lease state |
|
||||
| `checkpoint_completeness` | Pure — the drain-required fields missing from a record |
|
||||
|
||||
### Hard rules
|
||||
|
||||
1. **Redaction at write.** Free-text columns are redaction-filtered and JSON columns are redacted recursively before storage, so a pasted token or credential URL cannot land in a checkpoint. Secrets are never stored (#660 AC4).
|
||||
2. **Reconcile, never blind-restore.** `reconcile_session_checkpoint` returns a diagnosis (`stale`, `head_mismatch`, `lease_mismatch`, `reconcile_action`) and the caller decides. A stored head that no longer matches live Git, or a lease that is gone or reassigned, marks the checkpoint stale (#660 AC3).
|
||||
3. **Unknown live state is not a mismatch.** A `None` live input means "not checked" and never flags staleness on its own.
|
||||
4. **Drain fails closed.** A write with `require_complete=True` refuses when any of `session_id`, `role`, `workflow_stage`, `next_valid_action`, `recovery_instructions` is missing or blank, so drain cannot complete on a checkpoint that could not resume.
|
||||
5. **No transcript storage.** Checkpoints hold resume state, not conversation history.
|
||||
|
||||
## Non-goals (intentionally deferred)
|
||||
|
||||
- Full unsupervised watchdog auto-filing (prefer explicit reconcile first)
|
||||
|
||||
@@ -0,0 +1,167 @@
|
||||
# ADR: High-availability and rolling-restart architecture for Gitea MCP control plane
|
||||
|
||||
- **Status:** Proposed (Design ADR under [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668))
|
||||
- **Date:** 2026-07-25
|
||||
- **Tracking Issue:** [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668)
|
||||
- **Policy Version:** `mcp-ha-rolling-restart/v1`
|
||||
- **Related:**
|
||||
- Parent: [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655) — Governed MCP restart coordination and zero-disruption recovery
|
||||
- Governance Policy: [#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656) / `docs/architecture/mcp-restart-governance.md`
|
||||
- Control-Plane DB Substrate: [#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613) / `docs/architecture/control-plane-db-substrate.md`
|
||||
- Runtime Policy: [#615](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/615) / `docs/architecture/mcp-stable-control-runtime-policy-adr.md`
|
||||
- Product Vision: [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) (Phase 5 Maturity)
|
||||
- Delivery Roadmap: [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||
|
||||
---
|
||||
|
||||
## 1. Context & Problem Statement
|
||||
|
||||
The Gitea MCP server operates as the authoritative **control plane** for managing issues, Pull Requests, code mutations, formal reviews, and workflow reconciliations. Under single-process governance ([#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656)), process restarts are strictly controlled using pre-flight checks, drain phases, and operator approvals.
|
||||
|
||||
However, a single-instance control plane inherently presents fundamental constraints:
|
||||
|
||||
1. **Downtime during updates:** Even a perfectly executed single-process drain requires a window where incoming client requests must be paused or rejected while the server binary or python environment reloads.
|
||||
2. **Single point of failure:** Infrastructure issues, process crashes, or unhandled host-level terminations immediately disconnect active LLM sessions and leave transient workflows incomplete.
|
||||
3. **Multi-agent concurrency bottlenecks:** High volumes of concurrent multi-LLM tasks put all lock management, lease allocation, and Gitea API interactions through a single process event loop.
|
||||
|
||||
To achieve true zero-disruption operation and seamless rolling deployments without stopping active work, the system requires a high-availability (HA), multi-instance MCP architecture.
|
||||
|
||||
---
|
||||
|
||||
## 2. Architectural Principles & Non-Goals
|
||||
|
||||
### 2.1 Core Architectural Principles
|
||||
* **Gitea as Canonical Work SoT:** Gitea remains the ultimate System of Record (SoT) for issue states, pull requests, labels, and audit comments. The MCP control plane does not duplicate domain entities.
|
||||
* **Control-Plane DB as Multi-Instance State Substrate:** The control-plane SQLite/durable database ([#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613)) acts as the single source of truth for workflow leases, session tokens, assignment records, and lock fences across all MCP nodes.
|
||||
* **Stateless Worker Nodes:** MCP role server processes (`gitea-author`, `gitea-reviewer`, `gitea-merger`, `gitea-reconciler`, `gitea-controller`) maintain no unique in-memory state; any node can handle any request given a valid session resume token.
|
||||
* **Fail-Closed Split-Brain Defense:** In any network partition or quorum loss scenario, nodes must fail closed rather than risk double-mutations or conflicting Gitea states.
|
||||
|
||||
### 2.2 Non-Goals
|
||||
* **Replacing Gitea:** We do not replace Gitea issue/PR tracking with an independent database.
|
||||
* **Immediate Multi-Node Cluster Execution in v1:** This ADR defines the target architecture and phased roadmap; immediate implementation occurs incrementally post-[#655] v1.
|
||||
|
||||
---
|
||||
|
||||
## 3. High-Availability & Rolling-Restart Architecture
|
||||
|
||||
### 3.1 Architecture Overview
|
||||
|
||||
```
|
||||
+----------------------------+
|
||||
| LLM Clients / IDE Sessions |
|
||||
+--------------+-------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| HA Proxy / Router |
|
||||
| (Health-based & Affinity) |
|
||||
+------+--------------+------+
|
||||
| |
|
||||
+--------------+ +--------------+
|
||||
v v
|
||||
+--------------------+ +--------------------+
|
||||
| MCP Instance Node A| | MCP Instance Node B|
|
||||
| (Version N) | | (Version N+1) |
|
||||
+---------+----------+ +---------+----------+
|
||||
| |
|
||||
+----------------------+----------------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| Control-Plane DB Substrate|
|
||||
| (Shared Lease & Locks) |
|
||||
+--------------+-------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| Gitea API |
|
||||
+----------------------------+
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 3.2 Key System Components
|
||||
|
||||
#### A. Multiple MCP Instance Cohorts
|
||||
* The control plane runs across $N \ge 2$ redundant process nodes.
|
||||
* Dual-namespace deployment allows running the old version (Node A) alongside a updated version (Node B) during rolling upgrades.
|
||||
|
||||
#### B. Shared Durable Session Storage & Resume Tokens
|
||||
* Session context, preflight verification proofs, and capability resolution states are stored in the shared control-plane database.
|
||||
* Client requests carry an explicit `session_id` and `resume_token`. If an MCP instance restarts or a request routes to a different instance, the target node validates the token against the database without requiring full session re-initialization.
|
||||
|
||||
#### C. Shared Lease Authority & Fencing Counters
|
||||
* Workflow leases (`gitea_allocate_next_work`, `gitea_adopt_workflow_lease`) use monotonic fencing tokens (`lease_generation_id`).
|
||||
* When Node B acquires or renews a lease, it increments the generation counter. Any delayed or out-of-order write attempt from Node A using an older generation token is rejected by database constraints.
|
||||
|
||||
#### D. Leader Election & Coordinated Drain
|
||||
* Node clusters elect a primary coordinator node for administrative background tasks (such as stale lease cleanup or incident Watchdogs).
|
||||
* During a rolling deployment:
|
||||
1. Node B (new version) is launched and registers as healthy.
|
||||
2. Router directs new session creations to Node B.
|
||||
3. Node A enters `MAINTENANCE_DRAIN` status ([#659](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/659)), completing in-flight mutations while refusing new tasks.
|
||||
4. Once all active sessions migrate or complete, Node A shuts down cleanly.
|
||||
|
||||
#### E. Idempotent Mutations & Failover Safety
|
||||
* All state-changing tool executions (PR creation, review submission, merge operations, label changes) carry a deterministic `idempotency_key`.
|
||||
* If a network connection flaps or a node fails mid-mutation, the re-issued request with the same `idempotency_key` is recognized by the control-plane substrate, returning the existing recorded result without repeating side effects on Gitea.
|
||||
|
||||
#### F. Schema Version Compatibility
|
||||
* Database migrations follow non-breaking additive patterns.
|
||||
* During rolling upgrades where Node A (Version $N$) and Node B (Version $N+1$) run concurrently, both versions operate against the shared schema without structural conflicts.
|
||||
|
||||
---
|
||||
|
||||
## 4. Split-Brain & Failure Behavior
|
||||
|
||||
### 4.1 Split-Brain Risk Scenarios & Mitigation
|
||||
|
||||
| Scenario | Risk | Mitigation Strategy |
|
||||
|---|---|---|
|
||||
| **Network Partition between Nodes** | Both Node A and Node B attempt to process operations for the same issue/PR. | **Generation Fencing:** Lease renewal requires updating the DB generation counter. The node isolated from the DB fails closed immediately. |
|
||||
| **Stale Node Recovery** | Node A recovers after a long pause and executes a queued mutation. | **Lease Expiry & TTL Fencing:** Transactions verify that `expires_at > NOW()` within the atomic SQLite transaction boundaries. |
|
||||
| **Database Connection Loss** | Node loses access to shared control-plane DB substrate. | **Strict Fail-Closed:** The node immediately marks all task capabilities as `blocked` and rejects mutation tools until DB connectivity is re-established. |
|
||||
|
||||
---
|
||||
|
||||
## 5. Phased Implementation Milestones
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
M1[Milestone 1: Shared Control-Plane DB Schema & Resume Tokens] --> M2[Milestone 2: Idempotent Mutation Layer]
|
||||
M2 --> M3[Milestone 3: Health Routing & Standby Failover]
|
||||
M3 --> M4[Milestone 4: Active-Active Rolling Deployment & Auto-Drain]
|
||||
```
|
||||
|
||||
### Milestone 1: Shared Control-Plane DB Schema & Resume Tokens (Post-#655)
|
||||
* Extend [#613] Control-Plane DB schema to store multi-instance node heartbeat records and session resume tokens.
|
||||
* Enable session lookup across instances via `session_id`.
|
||||
|
||||
### Milestone 2: Idempotent Mutation Layer & Lease Fencing
|
||||
* Add mandatory `idempotency_key` tracking to all Gitea mutation tools.
|
||||
* Implement monotonic lease fencing counters in `gitea_allocate_next_work` and `gitea_adopt_workflow_lease`.
|
||||
|
||||
### Milestone 3: Health-Based Routing & Active-Passive Standby
|
||||
* Introduce lightweight proxy/router capable of checking node health endpoints.
|
||||
* Implement active-standby failover where standby node automatically assumes work if active node fails health checks.
|
||||
|
||||
### Milestone 4: Active-Active Horizontal Deployment & Rolling Upgrade Automation
|
||||
* Enable true active-active multi-instance execution.
|
||||
* Integrate automated zero-downtime rolling upgrades coordinated with `gitea_request_mcp_restart` maintenance drain.
|
||||
|
||||
---
|
||||
|
||||
## 6. Observability & Audit Requirements
|
||||
|
||||
High-availability control plane operations must expose clear telemetry and audit trails:
|
||||
|
||||
* **Node Registry Telemetry:** Active nodes, version numbers, uptime, and heartbeat timestamps reported via `gitea_get_runtime_context`.
|
||||
* **Lease Fencing Metrics:** Tracking lease acquire latency, fence rejection counts, and lease handoff durations.
|
||||
* **Failover & Re-route Audit Logs:** Durable logging of session migrations between nodes, drain initiation, and process retirement events.
|
||||
|
||||
---
|
||||
|
||||
## 7. Tradeoffs & Accepted Risks
|
||||
|
||||
* **Increased Architectural Complexity:** Moving from a single process to a multi-instance control plane requires robust DB locking, proxy routing, and migration governance.
|
||||
* **Database Dependency:** The control-plane database substrate becomes a critical shared dependency for multi-node deployments. High availability for the underlying SQLite file system / DB must be guaranteed.
|
||||
@@ -0,0 +1,223 @@
|
||||
# ADR: MCP restart governance and authorization policy
|
||||
|
||||
- **Status:** Accepted (policy effective immediately for LLM and operator sessions; enforcement tooling may lag)
|
||||
- **Date:** 2026-07-23
|
||||
- **Tracking issue:** [#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656)
|
||||
- **Policy version:** `restart-governance/v1`
|
||||
- **Related:**
|
||||
- Umbrella: [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655) — governed MCP restart coordination and zero-disruption recovery
|
||||
- Vision: [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) — MCP Control Plane Web Console product vision (§A system health and process control)
|
||||
- Roadmap: [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653) — Control Plane Web Console phased delivery (Phase 2 restart controls)
|
||||
- Contamination guard: [#630](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/630) — blocks manual process-kill recovery
|
||||
- Console restart UX: [#642](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/642) — sanctioned restart and graceful reload
|
||||
- Existing restart / reconnect paths to inventory: [#591](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/591) — auto-restart on master advance (closed); [#584](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/584) — host auto-reconnect on transport flap
|
||||
- Stable-control runtime split: `docs/architecture/mcp-stable-control-runtime-policy-adr.md` (#615)
|
||||
- Client-namespace health: `docs/mcp-namespace-health.md` (#543)
|
||||
- Reconnect-only EOF recovery: `docs/mcp-namespace-eof-recovery.md`
|
||||
|
||||
## 1. Context
|
||||
|
||||
The Gitea MCP server is the **control plane** for real issue and PR mutations
|
||||
(create, comment, lock, review, merge, reconcile). The same process serves every
|
||||
role namespace (`gitea-author`, `gitea-reviewer`, `gitea-merger`,
|
||||
`gitea-reconciler`, `gitea-controller`) and holds the in-memory capability-gate
|
||||
code loaded at startup.
|
||||
|
||||
Restarting that process is destructive to concurrent work:
|
||||
|
||||
- It resets every session's identity, preflight, and capability-lease binding.
|
||||
- It can interrupt a mutation mid-critical-section (a lock acquire, a review
|
||||
submit, a merge), leaving durable state half-written.
|
||||
- Relaunching from the wrong checkout or worktree silently changes which code
|
||||
the control plane runs, defeating master-parity gates (#420 / #615).
|
||||
|
||||
Today there is **no durable written policy** stating who may restart MCP, under
|
||||
what conditions, that restart is a last resort, and how controller approval,
|
||||
automated safety gates, and break-glass interact. Operators and LLM sessions
|
||||
therefore invent restart behavior ad hoc, which makes concurrent multi-role work
|
||||
unsafe. #630 and #642 need this policy as their backbone.
|
||||
|
||||
This ADR defines that policy. It does **not** implement coordinator code or HA
|
||||
multi-instance restart (those are later children of #655).
|
||||
|
||||
## 2. Decision
|
||||
|
||||
### 2.1 v1 decision (recorded)
|
||||
|
||||
**Restart authority in v1 is `controller approval + automated safety gates`.**
|
||||
|
||||
A restart of the stable control runtime is authorized only when **both** hold:
|
||||
|
||||
1. A **controller** role explicitly approves the restart, recording an audit
|
||||
entry (who, why, scope, affected sessions), **and**
|
||||
2. The **automated safety gates** pass: a completed drain acknowledgement (no
|
||||
affected session is mid-critical-section) or a declared break-glass incident
|
||||
(§2.5).
|
||||
|
||||
Quorum among multiple controllers is **not** required day-one. It is deferred
|
||||
unless a later investigation (tracked under #653) proves single-controller
|
||||
approval is insufficient. This ADR records the v1 decision so enforcement code
|
||||
(#630) has a fixed target; changing it requires a superseding ADR.
|
||||
|
||||
### 2.2 Restart is a last resort — the recovery ladder
|
||||
|
||||
Restart is the **last** rung. Before any restart, exhaust the narrower
|
||||
recoveries, in order:
|
||||
|
||||
1. **Reconnect** the IDE/client MCP namespace (transport EOF, `client is
|
||||
closing: EOF`, transient `#584` flap). No process change. See
|
||||
`docs/mcp-namespace-eof-recovery.md`.
|
||||
2. **Refresh / rebind** the session workspace: re-run `gitea_whoami`,
|
||||
`gitea_resolve_task_capability`, and pass an explicit validated
|
||||
`worktree_path`. Fixes stale session context without touching the process.
|
||||
3. **Scoped restart** of a single misbehaving namespace/service (where the
|
||||
deployment supports per-service restart) rather than the whole control plane.
|
||||
4. **Full restart** of the stable control runtime process — operator-owned,
|
||||
controller-approved, drained.
|
||||
5. **Host / infrastructure restart** — the broadest action; same authorization
|
||||
as a full restart plus infrastructure ownership.
|
||||
|
||||
A session **must** try rungs 1–2 and record why they were insufficient before
|
||||
requesting a restart at rung 3 or above. Skipping straight to restart is a
|
||||
policy violation.
|
||||
|
||||
### 2.3 Authorization matrix
|
||||
|
||||
| Role | Reconnect (1) | Refresh/rebind (2) | Scoped restart (3) | Full restart (4) | Host restart (5) |
|
||||
|---|---|---|---|---|---|
|
||||
| **author** | self | self | request only | **forbidden** | forbidden |
|
||||
| **reviewer** | self | self | request only | **forbidden** | forbidden |
|
||||
| **merger** | self | self | request only | **forbidden** | forbidden |
|
||||
| **reconciler** | self | self | request only | **forbidden** | forbidden |
|
||||
| **controller** | self | self | **approve** (+gates) | **approve** (+gates) | request to operator |
|
||||
| **operator** | self | self | execute (controller-approved) | execute (controller-approved) | execute (controller-approved) |
|
||||
| **admin** | self | self | execute | execute | execute (break-glass) |
|
||||
|
||||
Legend: *self* = may perform for its own client session; *request only* = may
|
||||
raise a restart request but not authorize or execute it; *approve* = may
|
||||
authorize under §2.1 gates; *execute* = may perform the process action after the
|
||||
authorization is recorded.
|
||||
|
||||
Key invariants:
|
||||
|
||||
- **No LLM worker role (author/reviewer/merger/reconciler) may perform or
|
||||
authorize a full or host restart.** They may only reconnect/rebind their own
|
||||
client and file a restart request.
|
||||
- **Controller approval authorizes; operator/admin executes.** The approving
|
||||
controller and the executing operator may be the same human, but both the
|
||||
approval and the execution are audited.
|
||||
- Privileged process actions (full restart, host restart) are reserved to
|
||||
**operator/admin**, never to an automated worker.
|
||||
|
||||
### 2.4 Approved conditions
|
||||
|
||||
A restart at rung 3+ is approved only under one of these recorded conditions:
|
||||
|
||||
- **No affected sessions:** the control plane has no live session that would be
|
||||
interrupted (verified, not assumed).
|
||||
- **Full drain acknowledged:** every affected session has drained
|
||||
(no open critical section — no held mutation lease mid-write) and the drain is
|
||||
acknowledged in the audit record.
|
||||
- **Controller + gates:** controller approval plus passing automated safety
|
||||
gates (§2.1), the standard v1 path.
|
||||
- **Quorum:** not required in v1; reserved for a future superseding ADR.
|
||||
- **Break-glass:** an incident-backed emergency exception (§2.5).
|
||||
|
||||
Restart **never** bypasses mutation gates mid-critical-section. Drain before
|
||||
restart is mandatory except under break-glass with a declared incident.
|
||||
|
||||
### 2.5 Break-glass
|
||||
|
||||
Break-glass is a **separate, narrower** authorization path for emergencies where
|
||||
the normal drain-and-approve path cannot complete (e.g. the control plane is
|
||||
wedged and cannot drain).
|
||||
|
||||
Break-glass conditions:
|
||||
|
||||
- A declared incident record exists (id, timestamp, declarer) **before** the
|
||||
action.
|
||||
- The action is taken by **operator or admin** authority only — never by an LLM
|
||||
worker role, and never unilaterally by an operator with active peers when a
|
||||
controller is reachable.
|
||||
- The scope is the minimum necessary rung of the ladder.
|
||||
- A **mandatory post-hoc audit** entry is filed: what was restarted, why the
|
||||
normal path was impossible, which sessions were affected, and the incident id.
|
||||
|
||||
Break-glass suspends the drain requirement, not the audit requirement.
|
||||
|
||||
### 2.6 Explicit prohibitions
|
||||
|
||||
- **A unilateral LLM or operator full restart while active peer sessions
|
||||
exist is forbidden.** An LLM worker role must not kill, restart, or relaunch
|
||||
the MCP process; a lone operator must not full-restart over live peer work
|
||||
without controller approval or a break-glass incident.
|
||||
- Process-kill recovery is forbidden as a routine tool (#630). This ADR does not
|
||||
introduce a kill path.
|
||||
- Ambiguous policy state **denies** restart (§4).
|
||||
|
||||
## 3. Security requirements
|
||||
|
||||
- Full restart and host restart are **privileged**; only operator/admin execute
|
||||
them, only after a controller approval or break-glass incident is recorded.
|
||||
- Break-glass is a distinct authorization path with its own audit mandate; it is
|
||||
never the default and never silent.
|
||||
- **Every approval and every restart action is audited** (who approved, who
|
||||
executed, scope, affected sessions, condition, policy version). No restart is
|
||||
authorized without a durable audit entry.
|
||||
|
||||
## 4. Failure behavior
|
||||
|
||||
**Ambiguous policy → deny restart.** If it cannot be established that a
|
||||
restart is authorized under §2 — unknown affected-session state, missing
|
||||
controller approval, absent break-glass incident, or an unclassifiable request —
|
||||
the safe action is to **refuse** the restart and stop with a recovery report,
|
||||
never to restart on assumption.
|
||||
|
||||
## 5. Policy IDs (for enforcement code)
|
||||
|
||||
Enforcement code — the restart coordinator (a later child of #655), the #630
|
||||
contamination guard, and the #642 console restart UX — binds to these stable
|
||||
policy identifiers rather than to prose:
|
||||
|
||||
| Policy ID | Statement |
|
||||
|---|---|
|
||||
| `RG-01` | Restart is last resort; rungs 1–2 must be tried and recorded first (§2.2). |
|
||||
| `RG-02` | v1 authority = controller approval + automated safety gates (§2.1). |
|
||||
| `RG-03` | No LLM worker role performs or authorizes full/host restart (§2.3). |
|
||||
| `RG-04` | Full/host restart executed by operator/admin only, post approval (§2.3). |
|
||||
| `RG-05` | Drain before restart is mandatory except break-glass with incident (§2.4). |
|
||||
| `RG-06` | Break-glass requires a pre-declared incident and post-hoc audit (§2.5). |
|
||||
| `RG-07` | Unilateral LLM/operator full restart with active peers is forbidden (§2.6). |
|
||||
| `RG-08` | Ambiguous policy state denies restart (§4). |
|
||||
|
||||
The `restart-governance/v1` **policy version** field is emitted on future
|
||||
restart audit events so approvals can be reconciled against the policy revision
|
||||
in force.
|
||||
|
||||
## 6. Dogfooding
|
||||
|
||||
Gitea-Tools governs its own MCP control plane by this policy. Author, reviewer,
|
||||
merger, and reconciler sessions operating on this repository use the recovery
|
||||
ladder (§2.2) — reconnect and rebind, never self-restart — and any real restart
|
||||
of the Gitea-Tools stable control runtime follows the controller-approval +
|
||||
drain path defined here.
|
||||
|
||||
## 7. Acceptance and cross-links
|
||||
|
||||
This ADR is the authoritative restart-governance policy. It **must** stay
|
||||
cross-linked from the safety model and the web-console deployment boundary:
|
||||
|
||||
- `docs/safety-model.md` § Process restart governance references this ADR.
|
||||
- `docs/webui-deployment.md` references this ADR for restart/reload disposition.
|
||||
|
||||
It is linked to its issue lineage — umbrella **#655**, vision **#652**, roadmap
|
||||
**#653**, contamination guard **#630**, and console restart UX **#642** — in
|
||||
§ Related above.
|
||||
|
||||
## 8. Non-goals
|
||||
|
||||
- Implementing the restart coordinator or approval state machine (#630, later
|
||||
children of #655).
|
||||
- Implementing HA multi-instance restart or quorum machinery.
|
||||
- Introducing any process-kill or auto-restart tool; existing auto-restart
|
||||
behavior must be inventoried before any new restart tool is enabled.
|
||||
@@ -0,0 +1,83 @@
|
||||
# Incident #670: bare direct-to-master commit `2fa97c26` (retroactive audit)
|
||||
|
||||
Status: verified; disposition recommendation: **accept as-is, no revert** (final
|
||||
disposition owned by controller per issue #670).
|
||||
|
||||
## Summary
|
||||
|
||||
Commit `2fa97c26fbda555a1a83930ca5fdcea9d8e47b50`
|
||||
(`fix(mcp): load dotenv relative to project root`) landed on `prgs/master`
|
||||
as a single-parent commit with no PR wrapper and no review record, bypassing
|
||||
the sanctioned issue → branch → PR → review → merge workflow. It was
|
||||
discovered during the PR #654 post-merge audit. PR #654 itself merged
|
||||
cleanly via the Gitea API and did **not** introduce this commit.
|
||||
|
||||
## Verification evidence (acceptance criteria 1–3)
|
||||
|
||||
- **AC1 — present on `prgs/master`: yes.**
|
||||
`git merge-base --is-ancestor 2fa97c26fbda555a1a83930ca5fdcea9d8e47b50 prgs/master` → true.
|
||||
- **AC2 — no PR or review record: confirmed.**
|
||||
The commit is a single-parent, non-merge commit sitting directly on
|
||||
first-parent master between the #629 merge (`5ab5fe85`) and the #654
|
||||
merge (`ec903b0d`). A PR landing on master produces a merge commit (or a
|
||||
PR-linked head); neither exists here. The controller audit at issue-create
|
||||
time also found no PR wrapper and no review record for this SHA.
|
||||
- **AC3 — changed files and diff summary: confirmed.**
|
||||
`gitea_auth.py | 5 +++--` (+3/−2). Single parent
|
||||
`5ab5fe8583c07134d55dadf09381aecb67df246e`. The change moves
|
||||
`PROJECT_ROOT` derivation above `load_dotenv()` and loads
|
||||
`.env` relative to the project root instead of the process CWD.
|
||||
|
||||
## AC4 — why no immediate revert
|
||||
|
||||
- The dotenv fix is intentional and required for correct runtime behavior:
|
||||
without it, `load_dotenv()` resolves `.env` against the process working
|
||||
directory, which breaks MCP server launches whose CWD is not the project
|
||||
root.
|
||||
- The change is small (+3/−2), self-contained in `gitea_auth.py`, and has
|
||||
been running on master without incident since 2026-07-10.
|
||||
- Reverting would re-introduce a real bug to remove a provenance defect —
|
||||
the wrong trade. Provenance is repaired retroactively by this document,
|
||||
issue #670, and the hardening landed under #671.
|
||||
- If the controller later judges the change unsafe, a separate
|
||||
revert/repair issue is the sanctioned path (issue #670, recommended
|
||||
disposition option 4).
|
||||
|
||||
## AC5 — workflow-hardening linkage
|
||||
|
||||
Prevention already landed: **issue #671** (closed)
|
||||
*“Block direct pushes to stable branches from MCP workflow sessions”*,
|
||||
implemented by commit `5933d87647656643a67a50331c4c7b06ea751dad`
|
||||
(`feat(guard): block direct stable-branch pushes from MCP workflow sessions`).
|
||||
|
||||
Shipped guardrails include:
|
||||
|
||||
- `gitea_record_stable_branch_push_attempt` — classifies proposed commands
|
||||
for direct stable-branch push intent (`git push <remote> master`,
|
||||
refspecs, `HEAD:master`, `--force`, dry-run intent, `:master` delete),
|
||||
plus root/control-checkout local commits not carried by an issue branch,
|
||||
and writes a durable `stable_branch_contamination` marker.
|
||||
- `gitea_audit_stable_branch_contamination` — reconciler-only audit/clear
|
||||
path; a contaminated worker session cannot self-clear.
|
||||
- Review/merge/close/completion mutations fail closed while a
|
||||
contamination marker is active.
|
||||
|
||||
## AC6 — PR #654 was not the source
|
||||
|
||||
- `2fa97c26` is the **first parent** of the #654 merge commit
|
||||
`ec903b0d619e7a27d24aed272a890f4e5d381411`; it predates the #654 merge.
|
||||
- First-parent history `5ab5fe8..ec903b0`:
|
||||
`2fa97c2 fix(mcp): load dotenv relative to project root` followed by
|
||||
`ec903b0 Merge pull request 'feat: lifecycle role/hazard labels ... (#603)' (#654)`.
|
||||
- The #654 merger audit confirmed `ec903b0d` was a valid Gitea-API merge,
|
||||
the `git push prgs master` attempt during that run was a no-op, and the
|
||||
net change `2fa97c2..ec903b0` contained only the reviewed #603
|
||||
lifecycle-label files.
|
||||
- Conclusion: #654 merged reviewed content only; the unauthorized-path
|
||||
defect is solely the earlier bare commit `2fa97c26`.
|
||||
|
||||
## Explicit non-actions (unchanged by this audit)
|
||||
|
||||
- No revert of `2fa97c26`.
|
||||
- No force-push or history rewrite.
|
||||
- No master mutation from the audit session.
|
||||
@@ -0,0 +1,70 @@
|
||||
# MCP Config Drift Diagnostic & Sanctioned Repair Runbook (#672)
|
||||
|
||||
This document describes the diagnostic framework for detecting configuration drift between the active IDE MCP configuration (`~/.gemini/antigravity-ide/mcp_config.json`) and the offline/global canonical configuration (`~/.gemini/config/mcp_config.json`), and establishes the **sanctioned repair runbook**.
|
||||
|
||||
## Background & Problem Statement
|
||||
|
||||
Offline tools like `test_mcp_conn.py` test the global configuration (`~/.gemini/config/mcp_config.json`) via `subprocess.Popen`. However, the active IDE/client namespace uses `~/.gemini/antigravity-ide/mcp_config.json`. When required Gitea role servers (`gitea-author`, `gitea-reviewer`, `gitea-merger`, `gitea-reconciler`, `gitea-controller`, `gitea-tools`) are missing or carry mismatched profile environments in the active IDE config:
|
||||
|
||||
1. Offline tests pass (`test_mcp_conn.py` green).
|
||||
2. The IDE client returns `EOF` / `transport closed` when attempting role-scoped mutations.
|
||||
3. Operators misdiagnose missing server definitions as stale runtimes, leading to forbidden `pkill` attempts (#630) or `mtime` hacks (#655).
|
||||
|
||||
## Diagnostic Tool: `mcp_config_drift.py`
|
||||
|
||||
Run the diagnostic tool directly to compare configurations:
|
||||
|
||||
```bash
|
||||
python3 mcp_config_drift.py --json
|
||||
```
|
||||
|
||||
Or specify custom config locations:
|
||||
|
||||
```bash
|
||||
python3 mcp_config_drift.py \
|
||||
--active-config ~/.gemini/antigravity-ide/mcp_config.json \
|
||||
--global-config ~/.gemini/config/mcp_config.json
|
||||
```
|
||||
|
||||
### Key Diagnostic Outputs
|
||||
|
||||
- `in_sync`: Boolean indicating if all required Gitea role servers exist in the active IDE config with matching profile declarations.
|
||||
- `missing_role_servers`: List of role servers present in global config but missing from active IDE config.
|
||||
- `profile_mismatches`: List of profile environment mismatches per server.
|
||||
- `reasons`: Explicit, human-readable list of drift causes.
|
||||
|
||||
All returned payloads automatically redact secret tokens, DSNs, Authorization headers, and private keys.
|
||||
|
||||
---
|
||||
|
||||
## Sanctioned Repair Path (Step-by-Step)
|
||||
|
||||
When `mcp_config_drift.py` reports drift (`in_sync: false`), execute the following **sanctioned repair steps**:
|
||||
|
||||
1. **Backup Active IDE Config:**
|
||||
```bash
|
||||
cp ~/.gemini/antigravity-ide/mcp_config.json ~/.gemini/antigravity-ide/mcp_config.json.bak
|
||||
```
|
||||
2. **Patch Active IDE Config:**
|
||||
Copy the missing Gitea role server JSON blocks (`gitea-author`, `gitea-reviewer`, etc.) from `~/.gemini/config/mcp_config.json` into `~/.gemini/antigravity-ide/mcp_config.json`.
|
||||
3. **Reconnect via IDE/Client:**
|
||||
Use the IDE / client UI reconnection control (or restart the IDE client app).
|
||||
4. **Verify Active Namespace Health:**
|
||||
Invoke `gitea_whoami` (and optional `gitea_resolve_task_capability`) through the active IDE client on each required role namespace.
|
||||
|
||||
---
|
||||
|
||||
## FORBIDDEN Repair Actions (#630 / #655)
|
||||
|
||||
The following actions are **strictly forbidden** for config drift repair:
|
||||
|
||||
- ❌ **`pkill` or manual daemon process kill commands:** Process kills cause contamination and break active session leases.
|
||||
- ❌ **`mtime` touch edits:** Artificial mtime modifications mask stale runtimes without updating configuration.
|
||||
- ❌ **Source code edits:** Mutating python tool logic to bypass missing server entries.
|
||||
- ❌ **Session-state edits:** Direct database or lock-file state mutation.
|
||||
|
||||
---
|
||||
|
||||
## Final Report Guidelines
|
||||
|
||||
A workflow final report **must not** rely on offline `test_mcp_conn.py` output alone. Final reports must include active-config evidence from live `gitea_whoami` calls on the active IDE namespaces.
|
||||
@@ -0,0 +1,134 @@
|
||||
# Authoritative MCP fleet inventory (#949)
|
||||
|
||||
`gitea_assess_fleet_inventory` is the read-only native capability that answers a
|
||||
question no other surface could: **is exactly one server running for each
|
||||
configured PRGS profile, and do they all belong to one client cohort?**
|
||||
|
||||
## Why the existing surfaces were not enough
|
||||
|
||||
| Surface | What it proves | Why it cannot prove the fleet |
|
||||
| --- | --- | --- |
|
||||
| `gitea_get_runtime_context` | Profile, identity, workspace binding of **the process answering the call** | Says nothing about the other four namespaces |
|
||||
| `gitea_assess_master_parity` | Startup vs current vs live revision of **that same process** | Five self-reports of one revision do not establish five processes, nor the absence of a sixth |
|
||||
| `gitea_assess_mcp_namespace_health` | Whether one named namespace can invoke one tool | Accepts `process`, `probe_result` and `registered_tools` **from the caller** — a capability whose inputs come from the party it constrains is not evidence |
|
||||
| control-plane `sessions` table | Allocator **task** sessions | A session is a unit of work, not a server process; nothing recorded that a server exists |
|
||||
|
||||
Before this capability, satisfying a strict five-process/single-cohort gate
|
||||
required shell process inspection, cached JSON, or source reading — none of
|
||||
which are sanctioned workflow evidence.
|
||||
|
||||
## Evidence model
|
||||
|
||||
Two independent sources must agree before a member counts as running.
|
||||
|
||||
**1. Control-plane runtime registry** — table `mcp_server_runtimes`.
|
||||
Each server writes exactly one row *about itself*, from the official entrypoint,
|
||||
immediately after the native MCP transport bind. Only a transport-bound process
|
||||
reaches that line, and no MCP caller can reach it at all. The row is
|
||||
authoritative for identity: namespace, profile, role, repository binding,
|
||||
cohort, startup revision, transport, and PID.
|
||||
|
||||
**2. Server-side process observation** — a process listing performed by the
|
||||
server answering the inventory call, never by the caller. It is authoritative
|
||||
for existence and liveness, and it is the only source that can reveal a running
|
||||
server the registry does not know about.
|
||||
|
||||
A member is `live` only when a registry row has a matching running process whose
|
||||
start time precedes the registration — so a recycled PID cannot impersonate a
|
||||
server that has since exited.
|
||||
|
||||
### Deliberate non-inferences
|
||||
|
||||
* **Configuration is not existence.** A configured profile with no live
|
||||
corroborated row is `missing`, never `running` (AC8).
|
||||
* **Matching revisions are not a cohort.** `single_cohort` is derived only from
|
||||
recorded cohort identity. Five members at one revision with an unknown cohort
|
||||
yield `single_cohort: null` and a closed gate, never `true` (AC7).
|
||||
* **Parent-client status is not member health.** Each member is classified from
|
||||
its own evidence.
|
||||
|
||||
Cohort identity comes from `GITEA_MCP_CLIENT_COHORT_ID` when the client sets it,
|
||||
otherwise from the parent process that launched the server. A server whose
|
||||
parent has gone away (reparented to init) reports an unknown cohort rather than
|
||||
guessing.
|
||||
|
||||
## Result shape
|
||||
|
||||
Top-level verdict fields:
|
||||
|
||||
| Field | Meaning |
|
||||
| --- | --- |
|
||||
| `inventory_complete` | All evidence was obtainable. False whenever anything below is unknown. |
|
||||
| `incomplete_reasons` | Every distinct reason completeness failed. |
|
||||
| `configured_members` | The five expected members with their instance counts and health. |
|
||||
| `running_members` | Live, corroborated instances of expected profiles. |
|
||||
| `missing_members` | Expected profiles with no live instance. |
|
||||
| `duplicate_members` | Expected profiles with more than one live instance, with every PID. |
|
||||
| `unexpected_members` | Live members outside the expected roster. Never folded into duplicates. |
|
||||
| `stale_members` | Registry rows whose process is dead, unobserved, PID-recycled, or unknown. |
|
||||
| `unregistered_processes` | Running MCP server processes with no registry row. |
|
||||
| `repository_binding_mismatches` | Live members bound to another repository, or with an incomplete binding. |
|
||||
| `role_mismatches` | Live members whose declared role or namespace contradicts the configured profile. |
|
||||
| `single_cohort` / `mixed_cohort` | `true`/`false`, or `null` when cohort evidence is unknown. |
|
||||
| `mixed_revision` / `startup_revisions` | Revision spread across live members. |
|
||||
| `exactly_one_per_profile` | No missing and no duplicate expected members. |
|
||||
| `mutation_gate_satisfied` | The full invariant held. |
|
||||
| `blocked_reason` / `blocked_reasons` | Why the gate is closed; the first is the headline. |
|
||||
| `mutations_performed` | Always `[]`. |
|
||||
|
||||
Ordering is deterministic — every list sorts by namespace, profile, PID, then
|
||||
runtime id — so two callers reading one snapshot see identical structure.
|
||||
|
||||
## How controller and reconciler gates consume it
|
||||
|
||||
Classification is a **pure function of the snapshot**. The answering namespace is
|
||||
reported as `answering_namespace` metadata and never affects the verdict, so
|
||||
`gitea-controller` and `gitea-reconciler` return the same result for the same
|
||||
fleet. Neither is privileged over the other.
|
||||
|
||||
Consume it like this:
|
||||
|
||||
1. Call `gitea_assess_fleet_inventory` from `gitea-controller` **or**
|
||||
`gitea-reconciler` (any namespace holding `gitea.read` may call it).
|
||||
2. If `mutation_gate_satisfied` is `true`, the exact-one-instance-per-profile,
|
||||
single-cohort, single-revision invariant is proven; proceed.
|
||||
3. Otherwise **stop and report `blocked_reason` verbatim**. Distinguish the
|
||||
cases — they need different operator actions:
|
||||
* `missing_members` — the client did not launch that namespace; reconnect it.
|
||||
* `duplicate_members` — a second client or a manual launch is running that
|
||||
profile; the **operator** quits the extra client. This capability never
|
||||
terminates a process.
|
||||
* `unexpected_members` — an unconfigured PRGS server is live; investigate
|
||||
before trusting any gate.
|
||||
* `unregistered_processes` — a server is running code that predates this
|
||||
capability, or failed to register; the inventory is incomplete by
|
||||
construction and must not be reported as healthy.
|
||||
* `mixed_cohort` / `mixed_revision` — the fleet is not one coherent unit.
|
||||
* `single_cohort: null` — cohort evidence is missing; this is *unknown*, not
|
||||
*healthy*.
|
||||
|
||||
Never treat `inventory_complete: false` as a soft warning. It means the
|
||||
inventory cannot describe the whole fleet, which is exactly the state the
|
||||
mutation gate exists to refuse.
|
||||
|
||||
## Read-only guarantees
|
||||
|
||||
The capability performs no restart, reconnect, drain, lease mutation, issue
|
||||
mutation, repository write, or process termination, and it never manufactures
|
||||
historical restart evidence. The only signal it sends is `signal 0`, a
|
||||
liveness/permission check that delivers nothing to the target process. Registry
|
||||
writes happen exclusively on the startup path of the process being described,
|
||||
never on this read path.
|
||||
|
||||
## Scope boundaries
|
||||
|
||||
This capability is observation only. Related concerns live elsewhere and are
|
||||
deliberately not absorbed here:
|
||||
|
||||
* **#950** — controller role metadata and capability-routing consistency.
|
||||
* **#951** — durable synchronization and restart receipts.
|
||||
* **#952** — stale-lease inspection, dashboard, and executor consistency.
|
||||
* **#900** — cohort lifecycle supervision (draining and reaping superseded
|
||||
cohorts) is a *mutating* transition that would consume this evidence.
|
||||
* **#948** — the client/session/generation ownership model that this evidence
|
||||
feeds.
|
||||
@@ -47,18 +47,33 @@ Do the steps in order. Stop as soon as a live **client-namespace** call succeeds
|
||||
- Only the Gitea namespace fails → single-namespace transport close. Continue.
|
||||
- Every server fails → restart the whole MCP client, not just one namespace.
|
||||
|
||||
2. **Reconnect the namespace through the client, not the shell.** Use the IDE /
|
||||
client MCP-reconnect action for that server entry (in Claude Code:
|
||||
`/mcp` → reconnect the affected `gitea-*` server). Reconnecting forces the
|
||||
client to spawn a fresh subprocess and re-open the pipe. This clears the
|
||||
closed-client state that a bare `kill`/respawn from a terminal does **not**.
|
||||
2. **Request the sanctioned reconnect surface (#678), then reconnect through
|
||||
the client — not the shell.** From a still-reachable Gitea MCP namespace
|
||||
(or after host auto-reconnect), call:
|
||||
|
||||
```text
|
||||
gitea_request_mcp_reconnect(
|
||||
namespace="gitea-author", # or gitea-reviewer / gitea-merger / …
|
||||
reason="transport_eof",
|
||||
client="codex", # or claude_code / generic
|
||||
)
|
||||
```
|
||||
|
||||
The tool is **report-only**: it never restarts a process. It returns
|
||||
namespace, profile, pid/session, startup SHA, current master SHA, boundary
|
||||
status, and a **typed blocker** with exact operator UI steps for Codex
|
||||
(Reload Developer Tools / per-server reconnect) or Claude Code (`/mcp`).
|
||||
Then perform the host reconnect those steps describe so the client spawns a
|
||||
fresh subprocess and re-opens the pipe. That clears the closed-client state
|
||||
that a bare `kill`/respawn from a terminal does **not**.
|
||||
|
||||
3. **Do not "fix" it by importing the server or poking the process.** Reaching
|
||||
for `python -c 'import gitea_mcp_server ...'`, raw JSON-RPC from a shell,
|
||||
killing PIDs to force a respawn, or touching MCP config mtimes does **not**
|
||||
restore the *client's* view of the namespace and violates the daemon-import
|
||||
guard (#558, `docs/mcp-daemon-import-guard.md`). The only sanctioned repair
|
||||
is a **client reconnect / relaunch**.
|
||||
is a **client reconnect / relaunch** (or the typed operator path returned by
|
||||
`gitea_request_mcp_reconnect`).
|
||||
|
||||
4. **Verify through the same path the workflow will use.** After reconnect, call
|
||||
the specific tool the blocked workflow needs — not just any tool — through
|
||||
@@ -153,7 +168,19 @@ not a tool argument: a session must never be able to authorize itself.
|
||||
## Related
|
||||
|
||||
- #630 — manual daemon killing as contaminated recovery (this contrast, enforced).
|
||||
- #657 — restart-path inventory and daemon classification.
|
||||
- #686 — manual server launch detection & fail-closed provenance gate.
|
||||
- #531 / #544 — stale-runtime detection (`ps`-based); sibling failure mode.
|
||||
- #558 / `docs/mcp-daemon-import-guard.md` — why shell imports are not a repair.
|
||||
- `docs/mcp-client-registration.md` — per-server registration contract.
|
||||
- `docs/mcp-namespace-health.md` — probe sources and mutation enforcement.
|
||||
|
||||
## Sanctioned reconnect vs forbidden manual launch (#686)
|
||||
|
||||
In addition to manual process killing (#630), manually launching a duplicate role server from an ad hoc shell (`python3 mcp_server.py`) is forbidden and fail-closed:
|
||||
|
||||
- **Why manual launches are unsupported:** A terminal-launched `mcp_server.py` holds its own stdio transport; it can never bind to the IDE client's stdio pipes. It cannot restore a dropped IDE namespace, and a manual duplicate process masks stale client-managed runtimes for that profile, defeating stale-runtime gates.
|
||||
- **Sanctioned path:** Supported recovery is IDE/client-managed reconnect only (`/mcp reconnect`, IDE restart, or sanctioned reconnect exposure).
|
||||
- **Fail-closed enforcement (#686):** Mutating tools on a server lacking client-managed launch provenance (`GITEA_CLIENT_MANAGED=1`) refuse execution fail-closed with typed blocker `unsupported_manual_launch` and an exact next action. Unsupported `GITEA_*` env overrides (e.g. `GITEA_DUMMY`) are surfaced in diagnostics rather than silently ignored.
|
||||
- **Inventory & staleness:** Staleness diagnostics ignore non-client-managed duplicates when evaluating runtime freshness and inventory duplicate processes per profile (#657, #686).
|
||||
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
# MCP scoped recovery playbook (#669)
|
||||
|
||||
**Parent:** [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655)
|
||||
**Vision / roadmap:** [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) · [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||
**Class matrix:** [#663](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/663) · `docs/mcp-restart-classes.md`
|
||||
**Coordinator:** [#658](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/658) · `restart_coordinator.py`
|
||||
**Audit lineage:** [#665](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/665)
|
||||
|
||||
## Decision
|
||||
|
||||
Full-server MCP reset is a **last resort**. Prefer the narrowest recovery that
|
||||
can clear the symptom. The coordinator **refuses** `rolling_mcp_restart`,
|
||||
`full_mcp_restart`, and `host_restart` unless:
|
||||
|
||||
1. The inventory carries a prior **attempt log** of at least one *insufficient*
|
||||
narrower recovery, **or**
|
||||
2. **Break-glass** is authorized
|
||||
(`request_break_glass` + `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`).
|
||||
|
||||
Break-glass still never bypasses the #663 class matrix (role/permission).
|
||||
|
||||
## Ladder (narrow → broad)
|
||||
|
||||
| Rank | Action | Self-service | Implementation / delegation |
|
||||
|---:|---|---|---|
|
||||
| 0 | `client_reconnect` | yes | Host auto-reconnect / client reconnect · [#584](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/584) · `docs/mcp-namespace-eof-recovery.md` |
|
||||
| 1 | `capability_refresh` | yes | `gitea_resolve_task_capability` + `gitea_whoami` · [#610](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/610) · [#685](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/685) |
|
||||
| 2 | `session_reconnect` | yes | Runtime rebind + explicit `worktree_path` · [#543](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/543) · [#618](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/618) |
|
||||
| 3 | `configuration_reload` | no | Class `configuration_reload` · console reload · [#642](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/642) |
|
||||
| 4 | `lease_recovery` | no | Lock/lease recovery paths · [#702](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/702) · [#753](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/753) · [#790](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/790) |
|
||||
| 5 | `worker_restart` | no | Class `worker_restart` · #663 |
|
||||
| 6 | `role_runtime_restart` | no | Class `role_runtime_restart` · console restart · #642/#663 |
|
||||
| 7 | `connector_restart` | no | Class `connector_restart` · #663 |
|
||||
| 8 | `rolling_mcp_restart` | no | Class `rolling_mcp_restart` · design [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668) · **attempt log required** |
|
||||
| 9 | `full_mcp_restart` | no | Class `full_mcp_restart` · **attempt log required** |
|
||||
| 10 | `host_restart` | no | Class `host_restart` · **attempt log required** |
|
||||
|
||||
Machine-readable source of truth: `recovery_playbook.RECOVERY_LADDER` and
|
||||
`recovery_playbook.ladder_document()`.
|
||||
|
||||
## Attempt log shape
|
||||
|
||||
Each prior attempt is a mapping:
|
||||
|
||||
```json
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "transport still closed after IDE reconnect",
|
||||
"actor": "prgs-controller-12345",
|
||||
"recorded_at": "2026-07-25T21:00:00+00:00"
|
||||
}
|
||||
```
|
||||
|
||||
Outcomes that count toward escalation: `failed`, `insufficient`, `denied`,
|
||||
`unresolved`, `timeout`, `error`.
|
||||
|
||||
Pass attempts into the coordinator via inventory
|
||||
`prior_recovery_attempts` or the MCP tool argument
|
||||
`prior_recovery_attempts_json` on `gitea_request_mcp_restart`.
|
||||
|
||||
Helper: `recovery_playbook.build_attempt_record(...)`.
|
||||
|
||||
## Symptom → first rung
|
||||
|
||||
`recovery_playbook.recommend_actions(symptoms=[...])` maps symptoms such as
|
||||
`transport_eof`, `stale_capability`, `stale_lease`, `daemon_corrupt` to the
|
||||
narrowest recommended action, then walks the ladder. Soft recommendations
|
||||
never replace the hard gate on broad restarts.
|
||||
|
||||
## Enforcement points
|
||||
|
||||
1. **`recovery_playbook.assess_escalation`** — pure gate.
|
||||
2. **`restart_coordinator.evaluate_restart_impact`** — when `restart_class` is
|
||||
set (policy-enforced path), broad classes require the gate; report fields
|
||||
`attempt_log_satisfied`, `playbook_escalation`, `break_glass`.
|
||||
3. **`gitea_request_mcp_restart`** — accepts attempt JSON and env-authorized
|
||||
break-glass; never restarts a process.
|
||||
|
||||
## Metrics
|
||||
|
||||
`recovery_playbook.recovery_metrics(attempts)` reports the fraction of
|
||||
successful recoveries that avoided full/host restart
|
||||
(`fraction_avoided_full_restart`).
|
||||
|
||||
## Non-goals
|
||||
|
||||
* HA multi-instance execution (#668 design only here).
|
||||
* Normalizing `pkill` (#630 contamination stays forbidden).
|
||||
* Silent mutation of leases or processes from the playbook itself.
|
||||
|
||||
## Manual process kills
|
||||
|
||||
Remain forbidden and contaminating (#630). The playbook never recommends them.
|
||||
@@ -0,0 +1,53 @@
|
||||
# MCP restart classes and blast-radius permissions (#663)
|
||||
|
||||
This is the machine-enforced class matrix used by
|
||||
`restart_coordinator.RESTART_CLASS_POLICIES`. It implements the narrower-first
|
||||
recovery ladder from #655 and the authorization policy from #656, using the
|
||||
path inventory from #657 and the impact coordinator from #658. Product and
|
||||
delivery lineage: vision #652 and roadmap #653.
|
||||
|
||||
Unknown class names are denied. The coordinator requires both the class
|
||||
permission and an eligible request role. Approval gates are additional: a
|
||||
caller cannot turn a request permission into execution authority.
|
||||
|
||||
| Restart class | Required permission | Expected blast radius | Drain requirement | Approval requirement | Audit requirement | Recovery behavior |
|
||||
|---|---|---|---|---|---|---|
|
||||
| `client_reconnect` | `mcp.reconnect.client` | none | none | self service | class, actor, client namespace, reason, outcome | Reconnect only the caller's client transport. No daemon or peer work changes. |
|
||||
| `session_reconnect` | `mcp.reconnect.session` | low | requesting-session safe point | self service | class, actor, session, reason, outcome | Rebind identity, capability, and workspace state for one session. |
|
||||
| `worker_restart` | `mcp.restart.worker.request` | low | target worker | controller approval + automated gates | class, actor, worker, approval, scoped drain, outcome | Restart one worker after its own leases and mutations drain. |
|
||||
| `role_runtime_restart` | `mcp.restart.role_runtime.request` | medium | target role runtime | controller approval + automated gates | class, actor, role namespace, approval, scoped drain, outcome | Restart and re-probe one role runtime; unrelated roles remain available. |
|
||||
| `connector_restart` | `mcp.restart.connector.request` | medium | target connector | controller approval + automated gates | class, actor, connector, approval, scoped drain, outcome | Restart one connector while unrelated runtimes remain available. |
|
||||
| `configuration_reload` | `mcp.reload.configuration.request` | low | mutation quiesce | controller approval + automated gates | class, actor, configuration revision, approval, outcome | Gracefully reload configuration without replacing the daemon. |
|
||||
| `rolling_mcp_restart` | `mcp.restart.rolling.request` | medium | one instance at a time | controller approval + automated gates | class, actor, instance order, approval, per-instance drains, outcome | Drain, restart, verify, and restore each instance before advancing. |
|
||||
| `full_mcp_restart` | `mcp.restart.full.request` | high | all sessions and mutations | controller approval + automated gates | class, actor, full impact, approval, full drain proof, outcome | Replace the complete MCP runtime only after a verified full drain. |
|
||||
| `host_restart` | `mcp.restart.host.request` | high | all host work | controller approval + infrastructure operator | class, actor, host/change or incident id, approval, full drain proof, outcome | Hand off to infrastructure ownership and reconcile every runtime afterward. |
|
||||
|
||||
## Drain boundary
|
||||
|
||||
Only `full_mcp_restart` and `host_restart` set `full_drain_required=true`.
|
||||
Reconnects and configuration reloads do not disrupt peer sessions. Worker,
|
||||
role-runtime, and connector restarts evaluate only their explicitly named
|
||||
target. Rolling restart drains one instance at a time. Missing required target
|
||||
scope denies the request rather than silently widening it to a full restart.
|
||||
|
||||
## Permission and approval boundary
|
||||
|
||||
Author, reviewer, merger, and reconciler roles may self-request reconnects and
|
||||
request scoped worker/role/connector/reload recovery. They cannot request
|
||||
rolling, full, or host restart classes. Controller/operator/admin roles may
|
||||
request the broader classes, while execution remains operator/admin-owned.
|
||||
Controller approval is independently required for every class above a session
|
||||
reconnect. Host restart additionally requires infrastructure-operator proof.
|
||||
|
||||
The MCP request tool derives class permissions from its authenticated runtime
|
||||
role. It does not accept caller-supplied permissions. Controller and operator
|
||||
authorization are read from the already-running daemon environment, never
|
||||
from a request argument.
|
||||
|
||||
## Audit and failure behavior
|
||||
|
||||
Every impact audit and every console restart/reload audit includes a
|
||||
`restart_class` field. The impact audit also includes the exact
|
||||
`required_permission`. Unknown classes, missing permissions, ineligible roles,
|
||||
missing approval, missing scoped targets, and incomplete inventory all deny
|
||||
fail closed. Manual process kills remain forbidden and contaminating (#630).
|
||||
@@ -0,0 +1,162 @@
|
||||
# MCP restart coordinator and impact analysis (#658)
|
||||
|
||||
Before any sanctioned MCP restart, a central coordinator evaluates the live
|
||||
control-plane state and produces an **impact preview** so operators and the web
|
||||
console (#642 / #652) can see the blast radius *before* concurrent LLM work is
|
||||
disrupted. Uncoordinated restarts destroy in-flight author/reviewer/merger work
|
||||
and give operators no way to see what they are about to break.
|
||||
|
||||
This lands the coordinator + impact DTO + the MCP tool. It is the single
|
||||
sanctioned entry point for restart evaluation post-#657 (which inventoried the
|
||||
restart/reload/kill paths).
|
||||
|
||||
The **drain-proof hard gate now executes inside this tool** (#661, via PR #882):
|
||||
an apply request (`dry_run=False`) is evaluated against a drain proof here and
|
||||
denied when that proof is missing, expired, unclean, tampered with, or stale.
|
||||
It is no longer a separate child operation. What remains a later child is only
|
||||
the **execution** step — actually stopping and restoring a process. This tool
|
||||
still never restarts anything: `apply_supported` is always `false` and
|
||||
`restart_performed` is always `false`.
|
||||
|
||||
The coordinator now routes every request through the restart-class policy
|
||||
matrix defined for #663. See
|
||||
[`mcp-restart-classes.md`](./mcp-restart-classes.md) for permissions, expected
|
||||
blast radius, scoped drain and approval requirements, audit fields, and
|
||||
recovery behavior for all nine classes.
|
||||
|
||||
## Components
|
||||
|
||||
| Piece | Where | Responsibility |
|
||||
|-------|-------|----------------|
|
||||
| `restart_coordinator.evaluate_restart_impact` | `restart_coordinator.py` | Pure classification: inventory → impact report DTO. No I/O, no restart. |
|
||||
| `RestartImpactReport` / `SessionImpact` / `LeaseImpact` | `restart_coordinator.py` | Console-facing DTO (`.as_dict()` is JSON-serializable). |
|
||||
| `ControlPlaneDB.list_sessions` | `control_plane_db.py` | Read-only session inventory (the process-level unit a restart kills). |
|
||||
| `gitea_request_mcp_restart` | `gitea_mcp_server.py` | MCP tool: gathers inventory from the #613 DB, calls the coordinator, returns the report, and on `dry_run=False` runs the #661 drain-proof hard gate. Never restarts a process. |
|
||||
| `drain_proof.gate_apply_restart` | `drain_proof.py` | The #661 hard gate: verifies a drain proof against the current impact fingerprint, or records an authorized break-glass bypass. |
|
||||
|
||||
## Dimensions evaluated
|
||||
|
||||
The coordinator classifies the inventory across the dimensions #658 requires:
|
||||
|
||||
- **Sessions** — every active MCP session; a restart terminates all of them.
|
||||
Liveness = `status == active` **and** the owner pid is alive **and** the
|
||||
heartbeat is fresh (default window 15 min). Dead/stale sessions do not count
|
||||
toward blast radius.
|
||||
- **Leases / locks** — control-plane leases joined with work items and their
|
||||
freshness (`lease_lifecycle.classify_lease_freshness`). Only `active` (live
|
||||
owner) leases are *disruptive*; expired / released / dead-process leases never
|
||||
withhold a restart.
|
||||
- **Issue / PR work** — the issues and PRs behind disruptive leases.
|
||||
- **Mutations / critical sections** — a live lease carrying an author worktree
|
||||
or a mutating phase (`implementing`, `publishing`, `merging`, …) is a
|
||||
critical section a restart must not sever.
|
||||
- **Terminal (merge) lock** — an active terminal lock always makes a restart
|
||||
unsafe.
|
||||
- **Prior recovery attempts** — narrower recovery already tried (e.g. sanctioned
|
||||
client reconnects) is echoed so the operator sees the escalation history.
|
||||
|
||||
## Verdict
|
||||
|
||||
Exactly three verdicts, matching the acceptance criteria:
|
||||
|
||||
| Verdict | `allow_restart` | Meaning |
|
||||
|---------|-----------------|---------|
|
||||
| `safe` | `true` | No other live sessions, no live leases, no terminal lock. |
|
||||
| `unsafe` | `false` | Live work would be disrupted and no operator override is present — **or** the inventory could not be completed (fail closed). |
|
||||
| `override` | `true` | Live work present, but an operator override accepts the blast radius. |
|
||||
|
||||
`override_would_allow` tells the console whether an override path exists for the
|
||||
current state. `blast_radius` is a `none` / `low` / `medium` / `high` severity
|
||||
band derived from the affected session and work counts.
|
||||
|
||||
### Fail closed
|
||||
|
||||
If the control-plane inventory cannot be completed (DB unavailable, a listing
|
||||
failed), `inventory_complete` is `false` and the verdict is `unsafe` / deny. An
|
||||
incomplete evaluation must never green-light a restart.
|
||||
|
||||
### Operator override authority
|
||||
|
||||
Override authority is read from the environment variable
|
||||
`GITEA_OPERATOR_RESTART_OVERRIDE_AUTHORIZATION` and **never** from a tool
|
||||
argument. A worker session cannot set an environment variable on an
|
||||
already-running daemon, so override cannot be self-asserted (same pattern as the
|
||||
#630 daemon-maintenance authorization). The `request_override` tool argument only
|
||||
expresses caller intent; it takes effect solely when the environment
|
||||
authorization is present.
|
||||
|
||||
## The tool
|
||||
|
||||
```text
|
||||
gitea_request_mcp_restart(remote, host, org, repo,
|
||||
dry_run=True, request_override=False,
|
||||
session_id=None, limit=200,
|
||||
restart_class="full_mcp_restart",
|
||||
target_session_id=None, target_role=None,
|
||||
target_connector=None,
|
||||
drain_proof_json=None,
|
||||
request_break_glass=False,
|
||||
prior_recovery_attempts_json=None)
|
||||
```
|
||||
|
||||
It **never restarts anything**: `apply_supported` is always `false` and
|
||||
`restart_performed` is always `false`.
|
||||
|
||||
`prior_recovery_attempts_json` (#669) is an optional JSON array of prior
|
||||
narrow recovery attempts. Rolling / full / host classes require at least one
|
||||
*insufficient* narrower attempt (or authorized break-glass). See
|
||||
`docs/mcp-recovery-playbook.md`.
|
||||
|
||||
### Dry-run versus apply
|
||||
|
||||
| Call | Behavior |
|
||||
|------|----------|
|
||||
| `dry_run=True` (default) | Read-only impact preview. No drain proof is required or consulted. |
|
||||
| `dry_run=False` | The #661 drain-proof hard gate runs **in this tool**. The outcome is reported under `apply_gate` / `apply_authorized`; a denial also returns a durable `incident` descriptor. Still no restart. |
|
||||
|
||||
### Authorization ordering
|
||||
|
||||
An apply requires **both** authorizations, and they are independent:
|
||||
|
||||
1. **Restart-class authorization** (#663 / #669) — the requester's role and
|
||||
permissions must allow the requested class, the class's approval requirement
|
||||
must be satisfied, any target-scoped class must name its target, and broad
|
||||
classes must satisfy the recovery-playbook attempt-log gate. Failing any of
|
||||
these makes `allow_restart` `false`.
|
||||
2. **Drain-proof gate** (#661) — a valid, unexpired, clean proof bound to the
|
||||
current impact fingerprint, or an authorized break-glass.
|
||||
|
||||
`apply_authorized` is the conjunction: `gate.allow and allow_restart`. A clean
|
||||
drain proof therefore cannot override a class or requester-role denial, and a
|
||||
denied class never reports an authorized apply. `apply_gate` carries
|
||||
`drain_gate_allow` and `restart_class_authorized` so a denial is attributable to
|
||||
the authorization that produced it.
|
||||
|
||||
### Break-glass
|
||||
|
||||
Break-glass bypasses the **drain proof only** — never the restart-class matrix
|
||||
(role/permission). Separately, authorized break-glass also satisfies the #669
|
||||
attempt-log requirement for broad restarts (rolling/full/host), because that
|
||||
gate is not a class-matrix permission check.
|
||||
It is honoured solely when `request_break_glass` is set *and* the environment
|
||||
carries `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`; like operator override, the
|
||||
tool argument expresses caller intent and cannot be self-asserted by a worker
|
||||
session. `break_glass_requested` and `break_glass_authorized` are both reported,
|
||||
so a bypass is never silent.
|
||||
|
||||
### Fail closed on apply
|
||||
|
||||
A missing, malformed, expired, unclean, tampered, or fingerprint-stale drain
|
||||
proof denies the apply and returns an `incident` descriptor. An unknown restart
|
||||
class denies before any of this. Ambiguity always denies.
|
||||
|
||||
## Audit
|
||||
|
||||
Every evaluation carries an `audit_record` (event, coordinator version, verdict,
|
||||
restart class, required permission, allow decision, blast radius, counts,
|
||||
timestamp) so restart decisions are
|
||||
auditable. No secrets flow through the coordinator — session ids, pids, and
|
||||
profiles are operational metadata only.
|
||||
|
||||
A representative dry-run report is in
|
||||
[`mcp-restart-impact-sample.json`](./mcp-restart-impact-sample.json).
|
||||
@@ -0,0 +1,148 @@
|
||||
{
|
||||
"coordinator_version": "1.0.0-issue-658",
|
||||
"evaluated_at": "2026-07-24T06:00:00+00:00",
|
||||
"dry_run": true,
|
||||
"restart_performed": false,
|
||||
"inventory_complete": true,
|
||||
"incomplete_reasons": [],
|
||||
"verdict": "unsafe",
|
||||
"allow_restart": false,
|
||||
"override_would_allow": true,
|
||||
"operator_override": false,
|
||||
"blast_radius": "high",
|
||||
"reasons": [
|
||||
"live work would be disrupted; restart denied without operator override",
|
||||
"1 critical section(s) in flight (active lease with a live owner)"
|
||||
],
|
||||
"affected_sessions": [
|
||||
{
|
||||
"session_id": "prgs-author-30988-d6f43c25",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": 1,
|
||||
"status": "active",
|
||||
"alive": true,
|
||||
"heartbeat_stale": false,
|
||||
"is_requester": false,
|
||||
"live": true
|
||||
},
|
||||
{
|
||||
"session_id": "prgs-reviewer-4157-0ce9",
|
||||
"role": "reviewer",
|
||||
"profile": "prgs-reviewer",
|
||||
"pid": 1,
|
||||
"status": "active",
|
||||
"alive": true,
|
||||
"heartbeat_stale": false,
|
||||
"is_requester": true,
|
||||
"live": true
|
||||
}
|
||||
],
|
||||
"affected_leases": [
|
||||
{
|
||||
"lease_id": "lease-abc",
|
||||
"session_id": "prgs-author-30988-d6f43c25",
|
||||
"role": "author",
|
||||
"phase": "implementing",
|
||||
"freshness": "active",
|
||||
"work_kind": "issue",
|
||||
"work_number": 658,
|
||||
"worktree_path": "/repo/branches/feat-issue-658",
|
||||
"disruptive": true,
|
||||
"is_mutation": true,
|
||||
"is_critical_section": true
|
||||
},
|
||||
{
|
||||
"lease_id": "lease-dead",
|
||||
"session_id": "prgs-author-91485",
|
||||
"role": "author",
|
||||
"phase": "allocated",
|
||||
"freshness": "stale_dead_process",
|
||||
"work_kind": "issue",
|
||||
"work_number": 651,
|
||||
"worktree_path": null,
|
||||
"disruptive": false,
|
||||
"is_mutation": false,
|
||||
"is_critical_section": false
|
||||
}
|
||||
],
|
||||
"critical_sections": [
|
||||
{
|
||||
"lease_id": "lease-abc",
|
||||
"session_id": "prgs-author-30988-d6f43c25",
|
||||
"role": "author",
|
||||
"phase": "implementing",
|
||||
"freshness": "active",
|
||||
"work_kind": "issue",
|
||||
"work_number": 658,
|
||||
"worktree_path": "/repo/branches/feat-issue-658",
|
||||
"disruptive": true,
|
||||
"is_mutation": true,
|
||||
"is_critical_section": true
|
||||
}
|
||||
],
|
||||
"affected_issues": [
|
||||
658
|
||||
],
|
||||
"affected_prs": [],
|
||||
"mutations": [
|
||||
{
|
||||
"lease_id": "lease-abc",
|
||||
"session_id": "prgs-author-30988-d6f43c25",
|
||||
"role": "author",
|
||||
"phase": "implementing",
|
||||
"freshness": "active",
|
||||
"work_kind": "issue",
|
||||
"work_number": 658,
|
||||
"worktree_path": "/repo/branches/feat-issue-658",
|
||||
"disruptive": true,
|
||||
"is_mutation": true,
|
||||
"is_critical_section": true
|
||||
}
|
||||
],
|
||||
"terminal_lock": null,
|
||||
"ack_state": {
|
||||
"prgs-author-30988-d6f43c25": "pending"
|
||||
},
|
||||
"prior_recovery_attempts": [
|
||||
{
|
||||
"kind": "client_reconnect",
|
||||
"at": "2026-07-24T06:00:00+00:00",
|
||||
"outcome": "insufficient"
|
||||
}
|
||||
],
|
||||
"counts": {
|
||||
"sessions_total": 2,
|
||||
"sessions_live_other": 1,
|
||||
"leases_total": 2,
|
||||
"leases_disruptive": 1,
|
||||
"critical_sections": 1,
|
||||
"mutations": 1,
|
||||
"affected_issues": 1,
|
||||
"affected_prs": 0,
|
||||
"prior_recovery_attempts": 1
|
||||
},
|
||||
"audit_record": {
|
||||
"event": "restart_impact_evaluated",
|
||||
"coordinator_version": "1.0.0-issue-658",
|
||||
"evaluated_at": "2026-07-24T06:00:00+00:00",
|
||||
"dry_run": true,
|
||||
"operator_override": false,
|
||||
"requesting_session_id": "prgs-reviewer-4157-0ce9",
|
||||
"inventory_complete": true,
|
||||
"verdict": "unsafe",
|
||||
"allow_restart": false,
|
||||
"blast_radius": "high",
|
||||
"counts": {
|
||||
"sessions_total": 2,
|
||||
"sessions_live_other": 1,
|
||||
"leases_total": 2,
|
||||
"leases_disruptive": 1,
|
||||
"critical_sections": 1,
|
||||
"mutations": 1,
|
||||
"affected_issues": 1,
|
||||
"affected_prs": 0,
|
||||
"prior_recovery_attempts": 1
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,91 @@
|
||||
# MCP restart / reload / kill path inventory (#657)
|
||||
|
||||
Complete inventory of every code, script, and host path that can **restart,
|
||||
reload, reconnect, kill, or force-recreate** an MCP process in this project,
|
||||
with each path classified and linked to the guard that constrains it.
|
||||
|
||||
This document is the human-readable companion to the machine-readable registry
|
||||
in [`mcp_restart_paths.py`](../mcp_restart_paths.py). The two are kept in
|
||||
lock-step by [`tests/test_mcp_restart_paths.py`](../tests/test_mcp_restart_paths.py):
|
||||
every `path_id` below must appear in this file, and the source guards are run
|
||||
against the live tree.
|
||||
|
||||
Roadmap linkage: this inventory is the enumeration step of the restart
|
||||
governance work — parent **#655**, restart-governance ADR **#656**, vision
|
||||
**#652**, roadmap **#653**. Related detection/guard work: master-advance
|
||||
staleness **#591**/**#420**, side-effect-free resolver **#685**, transport flap
|
||||
**#584**, manual-kill contamination **#630**.
|
||||
|
||||
## Classifications
|
||||
|
||||
| Classification | Meaning |
|
||||
|---|---|
|
||||
| `sanctioned_narrow_recovery` | One-shot, safe-by-construction recovery that never targets the running daemon. |
|
||||
| `guarded_fail_closed` | Detects a restart-requiring condition, then fails mutations closed and emits reconnect guidance. Never self-restarts. |
|
||||
| `forbidden` | A workflow-safety violation; where an LLM tool could invoke it, it is marked contamination. |
|
||||
| `removed` | A previously-existing unguarded restart primitive that has been deleted; a regression guard keeps it absent. |
|
||||
| `host_residual` | Behavior owned by the host/IDE, outside this process's control. Documented, not code-guarded here. |
|
||||
|
||||
## The rule
|
||||
|
||||
**No component may perform an unguarded full restart of the MCP daemon.** The
|
||||
in-process daemon (`gitea_mcp_server.py`, `mcp_server.py`,
|
||||
`role_session_router.py`) must never replace or terminate its own process:
|
||||
replacing the process after the host has wired up the stdio pipes desyncs the
|
||||
JSON-RPC transport (observed with Antigravity/Cascade hosts). Recovery is owned
|
||||
by the host/operator via a client reconnect — the daemon only ever *detects*
|
||||
and *fails closed*.
|
||||
|
||||
## Inventory
|
||||
|
||||
| path_id | Classification | Mechanism | Guard | Refs |
|
||||
|---|---|---|---|---|
|
||||
| `cli_venv_bootstrap_execv` | sanctioned_narrow_recovery | CLI wrapper scripts re-exec into `venv/bin/python3` via `os.execv`, guarded by `sys.executable != venv_python`. | One-shot pre-import bootstrap; runs before any MCP transport exists and only when not already on the venv interpreter; idempotent guard prevents a re-exec loop. | #657 |
|
||||
| `daemon_self_replacement` | forbidden | The daemon replacing/terminating its own process (`os.execv`/`os.kill`/`os._exit`) to reload code. | Forbidden by design; enforced against the source tree by `assert_no_daemon_self_replacement()`. | #657, #584 |
|
||||
| `legacy_auto_restart_helper` | removed | A helper (`_trigger_mcp_auto_restart`) that actively restarted the server from the read-only resolver path. | Removed in #685; kept absent by `assert_auto_restart_helper_absent()`. | #685, #657 |
|
||||
| `config_touch_reload` | removed | Touching (utime) the MCP client config to make the host reload the server. | Removed from the resolver in #685: stale detection is report-only, never mutating config, spawning threads, or calling `os._exit`. | #685, #657 |
|
||||
| `master_advance_auto_restart` | guarded_fail_closed | On-disk master advancing past the running code. | `master_parity_gate` captures startup parity and blocks mutations while stale, emitting restart guidance; the process never self-restarts. | #420, #591, #657 |
|
||||
| `stale_runtime_resolver_reconnect` | guarded_fail_closed | The capability resolver detecting a stale serving process. | Report-only (#685): returns `restart_required`/`stop_required` and an exact reconnect action; no restart, thread, config touch, or `os._exit`. | #685, #657, #678 |
|
||||
| `codex_client_reconnect_request` | guarded_fail_closed | `gitea_request_mcp_reconnect` report-only tool for Codex/LLM sessions. | Report-only (#678): returns namespace/profile/pid/startup SHA/master SHA/boundary status and a typed operator blocker with exact client UI steps; never restarts or kills. | #678, #630, #685, #657 |
|
||||
| `manual_daemon_kill` | forbidden | Shell kills of the daemon: `pkill -f mcp_server.py`, `killall`, broad `pkill -f python` sweeps, or `kill <pid>` of a daemon pid. | Forbidden (#630): `runtime_recovery_guard` classifies these as contamination and `gitea_record_daemon_process_kill_attempt` writes a durable marker that fails later mutations closed. Operator maintenance authorization is read only from the environment. | #630, #657 |
|
||||
| `conflict_marker_infra_stop` | guarded_fail_closed | The daemon entrypoint scans for unresolved merge-conflict markers at startup and stops (`sys.exit(1)`). | Fail-closed startup stop, not a restart: the process exits and waits for the operator to resolve conflicts and relaunch; never loops. | #657 |
|
||||
| `ide_client_reconnect` | host_residual | A manual `/mcp reconnect` (or equivalent host action) that recreates the MCP client connection. Agents obtain exact UI steps via `gitea_request_mcp_reconnect` (#678). | Outside this process's control; the sanctioned recovery the gates point operators toward. No in-process code initiates it. | #584, #656, #657, #678 |
|
||||
| `profile_switch_runtime` | sanctioned_narrow_recovery | Switching the active execution profile at runtime (dynamic-profile mode). | In-process and restart-free: `runtime_switching_supported` is true, so a switch rebinds capability without recreating the process. | #656, #657 |
|
||||
|
||||
## Guards enforced in CI
|
||||
|
||||
`tests/test_mcp_restart_paths.py` asserts, against the live source tree:
|
||||
|
||||
1. **Registry well-formedness** — every path has a valid classification, a
|
||||
non-empty guard description, references, and locations; ids are unique; all
|
||||
five classifications are represented.
|
||||
2. **Unknown restart attempts fail closed** —
|
||||
`assert_restart_attempt_registered()` raises `UnknownRestartPathError` for
|
||||
any path id not in this inventory, so a novel/unnamed restart primitive
|
||||
cannot slip through silently.
|
||||
3. **Daemon never self-replaces** — `assert_no_daemon_self_replacement()` scans
|
||||
the daemon modules for `os.execv`/`os.kill`/`os._exit`/`os.abort` calls
|
||||
(comment/docstring mentions are ignored) and finds none.
|
||||
4. **Legacy helper stays removed** — `assert_auto_restart_helper_absent()`
|
||||
confirms `_trigger_mcp_auto_restart` has not returned.
|
||||
5. **pkill stays forbidden** — a daemon `pkill` command still classifies as
|
||||
contamination via `runtime_recovery_guard`.
|
||||
|
||||
## Residual host behaviors (outside process control)
|
||||
|
||||
* `/mcp reconnect` in the IDE/host — the sanctioned recovery for stale-runtime,
|
||||
transport-flap (#584), and worktree-binding conditions. The daemon can only
|
||||
emit guidance toward it.
|
||||
* Host-level process management (the operator relaunching the daemon after a
|
||||
fail-closed stop, or after resolving merge conflicts).
|
||||
|
||||
These are documented rather than code-guarded because the process cannot
|
||||
observe or gate them from inside itself.
|
||||
|
||||
## Rollout
|
||||
|
||||
Per #657, guards are introduced flag-free as **regression assertions** (they
|
||||
codify invariants that already hold) before any hard runtime block is layered
|
||||
on. When the restart coordinator (#655/#656) lands, registered paths gain a
|
||||
coordinator token/capability check; unregistered attempts already fail closed
|
||||
today via `assert_restart_attempt_registered()`.
|
||||
@@ -54,6 +54,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_assess_already_landed_reconciliation`
|
||||
- `gitea_assess_conflict_fix_classification`
|
||||
- `gitea_assess_conflict_fix_push`
|
||||
- `gitea_assess_fleet_inventory`
|
||||
- `gitea_assess_gitea_operation_path`
|
||||
- `gitea_assess_master_parity`
|
||||
- `gitea_assess_mcp_namespace_health`
|
||||
@@ -69,6 +70,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_audit_worktree_cleanup`
|
||||
- `gitea_authorize_reconciliation_cleanup_phase`
|
||||
- `gitea_authorize_review_correction`
|
||||
- `gitea_bootstrap_author_issue_worktree`
|
||||
- `gitea_capability_stop_terminal_report`
|
||||
- `gitea_capture_branches_worktree_snapshot`
|
||||
- `gitea_check_pr_eligibility`
|
||||
@@ -100,6 +102,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_get_profile`
|
||||
- `gitea_get_runtime_context`
|
||||
- `gitea_get_shell_health`
|
||||
- `gitea_heartbeat_issue_lock`
|
||||
- `gitea_heartbeat_reviewer_pr_lease`
|
||||
- `gitea_inspect_workflow_lease`
|
||||
- `gitea_issue_irrecoverable_provenance_authorization`
|
||||
@@ -135,6 +138,8 @@ that gates each call, not which tools exist.
|
||||
- `gitea_release_merger_pr_lease`
|
||||
- `gitea_release_reviewer_pr_lease`
|
||||
- `gitea_release_workflow_lease`
|
||||
- `gitea_request_mcp_reconnect`
|
||||
- `gitea_request_mcp_restart`
|
||||
- `gitea_resolve_task_capability`
|
||||
- `gitea_resume_review_draft`
|
||||
- `gitea_review_pr`
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
# Model Usage, Token Cost, Latency, and Workflow Analytics (Phase 4)
|
||||
|
||||
- **Tracking Issue:** [#651](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
|
||||
- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
|
||||
- **Console Surface:** `/analytics`, `/api/v1/analytics`, `/api/v1/analytics/usage`
|
||||
|
||||
## 1. Overview
|
||||
|
||||
The Web Console Analytics module provides durable, aggregate visibility into **model usage, token cost, latency percentiles, and workflow-stage performance** across projects, worker roles, AI models, issues, and PRs.
|
||||
|
||||
### Non-Goals
|
||||
- No mandatory client-side telemetry that leaks prompts or secret keys.
|
||||
- No third-party payment provider or billing integration.
|
||||
- No automatic model routing changes without controller policy (#647).
|
||||
|
||||
---
|
||||
|
||||
## 2. Event Schema (`usage_events`)
|
||||
|
||||
Usage metrics are stored in the control-plane database under table `usage_events`.
|
||||
|
||||
| Column | Type | Description |
|
||||
|---|---|---|
|
||||
| `usage_id` | `INTEGER` | Primary key (autoincrement) |
|
||||
| `session_id` | `TEXT` | Optional active session identifier |
|
||||
| `remote` | `TEXT` | Known Gitea instance (`dadeschools` or `prgs`) |
|
||||
| `org` | `TEXT` | Repository owner / organization |
|
||||
| `repo` | `TEXT` | Repository name |
|
||||
| `project_id` | `TEXT` | Optional project identifier |
|
||||
| `role` | `TEXT` | Active worker role (`author`, `reviewer`, `merger`, `reconciler`, `controller`) |
|
||||
| `model` | `TEXT` | LLM model identifier (e.g. `gemini-3.6-flash`, `claude-3-5-sonnet`) |
|
||||
| `issue_number` | `INTEGER` | Correlated Gitea issue number (optional) |
|
||||
| `pr_number` | `INTEGER` | Correlated Gitea PR number (optional) |
|
||||
| `stage` | `TEXT` | Workflow stage (`preflight`, `implementation`, `review`, `merge`, `reconciliation`) |
|
||||
| `input_tokens` | `INTEGER` | Input token count (optional / nullable) |
|
||||
| `output_tokens` | `INTEGER` | Output token count (optional / nullable) |
|
||||
| `total_tokens` | `INTEGER` | Total token count (optional / nullable) |
|
||||
| `estimated_cost_usd` | `REAL` | Estimated USD cost (optional / nullable) |
|
||||
| `latency_ms` | `INTEGER` | Request latency in milliseconds (optional / nullable) |
|
||||
| `duration_ms` | `INTEGER` | Stage execution duration in milliseconds (optional / nullable) |
|
||||
| `status` | `TEXT` | Outcome status (`success`, `failure`, `timeout`) |
|
||||
| `metadata` | `TEXT` | Redacted metadata or summary string |
|
||||
| `created_at` | `TEXT` | ISO 8601 UTC timestamp |
|
||||
|
||||
---
|
||||
|
||||
## 3. Handling of Missing Data ("Unknown" vs. Zero Fabrication)
|
||||
|
||||
To ensure operational metrics accurately reflect evidence:
|
||||
- **Untracked or missing metrics are displayed as `Unknown`**, never zero-fabricated.
|
||||
- If an event omits `estimated_cost_usd`, `latency_ms`, or token counts, the aggregator marks those fields as missing (`None`) rather than defaulting to `0` or `$0.00`.
|
||||
- Summary tables and KPI cards explicitly indicate when data is unmeasured or partially reported.
|
||||
|
||||
---
|
||||
|
||||
## 4. Redaction & Security Rules
|
||||
|
||||
Per `#633` security policy:
|
||||
- Free-text fields (`metadata`, `prompt_summary`, `session_id`) are run through `console_redaction.redact_text` before persistence and output serialization.
|
||||
- Secret tokens, keychain commands, authorization headers, passwords, and JWTs are stripped automatically.
|
||||
|
||||
---
|
||||
|
||||
## 5. Opt-in Instrumentation Guide
|
||||
|
||||
Applications, MCP servers, and background sessions can report usage metrics through either Python API or HTTP ingestion.
|
||||
|
||||
### Python Ingestion
|
||||
|
||||
```python
|
||||
from webui.analytics_loader import record_usage
|
||||
|
||||
record_usage(
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role="author",
|
||||
model="gemini-3.6-flash",
|
||||
issue_number=651,
|
||||
stage="implementation",
|
||||
input_tokens=1420,
|
||||
output_tokens=380,
|
||||
total_tokens=1800,
|
||||
estimated_cost_usd=0.00045,
|
||||
latency_ms=320,
|
||||
duration_ms=4500,
|
||||
status="success",
|
||||
metadata={"note": "Implementation of analytics module"},
|
||||
)
|
||||
```
|
||||
|
||||
### HTTP Ingestion API (authorized write)
|
||||
|
||||
`POST /api/v1/analytics/usage` is a **gated write**. It runs through
|
||||
`console_authz` action `record_analytics_usage` (operator+, Phase 2 execution).
|
||||
Unauthenticated or phase-inactive requests receive **403** and do not write.
|
||||
Prefer in-process `record_usage` for MCP / session instrumentation.
|
||||
|
||||
```http
|
||||
POST /api/v1/analytics/usage HTTP/1.1
|
||||
Content-Type: application/json
|
||||
# Requires authenticated principal with record_analytics_usage execution enabled
|
||||
|
||||
{
|
||||
"remote": "dadeschools",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"role": "author",
|
||||
"model": "gemini-3.6-flash",
|
||||
"issue_number": 651,
|
||||
"stage": "implementation",
|
||||
"input_tokens": 1420,
|
||||
"output_tokens": 380,
|
||||
"total_tokens": 1800,
|
||||
"estimated_cost_usd": 0.00045,
|
||||
"latency_ms": 320,
|
||||
"duration_ms": 4500,
|
||||
"status": "success",
|
||||
"metadata": "Analytics schema landed"
|
||||
}
|
||||
```
|
||||
|
||||
### Retention
|
||||
|
||||
`usage_events` is retained with hard caps applied on every write:
|
||||
|
||||
| Limit | Default |
|
||||
|---|---|
|
||||
| Max rows | 10,000 (`ControlPlaneDB.USAGE_EVENTS_MAX_ROWS`) |
|
||||
| Max age | 90 days (`ControlPlaneDB.USAGE_EVENTS_RETENTION_DAYS`) |
|
||||
|
||||
Older rows (by `created_at`) and excess oldest rows (by `usage_id`) are deleted
|
||||
after each insert so unbounded growth / DoS-by-volume cannot fill the DB.
|
||||
|
||||
---
|
||||
|
||||
## 6. Querying Analytics API
|
||||
|
||||
```http
|
||||
GET /api/v1/analytics?role=author&stage=implementation HTTP/1.1
|
||||
```
|
||||
|
||||
Returns `AnalyticsSnapshot` JSON containing aggregations (`by_model`, `by_stage`, `by_role`, `by_work_item`, `by_project`) and latency percentiles (`p50`, `p90`, `p95`, `p99`).
|
||||
@@ -0,0 +1,35 @@
|
||||
# Web Console: Sentry/GlitchTip Observability & Incident Bridge Console (#649)
|
||||
|
||||
This document describes the Phase 4 observability console surface integrated into the MCP Control Plane Web Console (`webui/`), backed by the #612 incident bridge and the #613 control-plane DB substrate.
|
||||
|
||||
## Architectural Authority Model (ADR Alignment)
|
||||
|
||||
Per the Web Console Architecture ADR (`docs/architecture/webui-control-plane-console-architecture-adr.md`):
|
||||
|
||||
| Layer | Responsibility | Authority |
|
||||
|---|---|---|
|
||||
| **Gitea** | Durable work record | Issues, PRs, comments, reviews, labels, merges |
|
||||
| **Control-plane DB** | Live coordination & linkage | `incident_links` table, session leases, allocations |
|
||||
| **Sentry / GlitchTip** | Observability input | Unresolved incidents, error events, stack traces |
|
||||
| **Incident Bridge (#612)** | Reconciliation engine | Reconciles provider observations into Gitea issues |
|
||||
| **Web Console (`webui/`)** | Read-only projection & gated actions | Projects connection health & correlation links; gates writes |
|
||||
|
||||
> **Key Rule:** Raw monitoring incidents are **never** assignable control-plane `work_items`. They remain observation input only.
|
||||
|
||||
## Redaction Boundary Invariants
|
||||
|
||||
1. **No secrets in returns or rendering:** Auth tokens (`SENTRY_AUTH_TOKEN`, `GLITCHTIP_AUTH_TOKEN`), DSNs, `Authorization` headers, and sensitive local file paths are passed through `webui.console_redaction` before leaving the server.
|
||||
2. **Safe projection:** Connection objects report `credentials_present: true/false` rather than exposing raw keys or headers.
|
||||
|
||||
## Console Endpoints
|
||||
|
||||
- **HTML Surface:** `GET /observability` — Renders provider connection cards, error correlation tables, and gated reconcile controls.
|
||||
- **Versioned API:** `GET /api/v1/observability` — Returns structured JSON snapshot with `schema_version`, `providers`, `links`, and `metrics`.
|
||||
- **Legacy Compatibility Alias:** `GET /api/observability` — Read-only compatibility alias for Phase 4.
|
||||
|
||||
## Gated Actions
|
||||
|
||||
- `observability_reconcile_incident` (`gitea_observability_reconcile_incident`): Triggers or previews dry-run issue reconciliation for a provider incident.
|
||||
- `observability_link_issue` (`gitea_observability_link_issue`): Links a provider incident to an existing Gitea tracking issue.
|
||||
|
||||
Both actions require `operator` role and gate through `task_capability_map`. Execution fails closed in read-only MVP mode.
|
||||
@@ -0,0 +1,56 @@
|
||||
# Post-restart MCP reconciliation (#662)
|
||||
|
||||
After an MCP process restart, sessions, leases, capabilities, worktrees, and
|
||||
interrupted mutations must be reconciled before operators claim a clean runtime.
|
||||
This document describes the #662 completion-proof path.
|
||||
|
||||
## Components
|
||||
|
||||
| Piece | Where | Responsibility |
|
||||
|-------|-------|----------------|
|
||||
| `post_restart_reconcile.reconcile_after_restart` | `post_restart_reconcile.py` | Pure classification: inventory → completion proof DTO. No I/O. |
|
||||
| `RestartCompletionProof` | `post_restart_reconcile.py` | Machine-readable proof (`.as_dict()` is JSON-serializable). |
|
||||
| `gitea_reconcile_after_restart` | `gitea_mcp_server.py` | MCP tool: gathers inventory from the #613 control-plane DB + master-parity, classifies, returns the proof. Read-only. |
|
||||
| Boot hook | `gitea_assess_master_parity` | First post-restart parity probe also runs reconcile once (log-only by default). |
|
||||
|
||||
## Dimensions
|
||||
|
||||
The assessor classifies:
|
||||
|
||||
- **service_health** — process healthy / parity mutation-safe
|
||||
- **clients** — connected client descriptors (optional inventory)
|
||||
- **sessions** — active session rows with dead owner pids are unresolved
|
||||
- **checkpoints** — soft-depends on #660; skipped with reason when schema absent
|
||||
- **leases** — live control-plane leases after restart
|
||||
- **capabilities** — master-parity / stale-runtime (#610)
|
||||
- **worktrees** — lease-bound paths missing on disk
|
||||
- **interrupted_mutations** — mutating lease phases or explicit pending inventory; **never auto-resumed**
|
||||
- **duplicates** — multiple live claims on the same work item
|
||||
- **queue** — allocator resume safety
|
||||
|
||||
## Modes
|
||||
|
||||
| Mode | Env / arg | Behavior |
|
||||
|------|-----------|----------|
|
||||
| `log_only` (default) | unset or `GITEA_POST_RESTART_RECONCILE_MODE=log_only` | Proof only; `mutation_hold=false` |
|
||||
| `enforce` | `GITEA_POST_RESTART_RECONCILE_MODE=enforce` or `mode=enforce` | Sets `mutation_hold=true` when overall status is degraded/failed or interrupted mutations remain |
|
||||
|
||||
## Follow-up issues
|
||||
|
||||
Unresolved dimensions produce `proposed_follow_ups` entries suitable for durable
|
||||
Gitea issues. The MCP tool **does not create** those issues in v1 (rollout is
|
||||
log-only first). Controllers may file them from the proof payload.
|
||||
|
||||
## Links
|
||||
|
||||
- Umbrella: #655
|
||||
- Vision: #652 · Roadmap: #653
|
||||
- Checkpoint schema: #660 (soft dependency)
|
||||
- Drain proof: #661 (soft)
|
||||
- This issue: #662
|
||||
|
||||
## Non-goals
|
||||
|
||||
- HA multi-instance failover
|
||||
- Automatic silent mutation replay
|
||||
- Implementing the #660 checkpoint schema itself
|
||||
@@ -0,0 +1,230 @@
|
||||
# Remote-MCP coupling inventory
|
||||
|
||||
Every place the Gitea MCP server depends on being a local, client-spawned, stdio-attached
|
||||
process on the operator's machine.
|
||||
|
||||
- **Issue:** #930 (Remote-MCP 01), child 1 of epic #929.
|
||||
- **Generated against commit:** `7bf4f1258451823a55b36d2157e74f8457165088` (`master`).
|
||||
- **Anchors:** every `file:line` below resolves at the commit above and at the commit that
|
||||
adds this document. This change adds one new file and edits no existing file, so no
|
||||
existing line number shifts between the two.
|
||||
- **Scope:** documentation only. No server behavior changes in this child.
|
||||
|
||||
## How to read an entry
|
||||
|
||||
| Field | Meaning |
|
||||
| ----- | ------- |
|
||||
| **Anchor** | `file:line` at the commit under review. |
|
||||
| **Assumes today** | What the code takes for granted while running as a local stdio process. |
|
||||
| **Observes remotely** | What the same code would actually see on a shared remote host. |
|
||||
| **Class** | One of: *portable as written*, *needs a seam*, *needs a replacement*, *cannot be remote*. |
|
||||
| **Owner** | Exactly one epic child (#931–#939) responsible for the fix. |
|
||||
|
||||
Classification meanings:
|
||||
|
||||
- **portable as written** — the code is already transport-, host-, and principal-neutral; it
|
||||
moves unchanged once its inputs are supplied by a remote-aware caller.
|
||||
- **needs a seam** — the logic is correct but is wired to a hard-coded local source. It needs
|
||||
an injection point, not new semantics.
|
||||
- **needs a replacement** — the semantics themselves are local-only. A remote deployment
|
||||
needs a differently-defined mechanism, not the same mechanism relocated.
|
||||
- **cannot be remote** — the operation is inherently about the operator's own machine
|
||||
(its process table, its keychain, its checkout). It must either stay local behind an
|
||||
explicit boundary or be deleted from the remote surface.
|
||||
|
||||
---
|
||||
|
||||
## 1. Transport bind
|
||||
|
||||
The transport is bound literally, once, at process start, and the bound value is the root of
|
||||
the mutation-authorization chain.
|
||||
|
||||
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||
| T1 | `gitea_mcp_server.py:23750` | The single production bind call passes the literal `transport="stdio"` immediately before the server loop. | The literal is wrong for any non-stdio deployment; there is no parameter to change it. | needs a seam | #931 |
|
||||
| T2 | `mcp_daemon_guard.py:45` | `_PRODUCTION_TRANSPORTS = frozenset({"stdio"})` is the closed allowlist of production transports. | A remote transport name is rejected by the allowlist before any other check runs. | needs a seam | #931 |
|
||||
| T3 | `mcp_daemon_guard.py:174` | `bind_native_mcp_transport` raises `UnsanctionedRuntimeError` for any transport outside `_PRODUCTION_TRANSPORTS` (raise at `mcp_daemon_guard.py:187`). | The remote server fails to start rather than degrading; the failure is correct, but the allowlist is the only thing that must change. | needs a seam | #931 |
|
||||
| T4 | `mcp_daemon_guard.py:328` | `is_native_mcp_transport()` asserts a process-local runtime record whose `pid` matches `os.getpid()` and whose phase is `transport_bound`. The predicate itself names no transport. | Unchanged semantics: one server process that bound one transport. It stays true on a remote host. | portable as written | #931 |
|
||||
| T5 | `mcp_daemon_guard.py:349` | `is_production_native_mcp_transport()` adds only a `mode == production` check on top of T4. | Unchanged. | portable as written | #931 |
|
||||
| T6 | `irrecoverable_provenance.py:497` | `assess_transport_for_auth_mint()` requires production native transport before minting non-forgeable recovery authorization (#709 F1). | The gate is transport-agnostic in form, but its guarantee — "an ordinary Python process cannot reach this" — is currently underwritten by the stdio bind. Under a remote transport the guarantee must be re-derived from the authenticated session, not from the bind. | needs a seam | #931 |
|
||||
| T7 | `gitea_mcp_server.py:8375` | Consumer: refuses to proceed unless `assess_transport_for_auth_mint()` allows. | Unchanged given a corrected T6. | portable as written | #931 |
|
||||
| T8 | `gitea_mcp_server.py:8624` | Second consumer of the same gate on the confirmation path. | Unchanged given a corrected T6. | portable as written | #931 |
|
||||
| T9 | `mcp_server.py:4` | Module docstring asserts "Runs over stdio." as a property of the server. | The stated contract becomes false on the remote deployment and is load-bearing documentation for operators. | needs a replacement | #931 |
|
||||
|
||||
## 2. Launch provenance
|
||||
|
||||
Mutations fail closed unless the process can prove a client launched it with real stdio pipes
|
||||
and `GITEA_CLIENT_MANAGED` provenance. Every proof in this section is a statement about the
|
||||
local operating system.
|
||||
|
||||
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||
| P1 | `gitea_mcp_server.py:14588` | `_is_client_managed_process()` derives provenance from `GITEA_CLIENT_MANAGED` / `GITEA_MCP_CLIENT_MANAGED` / `GITEA_SERVER_PROVENANCE` / `GITEA_FORCE_CLIENT_MANAGED` on this process's own environment. | A long-lived remote process has one environment for all callers, so a per-process env var can no longer say anything about the caller that issued a request. | needs a replacement | #934 |
|
||||
| P2 | `gitea_mcp_server.py:14606` | Falls back to `sys.stdin.isatty()`: an active TTY on stdin means a human launched it from a terminal, so refuse. | A remote server has no meaningful stdin. The signal is absent, not merely different. | cannot be remote | #934 |
|
||||
| P3 | `gitea_mcp_server.py:14618` | `_provenance_mutation_block()` emits `blocker_kind: "unsupported_manual_launch"` and a "reconnect the IDE/client-managed MCP namespace" remediation. | The block shape is reusable; its predicate and its remediation text are both stdio-specific. | needs a seam | #934 |
|
||||
| P4 | `gitea_mcp_server.py:20599` | `_check_mcp_runtimes_diagnostics()` shells `ps -o pid,lstart,command -ax` and greps for `mcp_server.py` to find peer role servers. | On a shared host the process table lists unrelated tenants' processes, or none at all under a container. Peer discovery by `ps` has no remote meaning. | cannot be remote | #934 |
|
||||
| P5 | `gitea_mcp_server.py:20702` | More than one process per `GITEA_MCP_PROFILE` in the local process table is reported as a duplicate-launch fault. | A remote endpoint is expected to serve many concurrent sessions per role. "Two processes for one role" becomes the normal case, so the check inverts from a safety net into a false wall. | cannot be remote | #934 |
|
||||
| P6 | `gitea_mcp_server.py:20715` | Processes lacking client-managed provenance are ignored for runtime freshness and reported as manual launches. | Same defect as P5: correctness depends on enumerating local peers. | cannot be remote | #934 |
|
||||
| P7 | `gitea_config.py:1172` | `RECOGNIZED_GITEA_ENV_KEYS` is the allowlist of `GITEA_*` env vars a legitimately launched server may carry; anything else is contamination. | Configuration on a remote host arrives from deployment tooling, not from a client-authored env block. The allowlist keeps working mechanically but stops proving anything about provenance. | needs a replacement | #934 |
|
||||
| P8 | `gitea_mcp_server.py:20683` | The unsupported-env scan applies `RECOGNIZED_GITEA_ENV_KEYS` to *other* processes' environments harvested via `ps eww <pid>`. | Reading another process's environment is unavailable or prohibited across tenants, and is not exposed in this form outside macOS/BSD `ps`. | cannot be remote | #934 |
|
||||
| P9 | `mcp_daemon_guard.py:126` | `mark_sanctioned_daemon()` requires the claiming stack frame's resolved absolute path to be the canonical `mcp_server.py` / `gitea_mcp_server.py` next to the guard module; basename spoofing is rejected. | Entrypoint-path identity still exists on a remote host, but it authenticates the *deployment*, not the *caller*. It must be kept and demoted from "authorizes mutations" to "authorizes the process". | needs a seam | #934 |
|
||||
| P10 | `gitea_config.py:1233` | The client-config generator emits `"GITEA_CLIENT_MANAGED": "1"` into each generated MCP client entry, alongside `GITEA_MCP_CONFIG` / `GITEA_MCP_PROFILE`. | A remote endpoint is addressed by URL and credential, not by a spawn command with an env block. This generator produces the wrong artifact entirely. | needs a replacement | #938 |
|
||||
| P11 | `mcp_namespace_health.py:232` | Namespace health classifies a namespace as `client_managed` or `manual_launch` from the reported env summary. | During dual-run, local and remote namespaces coexist and must both be classifiable; a two-valued local/manual axis cannot express "remote endpoint, authenticated session". | needs a replacement | #939 |
|
||||
| P12 | `gitea_mcp_server.py:18161` | The diagnostics payload reports `server_provenance` as exactly `"client_managed"` or `"manual_launch"`. | This is the field a cutover operator reads to confirm which deployment served a call. It must gain a remote value before dual-run parity can be validated. | needs a replacement | #939 |
|
||||
|
||||
## 3. Role binding
|
||||
|
||||
Role separation is currently enforced by *which process a call reaches*. The process is pinned
|
||||
to one role for its lifetime by an environment variable.
|
||||
|
||||
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||
| R1 | `gitea_config.py:54` | `ENV_PROFILE = "GITEA_MCP_PROFILE"` is the single source of the active profile, read from the process environment. | One shared process serves several principals; a process-wide profile cannot answer "who is calling now". This is the root of the coupling. | needs a replacement | #932 |
|
||||
| R2 | `review_workflow_load.py:95` | Reads `GITEA_MCP_PROFILE` directly to decide the reviewer workflow binding. | Reads the deployment's profile, not the caller's, silently granting or denying the wrong role. | needs a replacement | #932 |
|
||||
| R3 | `mcp_discoverability.py:152` | Reads `GITEA_MCP_PROFILE` to describe the namespace to the client. | Correct logic, wrong input source; it needs the request principal injected. | needs a seam | #932 |
|
||||
| R4 | `webui/deployment_boundary.py:115` | Reads `GITEA_MCP_PROFILE` to classify the deployment boundary for the console. | Same as R3. | needs a seam | #932 |
|
||||
| R5 | `gitea_mcp_server.py:21106` | Remediation text instructs the operator to "Relaunch the server with `GITEA_MCP_PROFILE` set to a profile that has the required permission". | Relaunching a shared remote endpoint to change one caller's role is not a valid instruction; it would re-role every other session. | needs a replacement | #932 |
|
||||
| R6 | `native_mcp_preference.py:93` | Detects shell commands that override `GITEA_MCP_PROFILE` away from the session (`native_mcp_preference.py:223`) and flags them as CLI auth divergence. | The divergence check is genuinely useful and survives, but its notion of "the session's profile" must come from the request principal. | needs a seam | #932 |
|
||||
| R7 | `gitea_mcp_server.py:20671` | Recovers a peer server's role by regexing `GITEA_MCP_PROFILE=` out of that process's environment. | Depends on P4/P8 process-table access; role discovery by peer-env scraping has no remote analogue. | cannot be remote | #932 |
|
||||
|
||||
## 4. Credentials
|
||||
|
||||
Every token resolves, directly or indirectly, from one human's macOS keychain.
|
||||
|
||||
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||
| C1 | `gitea_config.py:956` | `_keychain_token()` shells `security find-generic-password -s <item> -w`. | `security(1)` is a macOS binary reading the calling user's login keychain. It does not exist on a Linux host and would be the wrong identity even on a shared Mac. | cannot be remote | #933 |
|
||||
| C2 | `gitea_config.py:974` | `resolve_token(profile, keychain_lookup=_keychain_token)` dispatches on `auth.type` of `env` or `keychain`, defaulting the lookup to C1. | The injectable `keychain_lookup` parameter is the existing seam; a remote credential provider plugs in here without changing the dispatch. | needs a seam | #933 |
|
||||
| C3 | `gitea_config.py:1015` | `keychain_auth(item_id)` constructs the `{"type": "keychain", "id": ...}` reference stored in profiles. | The reference type itself encodes "macOS keychain" into persisted config. A remote provider needs a new auth reference type, not a new value of this one. | needs a replacement | #933 |
|
||||
| C4 | `mcp_daemon_guard.py:440` | `assert_keychain_access_allowed()` fails closed for git-credential keychain fill outside a sanctioned daemon, with an operator opt-out env var. | The gate protects a mechanism that will not exist remotely. Its replacement must gate the *credential provider* call, not the keychain call, or the protection silently lapses. | needs a replacement | #933 |
|
||||
| C5 | `sentry_incident_bridge.py:190` | `resolve_token(env)` resolves the Sentry token from an injected env mapping with no keychain path. | Already host-neutral; it is the shape the Gitea credential path should converge on. | portable as written | #933 |
|
||||
| C6 | `gitea_mcp_server.py:18469` | The profile-audit tool calls `gitea_config.resolve_token(p)` for every configured profile to report "credentials present" without networking. | On a remote host this would materialize every principal's credential inside one process — an audit surface that becomes a credential-aggregation risk. | needs a seam | #933 |
|
||||
|
||||
## 5. Runtime freshness
|
||||
|
||||
The mutation gate is defined as "the commit this process started at matches the checkout on
|
||||
this disk, and both match live master". Two of those three terms are local-disk facts.
|
||||
|
||||
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||
| F1 | `master_parity_gate.py:168` | `capture_startup_parity(root)` reads git `HEAD` from the server's own root once at startup and returns it as the baseline. | A remote host carries a deployed artifact, not the operator's checkout. Its `HEAD` says nothing about the operator's working tree, which is the thing the gate exists to protect. | cannot be remote | #935 |
|
||||
| F2 | `master_parity_gate.py:255` | `mutation_safe = determinable and in_parity and live_known and not live_stale` — a conjunction of two local-HEAD comparisons and one live-remote comparison. | Two of the three conjuncts lose meaning, so the whole verdict does. A remote deployment needs a redefined, testable freshness predicate rather than this one relocated. | needs a replacement | #935 |
|
||||
| F3 | `master_parity_gate.py:164` | The live-remote head is probed and cached per `(root, remote, branch)`, keyed on the local root. | The live-remote probe is the one conjunct that survives; it needs a key that is not the operator's filesystem path. | needs a seam | #935 |
|
||||
| F4 | `gitea_mcp_server.py:18262` | `gitea_assess_master_parity` publishes `startup_head` / `local_head` / `live_remote_head` / `mutation_safe` as the authoritative mutation-safety verdict. | The tool's contract is consumed by every mutation caller and by the operator; it must keep its shape while its semantics are redefined, or every consumer breaks at once. | needs a replacement | #935 |
|
||||
| F5 | `gitea_mcp_server.py:23054` | Falls back to `_process_boot_head_sha` — the commit this process booted at — when the parity payload has no `startup_head`. | Same defect as F1, in a fallback path that is easy to miss when F1 is fixed. | needs a seam | #935 |
|
||||
| F6 | `gitea_mcp_server.py:20615` | Staleness is also inferred from `os.path.getmtime()` of `gitea_mcp_server.py` under `PROJECT_ROOT` (`gitea_mcp_server.py:20611`), compared against peer process start times. | File mtime on a deployed artifact tracks the deploy, not the operator's edits, and the peer start times it is compared against come from the unavailable process table (P4). | cannot be remote | #935 |
|
||||
|
||||
## 6. Local filesystem
|
||||
|
||||
Author and reviewer tools act directly on the operator's checkout.
|
||||
|
||||
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||
| L1 | `gitea_mcp_server.py:10122` | `gitea_bootstrap_author_issue_worktree` creates and binds a git worktree on the server's own disk. | The remote host has no operator checkout to add a worktree to. Executing this remotely would act on the wrong disk while reporting success. | cannot be remote | #936 |
|
||||
| L2 | `gitea_mcp_server.py:190` | `ACTIVE_WORKTREE_ENV = "GITEA_ACTIVE_WORKTREE"` and `AUTHOR_WORKTREE_ENV` (`gitea_mcp_server.py:191`) carry the active workspace as process-wide environment. | Process-wide workspace state cannot represent per-session workspaces on a shared endpoint. | needs a replacement | #936 |
|
||||
| L3 | `gitea_mcp_server.py:9801` | Binding a worktree writes `os.environ["GITEA_AUTHOR_WORKTREE"]` and `os.environ["GITEA_ACTIVE_WORKTREE"]` (`gitea_mcp_server.py:9802`), mutating global process state. | One session's bind would silently retarget every other concurrent session in the same process. This is a correctness bug the moment concurrency is real. | needs a replacement | #936 |
|
||||
| L4 | `reviewer_inventory_worktree.py:48` | `_BRANCHES_WORKTREE_RE = re.compile(r"\bbranches/", re.I)` requires review worktree paths to sit under `branches/`. | A path convention on the operator's machine, asserted as a validation rule. It needs to become a property of a declared workspace, not a substring test. | needs a seam | #936 |
|
||||
| L5 | `stable_control_runtime.py:54` | `DEV_WORKTREE_SEGMENT = "branches"` classifies a process root as a development worktree by path segment. | Same class of assumption as L4, on the runtime-classification side. | needs a seam | #936 |
|
||||
| L6 | `mcp_server.py:42` | `check_conflict_markers()` runs at import and `os.walk`s the install directory for unresolved conflict markers, `sys.exit(1)` on a hit. | On a remote host it scans a deployed artifact, which by construction never has conflict markers — so the guard passes trivially and stops protecting the thing it was written to protect. | needs a replacement | #936 |
|
||||
| L7 | `role_session_router.py:487` | `check_mid_merge()` reports infra-stop from `.git/MERGE_HEAD`, `rebase-merge`, `rebase-apply` and a source conflict scan under the server's project root. | Same inversion as L6: it would report the deployment's git state, not the operator's. | needs a replacement | #936 |
|
||||
| L8 | `author_issue_bootstrap.py:996` | Enumerates worktrees with `git -C <root> worktree list --porcelain`. | Requires a real local clone with real worktrees; there is nothing equivalent to enumerate remotely. | cannot be remote | #936 |
|
||||
| L9 | `mcp_server.py:10` | Redirects `sys.stderr` to the fixed path `/tmp/mcp_server_stderr.log` outside pytest. | A single fixed `/tmp` path is shared by every concurrent server on a host and is not a deployment's logging surface. | needs a replacement | #938 |
|
||||
| L10 | `gitea_mcp_server.py:2314` | `ISSUE_LOCK_FILE = "/tmp/gitea_issue_lock.json"` — the legacy single global lock slot. | One global `/tmp` slot per host cannot represent concurrent remote sessions and is world-visible on a shared machine. | needs a replacement | #937 |
|
||||
| L11 | `issue_lock_provenance.py:14` | `ISSUE_LOCK_FILE = os.environ.get("GITEA_ISSUE_LOCK_FILE", "/tmp/gitea_issue_lock.json")` keeps the same `/tmp` default in the provenance path. | Same as L10; the env override is a local escape hatch, not a remote design. | needs a replacement | #937 |
|
||||
|
||||
## 7. Durable state
|
||||
|
||||
Locks, leases, session state, and the control-plane database live in the operator's home
|
||||
directory and are keyed on local PIDs.
|
||||
|
||||
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||
| S1 | `issue_lock_store.py:26` | `DEFAULT_LOCK_DIR = ~/.cache/gitea-tools/issue-locks` — per-issue lock files under one user's home. | A shared endpoint has no single operator home; per-user paths make locks invisible across sessions and hosts. | needs a replacement | #937 |
|
||||
| S2 | `issue_lock_store.py:83` | `session_pointer_path()` names the session pointer file `session-<os.getpid()>.json`. | Many sessions share one PID on a remote server, so the pointer collapses to a single slot and sessions overwrite each other. | cannot be remote | #937 |
|
||||
| S3 | `issue_lock_store.py:98` | `is_process_alive(pid)` decides lock liveness by probing the local process table. | A PID recorded by one host is meaningless on another, and may coincidentally match a live unrelated process. | cannot be remote | #937 |
|
||||
| S4 | `issue_lock_store.py:213` | Lock records stamp `session_pid` and `pid` from `os.getpid()`. | The recorded identity no longer distinguishes sessions; ownership checks silently pass for the wrong caller. | needs a replacement | #937 |
|
||||
| S5 | `mcp_session_state.py:27` | `DEFAULT_STATE_DIR = ~/.cache/gitea-tools/session-state`, mode `0o700`. | Same home-directory coupling as S1, for review decision locks and workflow proofs. | needs a replacement | #937 |
|
||||
| S6 | `mcp_session_state.py:559` | Session bodies stamp `session_pid` and `writer_pid` from `os.getpid()` (`mcp_session_state.py:560`). | Writer attribution collapses across concurrent sessions in one process. | needs a replacement | #937 |
|
||||
| S7 | `control_plane_db.py:47` | `DEFAULT_DB_PATH = ~/.cache/gitea-tools/control-plane/control_plane.sqlite3`. | A per-user SQLite file is not reachable by, or safe for, multiple remote sessions or multiple hosts. | needs a replacement | #937 |
|
||||
| S8 | `control_plane_db.py:386` | `sqlite3.connect(self.db_path, timeout=30)` — single-writer file locking tuned for one local process. | SQLite's write lock does not extend across hosts and degrades sharply under real concurrency; the store needs a concurrency-safe backend. | needs a replacement | #937 |
|
||||
| S9 | `control_plane_db.py:1145` | Lease rows record `owner_pid` defaulting to `os.getpid()` (also `control_plane_db.py:2039`). | PID-keyed lease ownership is unusable across hosts and ambiguous within one shared process. | cannot be remote | #937 |
|
||||
| S10 | `mcp_daemon_guard.py:53` | `_DEFAULT_SESSION_STATE_DIR` is pinned once at transport bind so a later `GITEA_MCP_SESSION_STATE_DIR` change cannot manufacture a second authority domain (#695 AC2). | The single-authority-domain invariant is exactly right and must be preserved; only its backing location needs to move. | needs a seam | #937 |
|
||||
| S11 | `gitea_mcp_server.py:11875` | Reviewer-lease reclaim reads `owner_pid_alive` from the lease freshness record to decide whether an owner is dead. | Consumes S3/S9; a false "owner alive" or "owner dead" here reclaims or refuses a live lease. This is the highest-consequence consumer of PID liveness. | cannot be remote | #937 |
|
||||
|
||||
---
|
||||
|
||||
## Summary
|
||||
|
||||
### Entries per category
|
||||
|
||||
| Category | Entries |
|
||||
| -------- | ------: |
|
||||
| 1. Transport bind | 9 |
|
||||
| 2. Launch provenance | 12 |
|
||||
| 3. Role binding | 7 |
|
||||
| 4. Credentials | 6 |
|
||||
| 5. Runtime freshness | 6 |
|
||||
| 6. Local filesystem | 11 |
|
||||
| 7. Durable state | 11 |
|
||||
| **Total** | **62** |
|
||||
|
||||
No category is empty, so no "this category has no coupling" justification is required.
|
||||
|
||||
### Entries per classification
|
||||
|
||||
| Classification | Entries |
|
||||
| -------------- | ------: |
|
||||
| portable as written | 5 |
|
||||
| needs a seam | 16 |
|
||||
| needs a replacement | 26 |
|
||||
| cannot be remote | 15 |
|
||||
| **Total** | **62** |
|
||||
|
||||
### Category × classification
|
||||
|
||||
| Category | portable | seam | replacement | cannot | Total |
|
||||
| -------- | -------: | ---: | ----------: | -----: | ----: |
|
||||
| 1. Transport bind | 4 | 4 | 1 | 0 | 9 |
|
||||
| 2. Launch provenance | 0 | 2 | 5 | 5 | 12 |
|
||||
| 3. Role binding | 0 | 3 | 3 | 1 | 7 |
|
||||
| 4. Credentials | 1 | 2 | 2 | 1 | 6 |
|
||||
| 5. Runtime freshness | 0 | 2 | 2 | 2 | 6 |
|
||||
| 6. Local filesystem | 0 | 2 | 7 | 2 | 11 |
|
||||
| 7. Durable state | 0 | 1 | 6 | 4 | 11 |
|
||||
| **Total** | **5** | **16** | **26** | **15** | **62** |
|
||||
|
||||
### Entries per epic child
|
||||
|
||||
Every child from 2 through 10 is named by at least one entry, and every entry names exactly
|
||||
one child.
|
||||
|
||||
| Child | Issue | Title | Entries | IDs |
|
||||
| ----: | ----- | ----- | ------: | --- |
|
||||
| 2 | #931 | Transport-neutral bind seam | 9 | T1–T9 |
|
||||
| 3 | #932 | Per-request principal resolution | 7 | R1–R7 |
|
||||
| 4 | #933 | Server-side credential provider | 6 | C1–C6 |
|
||||
| 5 | #934 | Remote-session provenance | 9 | P1–P9 |
|
||||
| 6 | #935 | Redefined master-parity gate | 6 | F1–F6 |
|
||||
| 7 | #936 | Local-filesystem vs remotable tool split | 8 | L1–L8 |
|
||||
| 8 | #937 | Concurrency-safe session, lock, and lease state | 13 | L10, L11, S1–S11 |
|
||||
| 9 | #938 | Authenticated remote MCP endpoint | 2 | P10, L9 |
|
||||
| 10 | #939 | Dual-run cutover and rollback | 2 | P11, P12 |
|
||||
| | | **Total** | **62** | |
|
||||
|
||||
## Notes for downstream children
|
||||
|
||||
- **The three highest-risk entries are P5, F2, and S11.** Each is a guard that does not
|
||||
merely stop working remotely — it inverts. P5 turns concurrency into a reported fault,
|
||||
F2 returns a verdict computed from terms that no longer mean anything, and S11 reclaims
|
||||
or refuses leases on a PID-liveness answer that is wrong rather than unknown. A gate that
|
||||
fails open while still reporting green is worse than one that fails to start.
|
||||
- **T4, T5, T7, T8, and C5 are the portable core.** They show the target shape: predicates
|
||||
over injected inputs, with no reference to the host, the process table, or the operator's
|
||||
disk.
|
||||
- **The keychain seam already exists** at C2 (`resolve_token`'s injectable `keychain_lookup`).
|
||||
#933 should widen that seam rather than introduce a parallel path, and must remember C4 —
|
||||
the guard protecting the old mechanism has to be re-pointed, or the protection lapses
|
||||
silently when the mechanism is replaced.
|
||||
- **`branches/` appears as a validation rule in at least two independent places** (L4, L5).
|
||||
Path-substring conventions tend to have more copies than expected; #936 should re-grep
|
||||
rather than trust this list to be exhaustive for that specific pattern.
|
||||
@@ -46,3 +46,17 @@ If shell helpers are unavailable and MCP commit cannot run, stop with a recovery
|
||||
report (restart session, clear hung terminals, use MCP-native commit). See
|
||||
[`llm-workflow-runbooks.md`](llm-workflow-runbooks.md) § MCP-native commit path
|
||||
(#260) and agent temp artifact cleanup (#261).
|
||||
|
||||
## 7. Process restart governance
|
||||
|
||||
Restarting the MCP control-plane process is destructive to concurrent multi-role
|
||||
work and is governed by a dedicated policy. Restart is a **last resort** behind
|
||||
narrower recoveries (reconnect, rebind), full/host restart is reserved to
|
||||
operator/admin under **controller approval + automated safety gates**, a
|
||||
unilateral LLM or operator full restart with active peers is **forbidden**, and
|
||||
ambiguous policy state **denies** restart. Break-glass is a separate,
|
||||
incident-backed path with a mandatory audit.
|
||||
|
||||
See [`architecture/mcp-restart-governance.md`](architecture/mcp-restart-governance.md)
|
||||
(#656) for the authorization matrix, the recovery ladder, break-glass
|
||||
conditions, and the `RG-01`–`RG-08` policy IDs.
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# Sanctioned Recovery Playbooks & Controls (Phase 2 #644)
|
||||
|
||||
## Overview
|
||||
|
||||
Stale runtimes, worktree binding mismatches, and un-reconciled merged branches previously required expert manual shell recovery. Manual process kills (`pkill -f mcp_server.py`) are strictly forbidden and classified as runtime contamination ([#630](sanctioned-restart-controls.md)).
|
||||
|
||||
Phase 2 introduces **sanctioned recovery playbooks and controls** into the Web Console:
|
||||
- **Diagnose**: Surface stale runtimes, worktree binding errors, contamination markers, and worktree anomalies via health & inventory APIs.
|
||||
- **Preview**: Render mutation ledgers and exact confirmation phrases for recovery playbooks.
|
||||
- **Confirm & Apply**: Execute sanctioned recovery actions through gated, audited paths.
|
||||
- **Verify**: Revalidate control-plane state post-recovery before claiming clean status.
|
||||
|
||||
---
|
||||
|
||||
## Recovery Playbook Taxonomy
|
||||
|
||||
| Playbook ID | Action ID | Minimum Role | Target / Scope | Description |
|
||||
|---|---|---|---|---|
|
||||
| `clear_stale_binding` | `system.clear_stale_binding` | Operator | Active worktree binding | Clear provably missing or superseded `GITEA_ACTIVE_WORKTREE` binding ([#702](../stale_binding_recovery.py)). |
|
||||
| `rebind_session_worktree` | `system.rebind_session_worktree` | Operator | Session worktree | Rebind or synchronize session worktree to verified lease worktree ([#864](../dirty_same_claimant_session_rebind.py)). |
|
||||
| `reconcile_cleanups` | `system.reconcile_cleanups` | Controller | Worktree hygiene | Execute reconciler cleanup preview and apply for merged/superseded PR branches. |
|
||||
| `sanctioned_restart` | `system.restart_namespace` | Admin | MCP Namespace | Restart MCP daemon gracefully via host supervisor ([#642](sanctioned-restart-controls.md)). |
|
||||
|
||||
---
|
||||
|
||||
## Wizard Workflow (Diagnose → Preview → Confirm → Verify)
|
||||
|
||||
### 1. Diagnose (`GET /api/v1/system/recovery/diagnose`)
|
||||
Runs control-plane diagnostics:
|
||||
- **Stale Runtime**: Mismatch between running daemon HEAD, local checkout HEAD, and remote-tracking HEAD.
|
||||
- **Worktree Binding**: Missing path (`provably_stale_missing_path`), unverified inherited binding (`unverified_inherited`), or superseded binding (`superseded_by_session_lease`).
|
||||
- **Contamination**: Checks for live contamination markers from unmanaged process kills.
|
||||
- **Worktree Anomalies**: Scans `branches/` directory for un-reconciled cleanups or missing preserved worktrees.
|
||||
|
||||
Returns `RecoveryDiagnosis` with eligible playbooks.
|
||||
|
||||
### 2. Preview (`POST /api/v1/system/recovery/preview`)
|
||||
Takes `playbook_id` and optional `target`/`params`.
|
||||
Returns:
|
||||
- **Mutation Ledger**: Step-by-step sequence of actions.
|
||||
- **Confirmation Phrase**: Exact phrase required to authorize execution (e.g., `confirm clear_stale_binding`).
|
||||
- **Authorization Decision**: RBAC check against the operator's principal.
|
||||
|
||||
### 3. Apply (`POST /api/v1/system/recovery/apply`)
|
||||
Requires `playbook_id` and matching `confirmation` phrase. Gates run in this order, and each fails closed before anything is mutated:
|
||||
|
||||
1. **RBAC and execution phase** (`console_authz.authorize(..., for_execution=True)`). The phase branch only applies when `for_execution` is set. While `ACTIVE_PHASE` is `1`, every phase-2 recovery action is refused with `phase_not_active`, so no recovery playbook writes yet. Preview reports the same decision under `execution_authorization` / `execution_blocked_reason`.
|
||||
2. **Confirmation phrase** (`confirmation_matches`).
|
||||
3. **Contamination rules** ([#630](sanctioned-restart-controls.md)): the live marker is read from the session inventory and assessed under the gated task key `console_recovery_apply`. A contaminated runtime must be cleared through the reconciler cleanup playbook, which is the one playbook exempted from this gate because it is the designated remedy. The marker is also forwarded to `sanctioned_restart.execute_restart`, so a restart cannot launder a contaminated runtime.
|
||||
|
||||
Apply then executes the sanctioned recovery logic against the **live** process environment — not a copy — and records an audit entry in `console_audit`. A playbook that leaves the binding unchanged reports `performed: false`; `binding_before`, `binding_after`, and `binding_changed` are returned so a no-op cannot read as success.
|
||||
|
||||
Apply does **not** enforce master parity. Parity is reported by Diagnose ([#610](../master_parity_gate.py)) as evidence for the operator; it is not a precondition of this endpoint.
|
||||
|
||||
### 4. Verify (`POST /api/v1/system/recovery/verify`)
|
||||
Re-evaluates control-plane diagnostics post-recovery and **reports** `clean`, `stale_runtime_clean`, `binding_clean`, `binding_classification`, and `contamination_clean`. It reports; it does not assert or block. State is read fresh rather than from the mapping a mutation just wrote. An `unverified_inherited` binding is reported as not clean, because unproven is not clean.
|
||||
|
||||
---
|
||||
|
||||
## Safety & Governance Principles
|
||||
|
||||
1. **No Manual `pkill`**: Direct process killing remains forbidden and is recorded as contamination.
|
||||
2. **Auditability**: Every recovery preview and execution is logged in the console audit trail.
|
||||
3. **Master Parity & Dual Control**: High-privilege recovery actions require controller/admin roles and explicit confirmation phrases.
|
||||
@@ -0,0 +1,122 @@
|
||||
# Sanctioned restart and graceful reload controls (#642)
|
||||
|
||||
Sessions used to recover MCP connectivity by killing the host daemon
|
||||
(`pkill -f mcp_server.py`, #630). That is forbidden and stays forbidden: it
|
||||
kills every namespace on the host, contaminates whichever session survives, and
|
||||
leaves no audit trail. This document describes the sanctioned replacement,
|
||||
implemented in `webui/sanctioned_restart.py`.
|
||||
|
||||
## What the console will and will not do
|
||||
|
||||
The console **never** restarts anything. It authorizes an intent, records it,
|
||||
and hands off to a host supervisor. There is no code path in which the console
|
||||
sends a signal, spawns a process, or renders a kill command — a regression test
|
||||
asserts the module contains no `subprocess`, `signal`, `os.kill`, `os.system`,
|
||||
or `popen` reference, and that no returned payload contains a kill command.
|
||||
|
||||
## Operations
|
||||
|
||||
| Mode | Action | Minimum role | Behaviour |
|
||||
|------|--------|--------------|-----------|
|
||||
| `reload` | `system.reload_namespace` | controller | Host supervisor reloads the namespace in place, draining in-flight requests. |
|
||||
| `restart` | `system.restart_namespace` | admin | Host supervisor restarts the namespace. In-flight requests are lost. |
|
||||
|
||||
Scope is always exactly one namespace. A fleet-wide restart is an explicit
|
||||
non-goal: `all`, `*`, `fleet`, and an empty scope are refused with
|
||||
`fleet_scope_not_permitted`, because that is precisely the blast radius the
|
||||
forbidden kill already had. An unrecognised namespace is refused rather than
|
||||
passed through to the host.
|
||||
|
||||
## The gate sequence
|
||||
|
||||
`assess_restart_request()` applies every gate in order and reports the first
|
||||
failure with a stable reason code:
|
||||
|
||||
| Order | Gate | Reason code on failure |
|
||||
|-------|------|------------------------|
|
||||
| 1 | Mode is `restart` or `reload` | `unknown_mode` |
|
||||
| 2 | Scope is a single known namespace | `fleet_scope_not_permitted`, `unknown_namespace` |
|
||||
| 3 | Principal holds the required console role | `unauthorized` |
|
||||
| 4 | Confirmation phrase supplied | `confirmation_required` |
|
||||
| 5 | Confirmation names this namespace and mode | `confirmation_mismatch` |
|
||||
| 6 | Out-of-band operator authorization present | `operator_authorization_missing` |
|
||||
| 7 | Runtime is not contaminated | `contaminated_runtime` |
|
||||
| 8 | Host restart hook configured | `restart_hook_not_configured` |
|
||||
|
||||
Passing every gate yields `host_action_required`, never "restarted".
|
||||
|
||||
### Confirmation binds the namespace
|
||||
|
||||
The required phrase is `"<mode> <namespace>"` — for example
|
||||
`restart gitea-author`. Binding the namespace into the phrase is the point: a
|
||||
confirmation typed for one namespace cannot be replayed against another.
|
||||
|
||||
### Operator authorization is not self-assertable
|
||||
|
||||
Host daemon maintenance is authorized out of band through
|
||||
`GITEA_OPERATOR_DAEMON_MAINTENANCE_AUTHORIZATION`, read from the process
|
||||
environment and nowhere else (#630; #710 finding F1). A worker session cannot
|
||||
set an environment variable for an already-running daemon, so this cannot be
|
||||
faked the way a tool argument could.
|
||||
|
||||
### The host hook
|
||||
|
||||
`GITEA_SANCTIONED_RESTART_HOOK` holds an opaque reference the *host* resolves —
|
||||
a supervisor label such as a launchd job name, never a command line. With no
|
||||
hook configured the request is refused; the console does not fall back to a
|
||||
process kill. The value is read server-side and never rendered to a client.
|
||||
|
||||
## Manual kill remains contamination
|
||||
|
||||
`classify_restart_command()` classifies an operator-proposed recovery command.
|
||||
A manual `pkill`/`kill`/`killall` of the MCP daemon is contamination, not a
|
||||
restart: it returns `clean_claim_allowed: false` and builds a durable
|
||||
contamination marker (redacted command only, never secrets) naming
|
||||
`system.restart_namespace` as the sanctioned alternative.
|
||||
|
||||
A live, uncleared contamination marker also blocks a restart. This is stricter
|
||||
than #630's task-scoped gate, which deliberately lets a contaminated worker keep
|
||||
commenting and handing off: restarting a contaminated runtime would launder the
|
||||
contamination rather than resolve it. Clear the marker through the reconciler
|
||||
path first.
|
||||
|
||||
## Post-restart health verification
|
||||
|
||||
After the host supervisor acts, `verify_post_restart_health()` decides whether
|
||||
the session may claim to be clean:
|
||||
|
||||
| Status | Meaning | Clean claim |
|
||||
|--------|---------|-------------|
|
||||
| `clean` | Required tool callable, proven through the live client namespace | Allowed |
|
||||
| `unproven` | Reported healthy without live client-namespace evidence | Refused |
|
||||
| `unhealthy` | Probe failed | Refused |
|
||||
|
||||
Only `probe_source=client_namespace` evidence clears a session. Static tool
|
||||
registration is not proof, and neither is an offline subprocess probe — an IDE
|
||||
client can hold a registered tool list while live calls fail with
|
||||
`client is closing: EOF` (see
|
||||
[`mcp-namespace-health.md`](mcp-namespace-health.md)).
|
||||
|
||||
## Audit
|
||||
|
||||
Every attempt — allowed or denied — is recorded through
|
||||
`webui.console_audit` with actor, target namespace, mode, result, and reason
|
||||
code, and is redacted before it is persisted. `system.restart_namespace` is
|
||||
break-glass, so its records are retained for 730 days. Records carry
|
||||
`process_kill_executed: false`, which is a fact about the code path rather than
|
||||
a claim: no such path exists.
|
||||
|
||||
## Environment variables
|
||||
|
||||
| Variable | Purpose |
|
||||
|----------|---------|
|
||||
| `GITEA_SANCTIONED_RESTART_HOOK` | Host supervisor reference; absent means restart is refused. |
|
||||
| `GITEA_OPERATOR_DAEMON_MAINTENANCE_AUTHORIZATION` | Out-of-band operator authorization reference. |
|
||||
| `WEBUI_AUDIT_LOG` | Console audit sink; absent means records are built but not persisted. |
|
||||
|
||||
## Non-goals
|
||||
|
||||
* No unrestricted `kill` from the UI, in any role, in any phase.
|
||||
* No fleet-wide restart.
|
||||
* No silent auto-restart loop: every attempt is confirmed and audited.
|
||||
* This does not implement the Phase 1 health API (#634).
|
||||
+58
-11
@@ -91,6 +91,15 @@ already define, and a regression test asserts each mapping matches.
|
||||
| `close_pr` | controller | privileged | `gitea.pr.close` | Yes | No | No | 3 |
|
||||
| `merge_pr` | controller | privileged | `gitea.pr.merge` | Yes | **Yes** | **Yes** | 3 |
|
||||
| `delete_branch` | admin | destructive | `gitea.branch.delete` | Yes | **Yes** | **Yes** | 3 |
|
||||
| `record_analytics_usage` | operator | gated_write | `runtime.record_analytics_usage` | Yes | No | No | 2 |
|
||||
| `system.reload_namespace` | controller | privileged | `runtime.reload_namespace` | Yes | No | No | 2 |
|
||||
| `system.restart_namespace` | admin | destructive | `runtime.restart_namespace` | Yes | **Yes** | **Yes** | 2 |
|
||||
| `system.clear_stale_binding` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||
| `system.rebind_session_worktree` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||
| `system.reconcile_cleanups` | controller | privileged | `gitea.pr.close` | Yes | No | No | 2 |
|
||||
| `initiate_workflow` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||
| `observability_reconcile_incident` | operator | gated_write | `gitea.read` | Yes | No | No | 4 |
|
||||
| `observability_link_issue` | operator | gated_write | `gitea.read` | Yes | No | No | 4 |
|
||||
|
||||
**Dual control** means the acting principal may not be the sole authority: a
|
||||
second distinct principal must confirm. **Break-glass** means the action is
|
||||
@@ -102,6 +111,19 @@ honouring it.
|
||||
`delete_branch` is admin-only rather than controller because it is the one
|
||||
irreversible action in the set.
|
||||
|
||||
`system.restart_namespace` is admin-only for the same reason: restarting a
|
||||
namespace drops every in-flight request on it. `system.reload_namespace` drains
|
||||
first, so it is privileged but not destructive. Neither action is ever executed
|
||||
by the console — both hand off to a host supervisor, and neither exposes a raw
|
||||
process kill. See
|
||||
[`sanctioned-restart-controls.md`](sanctioned-restart-controls.md) (#642).
|
||||
|
||||
`initiate_workflow` (#643) is operator-class because its outcome is a *claim*,
|
||||
not a Gitea verdict. Requesting reviewer or merger work reserves that work
|
||||
through the allocator; it does not grant the right to approve or merge, which
|
||||
stays with the MCP role profile and its own capability gates. See
|
||||
[`webui-requests.md`](webui-requests.md).
|
||||
|
||||
### Authorization decision
|
||||
|
||||
`authorize(action_id, principal, for_execution=False)` returns a decision
|
||||
@@ -116,9 +138,24 @@ record and **denies by default**. The deny reasons are closed and enumerated:
|
||||
| `phase_not_active` | Execution requested for an action whose phase is not open. |
|
||||
| `allowed_preview_only` | Authorized — preview only, execution still disabled. |
|
||||
|
||||
There is no implicit allow branch. Even the allow result reports
|
||||
`execution_enabled: false` while the console is in Phase 1, so no caller can
|
||||
read an allow as permission to mutate.
|
||||
There is no implicit allow branch.
|
||||
|
||||
`execution_enabled` on the decision reports whether the action has a live
|
||||
execution path at all, and is computed by `execution_wired(action)`. There are
|
||||
exactly two ways to be wired:
|
||||
|
||||
1. the action's `phase` is at or below `ACTIVE_PHASE`; or
|
||||
2. the action declares an `execution_env_flag` **and** that variable is set.
|
||||
|
||||
Every action that declares no flag therefore reports `execution_enabled: false`
|
||||
while the console is in Phase 1, so no caller can read an allow as permission
|
||||
to mutate. The per-action flag exists because raising `ACTIVE_PHASE` would
|
||||
enable execution for every action of that phase at once, including ones whose
|
||||
execution path is not implemented. One implemented action goes live on its own
|
||||
flag instead of dragging its unimplemented phase-mates with it.
|
||||
|
||||
`initiate_workflow` is the only action that currently declares a flag
|
||||
(`WEBUI_REQUESTS_EXECUTION`), and it stays denied until an operator sets it.
|
||||
|
||||
## Secret redaction
|
||||
|
||||
@@ -199,8 +236,8 @@ breaking the request it describes.
|
||||
| Class | Applies to | Default |
|
||||
|-------|-----------|---------|
|
||||
| `standard` | Routine gated writes | 90 days |
|
||||
| `privileged` | `review_pr`, `close_pr`, and any unclassifiable action | 365 days |
|
||||
| `break_glass` | `merge_pr`, `delete_branch` | 730 days |
|
||||
| `privileged` | `review_pr`, `close_pr`, `system.reload_namespace`, and any unclassifiable action | 365 days |
|
||||
| `break_glass` | `merge_pr`, `delete_branch`, `system.restart_namespace` | 730 days |
|
||||
|
||||
Each record carries its own class, day count, and computed `expires_at`, so
|
||||
retention is auditable per record rather than inferred from file age. An
|
||||
@@ -225,13 +262,22 @@ second one. The integration points are already wired and observable:
|
||||
instead of adding a parallel check.
|
||||
- **`GET /api/console/security-model`** publishes the RBAC matrix, redaction
|
||||
policy, and audit policy as JSON for operators and tests.
|
||||
- **`POST /api/v1/requests/preview` and `.../apply`** (#643) are the first
|
||||
actions to use this model for a real execution path. Preview always returns a
|
||||
decision and an audited `previewed` record; apply requires `confirm=true`,
|
||||
emits `succeeded` or `denied`, and reserves work only through the allocator.
|
||||
See [`webui-requests.md`](webui-requests.md).
|
||||
|
||||
To open Phase 2, a child issue must: raise `ACTIVE_PHASE`, implement the
|
||||
confirmation and dual-control flow the matrix already declares, emit a
|
||||
`succeeded` or `failed` record alongside the `gitea_audit` mutation record, and
|
||||
keep `viewer` unable to reach any of it. Turning on execution without the
|
||||
confirmation flow contradicts a declared requirement and is a review failure,
|
||||
not a shortcut.
|
||||
A Phase 2 action must: use `execution_wired` rather than a private enable flag,
|
||||
implement the confirmation and dual-control flow the matrix already declares,
|
||||
emit a `succeeded` or `failed` record alongside the `gitea_audit` mutation
|
||||
record, and keep `viewer` unable to reach any of it. Turning on execution
|
||||
without the confirmation flow contradicts a declared requirement and is a
|
||||
review failure, not a shortcut.
|
||||
|
||||
Raising `ACTIVE_PHASE` remains the way to open a whole phase at once, and is
|
||||
deliberately *not* what #643 did: an action-scoped opt-in cannot enable an
|
||||
action whose execution path nobody wrote.
|
||||
|
||||
## Local-dev mode
|
||||
|
||||
@@ -284,6 +330,7 @@ Until Phase 2 wires it, probe protection rests on network placement alone, as
|
||||
| `WEBUI_ROLE_MAP` | unset | JSON subject → role map |
|
||||
| `WEBUI_REQUIRE_PROBE_AUTH` | unset | Require auth for non-public probes |
|
||||
| `WEBUI_CONSOLE_AUDIT_LOG` | unset | Append-only audit sink path |
|
||||
| `WEBUI_REQUESTS_EXECUTION` | unset | Opt in to `initiate_workflow` execution (#643) |
|
||||
|
||||
All are read server-side only. None is ever rendered into a page or returned by
|
||||
an API.
|
||||
|
||||
@@ -55,6 +55,15 @@ shipped to the browser.
|
||||
assumption paths, and the client-secret policy. Use it to verify an instance is
|
||||
configured for internal-only operation.
|
||||
|
||||
## Process restart / reload disposition
|
||||
|
||||
The console never exposes a restart or reload control; process restart of the
|
||||
MCP control-plane runtime is governed separately. Restart is a last resort behind
|
||||
reconnect/rebind, full restart is operator/admin-only under controller approval
|
||||
plus safety gates, and break-glass is an incident-backed path. See
|
||||
[`architecture/mcp-restart-governance.md`](architecture/mcp-restart-governance.md)
|
||||
(#656).
|
||||
|
||||
## Non-goals (MVP)
|
||||
|
||||
- Full SSO or session login in the UI
|
||||
|
||||
+419
-1
@@ -52,9 +52,13 @@ status, onboarding checklist state, and the fail-closed error payloads (#635).
|
||||
| Path | Description |
|
||||
|------|-------------|
|
||||
| `/` | Home / operator overview |
|
||||
| `/health` | JSON liveness (`status`, `service`, `mode`, `timestamp`) |
|
||||
| `/health` | JSON liveness (`status`, `service`, `mode`, `timestamp`, `uptime_seconds`) |
|
||||
| `/api/v1/system/health` | Structured read-only system health (#634) |
|
||||
| `/system-health` | System-health dashboard — readiness, version/uptime, dependencies, MCP namespaces, stale-runtime parity (#639) |
|
||||
| `/queue` | Live PR and issue queue dashboard (#429) |
|
||||
| `/api/queue` | JSON queue export with pagination metadata |
|
||||
| `/traffic` | Workflow traffic-control view — runnable, leased, blocked, needs-controller, terminal-complete (#640) |
|
||||
| `/api/traffic` | JSON traffic-control export with state classifications and next safe role actions |
|
||||
| `/projects` | Project registry list with status and onboarding progress (#427, #635) |
|
||||
| `/projects/{id}` | Project detail + onboarding checklist |
|
||||
| `/api/v1/projects` | Versioned JSON registry export (#635) |
|
||||
@@ -73,11 +77,128 @@ status, onboarding checklist state, and the fail-closed error payloads (#635).
|
||||
| `/api/actions/{id}/preview` | Mutation ledger preview (GET, read-only) |
|
||||
| `/leases` | Lease and collision visibility (#433) |
|
||||
| `/api/leases` | JSON lease/collision export |
|
||||
| `/sessions` | Runtime and session view (#641) — health + inventory sessions/namespaces/worktrees |
|
||||
| `/api/sessions` | JSON export for the runtime/session view |
|
||||
| `/api/v1/sessions` | Versioned alias of `/api/sessions` |
|
||||
| `/gitea` | Gitea issue↔PR linkage console (#645) — both directions, with the evidence for each edge |
|
||||
| `/api/v1/gitea/linkage` | JSON linkage export; `502` when the read could not be answered |
|
||||
| `/inventory` | Phase 1 shell stub — unified inventory (backed by #636) |
|
||||
| `/timeline` | Phase 1 shell stub — workflow event timeline |
|
||||
| `/policy` | Phase 1 shell stub — capability/role policy placeholder |
|
||||
| `/insights` | Phase 1 shell stub — operational insights placeholder |
|
||||
|
||||
Most routes are GET-only. POST/PUT/PATCH/DELETE return `405` with
|
||||
`read-only-mvp`, except `/audit` and `/api/audit` which accept POST for
|
||||
local validator preview only (no Gitea mutations, no server-side storage).
|
||||
|
||||
### Traffic-control state vocabulary (#640)
|
||||
|
||||
The traffic view classifies each open issue/PR into exactly one bucket:
|
||||
|
||||
| Bucket | Meaning | Operator implication |
|
||||
|--------|---------|----------------------|
|
||||
| **runnable** | No active lease, no block reason, safe for its expected role | Next role may start work |
|
||||
| **leased** | Active author claim or reviewer PR lease | Do not stomp; wait or adopt via role tools |
|
||||
| **blocked** | Dependency, missing head pin, conflict, or unmet dependency | Author remediation first |
|
||||
| **needs_controller** | Contaminated, controller-only diagnosis, or `status:blocked` | Controller only |
|
||||
| **terminal_complete** | Reconciler / terminal-lock territory | Reconciler cleanup path |
|
||||
|
||||
`status:blocked` items route to **needs_controller**, not **blocked**:
|
||||
`expected_role_for_candidate` sends them to the controller, and the blocker
|
||||
reason renders in either bucket.
|
||||
|
||||
**Live path contracts (do not invent):**
|
||||
|
||||
- PR head pins come from `QueueItem.signals["head_sha"]` (full SHA). Display
|
||||
`extra["head_sha"]` is truncated and must never be used for routing.
|
||||
- Reviewer leases are keyed as `(pr, pr_number)` only — never via a linked
|
||||
`issue_number` on the same lease marker.
|
||||
- Issue claims come from `claim_inventory["entries"]`
|
||||
(`issue_claim_heartbeat.build_claim_inventory`). There is no `active_claims`
|
||||
key.
|
||||
- Queue display badges are only: `blocked`, `claimed`, `duplicate`, `stale`,
|
||||
`in-review`, `open`. Review verdicts (`request-changes`, `approved`) are
|
||||
**not** queue badges; traffic does not invent them from the queue loader.
|
||||
|
||||
## System health API (#634)
|
||||
|
||||
`GET /api/v1/system/health` is the structured, read-only health surface for
|
||||
automated readiness checks. It is the first console API under the `/api/v1`
|
||||
prefix; the unversioned MVP exports remain as compatibility aliases.
|
||||
|
||||
`/health` is unchanged for existing consumers — every MVP key is still present
|
||||
— and now also carries `started_at`, `uptime_seconds`, and a
|
||||
`system_health_api` pointer. It stays deliberately cheap and runs no dependency
|
||||
probe, because answering readiness costs real work.
|
||||
|
||||
**Status codes.** `200` when ready, `503` when a required dependency failed or
|
||||
was never probed. Automation can branch on the code without parsing the body.
|
||||
|
||||
**Query flags.** The Gitea check is a network call, so it is opt-in:
|
||||
`GET /api/v1/system/health?deep=1` runs it and caches the result for
|
||||
`WEBUI_HEALTH_PROBE_TTL_SECONDS` (default 15s) so dashboard polling does not
|
||||
amplify into remote load. Without the flag that probe reports `skipped`.
|
||||
|
||||
**Dependencies.** `control_plane_db` and `repository` are required and drive
|
||||
readiness. `gitea` is optional: when it fails the overall `status` degrades but
|
||||
`readiness.ready` stays true, because local inventory is still serveable. Each
|
||||
entry carries `status`, `detail`, `required`, and `latency_ms`.
|
||||
|
||||
Two honesty rules are worth knowing before reading the payload:
|
||||
|
||||
* `stale_runtime.mutation_safe` is true only when the runtime, checkout, and
|
||||
remote-tracking commits are all known and equal. An unfetched remote is
|
||||
reported as indeterminate, never as safe.
|
||||
* `mcp_namespaces` entries are always `unproven`. A web process runs outside
|
||||
the IDE-managed MCP client and cannot invoke a namespace tool, so per #543
|
||||
only a `client_namespace` probe can prove that path.
|
||||
|
||||
Sample response (abridged, healthy):
|
||||
|
||||
```json
|
||||
{
|
||||
"status": "ok",
|
||||
"service": "mcp-control-plane-webui",
|
||||
"mode": "read-only",
|
||||
"api": "/api/v1/system/health",
|
||||
"timestamp": "2026-07-22T11:04:18.512034+00:00",
|
||||
"readiness": { "ready": true, "complete": true, "reasons": [] },
|
||||
"version": {
|
||||
"git_sha": "620ed6e9a9550b8da2ceb82d9ab8744e8920490f",
|
||||
"git_describe": "v1.1.0-898-g620ed6e",
|
||||
"control_plane_schema_version": 4,
|
||||
"python_version": "3.14.5",
|
||||
"known": true
|
||||
},
|
||||
"process": { "started_at": "2026-07-22T10:58:02.114+00:00", "uptime_seconds": 376.4 },
|
||||
"deep_probes_requested": false,
|
||||
"dependencies": [
|
||||
{
|
||||
"name": "control_plane_db",
|
||||
"kind": "sqlite",
|
||||
"status": "ok",
|
||||
"detail": "schema v4 readable",
|
||||
"required": true,
|
||||
"healthy": true,
|
||||
"latency_ms": 1.482,
|
||||
"metadata": { "schema_version": 4, "active_leases": 3 }
|
||||
},
|
||||
{ "name": "repository", "kind": "git", "status": "ok", "required": true, "healthy": true },
|
||||
{ "name": "gitea", "kind": "http", "status": "skipped", "required": false, "healthy": false }
|
||||
],
|
||||
"mcp_namespaces": [
|
||||
{ "namespace": "gitea-author", "required_tool": "gitea_whoami", "status": "unproven" }
|
||||
],
|
||||
"stale_runtime": { "stale": false, "determinable": true, "mutation_safe": true, "reasons": [] },
|
||||
"probe_errors": []
|
||||
}
|
||||
```
|
||||
|
||||
No restart, reload, or process-kill control is exposed here: those are Phase 2
|
||||
at the earliest, and #630 forbids process-kill recovery outright. Every probe
|
||||
opens its subject read-only — the control-plane database is opened through a
|
||||
`mode=ro` URI so a health check can never create or migrate a schema.
|
||||
|
||||
## Report audit (#431)
|
||||
|
||||
Paste an LLM final report at `/audit` or POST JSON to `/api/audit`. The UI
|
||||
@@ -153,6 +274,154 @@ health, workflow/schema SHA-256 hashes, and stale-runtime warnings when the
|
||||
checkout is behind merged safety-gate changes. Restart guidance links to #420;
|
||||
no tokens or MCP restart actions are exposed.
|
||||
|
||||
## Application shell — Phase 1 (#638)
|
||||
|
||||
The console shell (`webui/layout.py`) renders a grouped navigation driven by a
|
||||
single nav-config module, `webui/nav.py`. Nav groups follow the epic #631
|
||||
Phase 1 information architecture: **Health, Traffic, Runtime/Sessions,
|
||||
Projects, Inventory, Timeline, Policy** (placeholder), and **Insights**
|
||||
(placeholder). Live views and Phase 1 placeholders (`stub`) are declared in one
|
||||
place so the layout and the route table cannot drift.
|
||||
|
||||
The header carries two read-only status badges — an **environment** badge
|
||||
(`local` for loopback binds, `remote` otherwise, derived from `WEBUI_HOST`) and
|
||||
a **mode: read-only** badge — plus a **Docs** link to this document. No
|
||||
privileged action controls are present in the Phase 1 shell.
|
||||
|
||||
Not-yet-implemented surfaces (`/inventory`, `/timeline`, `/policy`,
|
||||
`/insights`) resolve to graceful read-only stub pages instead of 404s; their
|
||||
backing views land in later child issues of #631 (the inventory surfaces are
|
||||
backed by #636). Mutating methods on stub routes still fail closed with
|
||||
`read-only-mvp`.
|
||||
|
||||
### Runtime and sessions (#641)
|
||||
|
||||
`/sessions` is a live Phase 1 read-only view that composes:
|
||||
|
||||
* runtime health from `#430` (profile, role, identity, master parity, stale warning)
|
||||
* control-plane sessions / leases and filesystem locks / worktrees / namespaces from `#636`
|
||||
* durable contamination markers when detectable (`#630` runtime recovery, `#671` stable-branch push)
|
||||
|
||||
It surfaces stale indicators (dead PID, expired lease) and never silences an
|
||||
active contamination marker. Recovery links point only at sanctioned
|
||||
reconnect/operator restart docs (`docs/mcp-namespace-eof-recovery.md`,
|
||||
`docs/mcp-namespace-health.md`, `docs/mcp-restart-path-inventory.md`, this
|
||||
document). The page does **not** restart, kill, or take over sessions; manual
|
||||
`pkill` of MCP daemons is contamination, not recovery.
|
||||
|
||||
Honesty rules specific to this view:
|
||||
|
||||
* **Ownership columns never assert absence they cannot prove.** When the
|
||||
`leases` or `locks` section is degraded or unavailable, the Leases and
|
||||
Worktree-binding cells render `unknown (inventory <status>)` with an
|
||||
*authority unproven* badge instead of `none` / `unbound`, and a caveat names
|
||||
the unreadable sections. A worktree binding is correlated through lease work
|
||||
numbers, so it is unproven when *either* section fails to read.
|
||||
`/api/sessions` carries the same facts as `ownership_authority_complete`,
|
||||
`ownership_section_status`, and per-row `lease_authority` /
|
||||
`worktree_authority`, so a JSON consumer can tell "holds none" from "could
|
||||
not be read".
|
||||
* **Contamination text is redacted at the display boundary.** Marker payloads
|
||||
(`command_summary`, `reason_class`, `session_id`, `role`) are
|
||||
operator-supplied free text that does not arrive through inventory scrubbing,
|
||||
so they pass through `webui.inventory.scrub_text`, which collapses `$HOME` and
|
||||
redacts credential-shaped tokens and URL userinfo *anywhere* in the string.
|
||||
The write-time redactor is a narrow denylist and is not relied on. The field
|
||||
itself is kept — it is the `#630` evidence naming which daemon was killed.
|
||||
|
||||
## Gitea issue/PR linkage (#645)
|
||||
|
||||
`/gitea` is the Phase 3 read-only linkage console: which PR carries which issue,
|
||||
which issues are claimed by more than one PR, and what the latest Canonical
|
||||
Thread Handoff on a thread said. Gitea remains the source of truth — this
|
||||
surface reads it and never writes to it. There is no issue/PR editor, no review,
|
||||
and no merge control.
|
||||
|
||||
Query parameters (all optional):
|
||||
|
||||
| Parameter | Meaning |
|
||||
|-----------|---------|
|
||||
| `project` | Registry project id to scope the read (default: first registry entry) |
|
||||
| `state` | `open` (default) or `all`; `all` widens the window to merged/closed items, where a landed edge lives |
|
||||
| `issue=N` / `pr=N` | Focus one thread and load *its* latest canonical handoff |
|
||||
|
||||
`GET /api/v1/gitea/linkage` returns the same model as JSON
|
||||
(`schema_version: 1`). It answers `502` when the read could not be answered, so
|
||||
an automated consumer cannot mistake a fail-closed payload for "no links exist".
|
||||
The HTML page always answers `200` and renders the reason instead — an operator
|
||||
view must show why a read failed rather than withhold the page.
|
||||
|
||||
### How an edge is found
|
||||
|
||||
Each edge carries the evidence that produced it, strongest first:
|
||||
|
||||
| Evidence | Meaning |
|
||||
|----------|---------|
|
||||
| `closes_keyword` | The PR title or body declares `closes/fixes/resolves #N`. Gitea itself acts on this keyword. |
|
||||
| `branch_marker` | The PR head branch carries the canonical `(fix\|feat\|docs\|chore)/issue-N-…` marker minted by the issue lock. |
|
||||
| `body_reference` | The PR body mentions `#N` with no closing keyword. A mention is not a claim to close. |
|
||||
|
||||
Only closing and branch-marker edges populate the **issue → PR** direction: a
|
||||
bare mention is a cross-link, and counting it as ownership would invent
|
||||
contested issues out of ordinary references. The mention stays visible on the
|
||||
**PR → issue** side, labelled as such. A PR whose two strongest edges tie is
|
||||
flagged `ambiguous`; an issue claimed by two PRs is flagged `contested`.
|
||||
|
||||
### Honesty rules specific to this view
|
||||
|
||||
* **A partial read never reads as an absence.** Linkage is a claim about the
|
||||
loaded window only. When pagination did not complete, every empty edge cell
|
||||
renders `none found (partial inventory)` rather than `none`, and the JSON
|
||||
carries `inventory_complete: false` plus per-row `links_authoritative: false`.
|
||||
* **A failed read renders no table at all.** Missing credentials, an unknown
|
||||
project, or a fetch error produce `ok: false` with a reason. An empty linkage
|
||||
table would assert that no issue is linked to any PR, which such a read is not
|
||||
in a position to claim.
|
||||
* **Handoffs are loaded, never assumed.** CTH comments are thread-scoped, so
|
||||
only the focused issue or PR has its comments fetched. Every other row reports
|
||||
`not_loaded` with the reason; a thread whose comments *were* loaded and carried
|
||||
no CTH says exactly that. A comment-source failure degrades the handoff alone —
|
||||
the linkage tables still render.
|
||||
* **Unrecognised handoff headings are reported, not republished.** A `## CTH:`
|
||||
heading outside `CTH_TYPES` renders as `unrecognized`.
|
||||
* **Redaction precedes display.** Titles, labels, handoff fields, and error
|
||||
reasons pass through `webui.console_redaction` before serialization, and the
|
||||
page HTML-escapes everything it renders.
|
||||
* **Deep links are opt-in.** A link out to the Gitea web UI appears only when
|
||||
`GITEA_MCP_REVEAL_ENDPOINTS=1` is set server-side, matching how the MCP tools
|
||||
gate URL exposure. Item numbers stay usable without it.
|
||||
|
||||
## System-health dashboard (#639)
|
||||
|
||||
`/system-health` renders the same snapshot the `/api/v1/system/health` API
|
||||
returns, so the page and the API can never disagree. Cards: overall readiness,
|
||||
stale-runtime parity, version and uptime, dependency probes, MCP namespaces,
|
||||
probe errors (only when present), and recovery pointers. `?deep=1` opts into
|
||||
the network probe exactly as the API does; the plain page load stays cheap.
|
||||
|
||||
Field authority and honesty rules:
|
||||
|
||||
* `ready` and `readiness_complete` are shown separately. A snapshot whose
|
||||
required probes never ran is not the same as one that ran them and passed,
|
||||
and the page never collapses the two into an unproven green.
|
||||
* A probe that did not run appears under **Not probed**, never as healthy.
|
||||
* `stale_runtime.mutation_safe` is displayed verbatim from the API. When the
|
||||
runtime is stale, or when parity is indeterminate, the page warns and does
|
||||
not claim mutation safety.
|
||||
* MCP namespaces are reported `unproven`: the web process runs outside the
|
||||
IDE-managed MCP client and cannot prove that path (#543).
|
||||
|
||||
Redaction is split by field kind. Free text — probe details, readiness and
|
||||
parity reasons, probe errors — passes through `system_health.redact`.
|
||||
Structured fields — commit SHAs, probe names, statuses, timestamps — are
|
||||
HTML-escaped only, because `redact`'s opaque-token rule matches any run of 32
|
||||
or more characters and would otherwise blank every 40-character git SHA, which
|
||||
is precisely the evidence the parity view exists to show.
|
||||
|
||||
The dashboard is read-only: no restart, reload, or process-kill control. Those
|
||||
arrive in Phase 2 (#642). Recovery guidance points at the sanctioned client
|
||||
reconnect / operator restart path — never a manual daemon kill (#630).
|
||||
|
||||
## Deployment boundary (#435)
|
||||
|
||||
MVP serves on loopback by default. Binding `0.0.0.0` or `::` is **refused**
|
||||
@@ -212,6 +481,155 @@ health, workflow/schema SHA-256 hashes, and stale-runtime warnings when the
|
||||
checkout is behind merged safety-gate changes. Restart guidance links to #420;
|
||||
no tokens or MCP restart actions are exposed.
|
||||
|
||||
## Inventory API (#636)
|
||||
|
||||
`GET /api/v1/inventory` returns one versioned, read-only snapshot that unifies
|
||||
what the lease (#433), worktree (#432), and runtime (#430) MVP views each show
|
||||
separately, so traffic-control and recovery consumers read the same source.
|
||||
`GET /api/v1/inventory/{section}` returns a single section under the identical
|
||||
schema (`sessions`, `leases`, `locks`, `worktrees`, `namespaces`); an unknown
|
||||
section is a `404` with `error: unknown_section`. Both routes are `GET`-only.
|
||||
|
||||
Each section carries its own `status` (`ok` / `degraded` / `unavailable`), a
|
||||
`reason` when not `ok`, and a `scan_ms`. A subsystem that cannot be read
|
||||
degrades to a reasoned section; it never raises and never emits an empty list
|
||||
that would read as "nothing is there".
|
||||
|
||||
### Field authority
|
||||
|
||||
Every section names where its rows came from; authorities are never blended.
|
||||
|
||||
| Section | Authority | Source |
|
||||
|---|---|---|
|
||||
| `sessions` | `control_plane_db` | #613 control-plane DB (`mode=ro`), authoritative for exclusive ownership (#600/#601) |
|
||||
| `leases` | `control_plane_db` | #613 control-plane DB; degrades if the `work_items` table is absent |
|
||||
| `locks` | `filesystem` | durable per-issue lock files (`issue_lock_store`) |
|
||||
| `worktrees` | `filesystem` | registered git worktrees via the #432 hygiene scanner |
|
||||
| `namespaces` | `filesystem` | the active profile serving this web process (others are not enumerable) |
|
||||
|
||||
The payload restates this map under `field_authority` for machine consumers.
|
||||
|
||||
### Ownership safety
|
||||
|
||||
`ownership_authority_complete` is true only when every ownership-bearing section
|
||||
(`sessions`, `leases`, `locks`) read cleanly. While it is false, nothing is
|
||||
reported as unowned and no collision is asserted from a degraded source —
|
||||
absence of evidence is reported as absence of evidence, never as free work.
|
||||
|
||||
`collisions` surfaces detectable conflicts, each with a `kind` and `severity`:
|
||||
`lock-without-worktree`, `duplicate-live-lock`, `live-lock-dead-owner` (unexpired
|
||||
lease, dead pid — a #753 recovery candidate that would read as live to a naive
|
||||
timestamp check), `stale-lock-dead-owner`, `expired-lock-live-owner` (the
|
||||
#635/#760 daemon-pid deadlock), `concurrent-active-lease`, `active-lease-past-expiry`,
|
||||
and `orphan-lease`. Collisions are emitted only from sections that read cleanly.
|
||||
|
||||
The control-plane DB is opened through a `mode=ro` URI so a read never creates
|
||||
or migrates it; paths are collapsed against `$HOME`, URLs lose userinfo and
|
||||
query strings, and credential-shaped values are redacted at the boundary. Lease
|
||||
steal/release and worktree deletion are Phase 2+ and have no representation here.
|
||||
|
||||
## Workflow-event timeline (#637)
|
||||
|
||||
`GET /api/v1/timeline` is a read-only, versioned aggregation of workflow
|
||||
events from every available source into one normalised, filterable stream. It
|
||||
is the model layer for the Phase 1 timeline console view (a later child issue
|
||||
of #631); this issue ships the schema, adapters, and read API only.
|
||||
|
||||
### Schema (versioned)
|
||||
|
||||
`webui/timeline.py` declares `TIMELINE_SCHEMA_VERSION` (currently `1`) and the
|
||||
frozen `WorkflowEvent` record. Every response carries `schema_version` so a
|
||||
consumer can branch on shape. One event:
|
||||
|
||||
```json
|
||||
{
|
||||
"source": "control_plane",
|
||||
"event_type": "lease.renew",
|
||||
"event_key": "cp:1421",
|
||||
"timestamp": "2026-07-23T02:00:00Z",
|
||||
"actor": null,
|
||||
"role": null,
|
||||
"issue_number": 637,
|
||||
"pr_number": null,
|
||||
"session_id": null,
|
||||
"tool_name": null,
|
||||
"decision": null,
|
||||
"message": "lease renewed",
|
||||
"correlation_id": "issue#637",
|
||||
"evidence_refs": [],
|
||||
"sensitive": true
|
||||
}
|
||||
```
|
||||
|
||||
`event_key` is stable and unique per source (`cp:<event_id>`,
|
||||
`cth:<kind>:<number>:<comment_id>`), so pagination and dedup are deterministic.
|
||||
|
||||
### Sources and field authority
|
||||
|
||||
| Source | Adapter | Authority |
|
||||
|---|---|---|
|
||||
| Control-plane `events` ⋈ `work_items` | `adapt_cp_events` | `event_type`, `message`, `timestamp`, issue/PR scope come from the CP database, read through a `mode=ro` URI (never creates the DB or runs migrations) |
|
||||
| Gitea Canonical Thread Handoff comments | `adapt_cth_comments` | `actor`, `role` (next owner), `decision`, `evidence_refs`, `timestamp` come from the parsed CTH comment body (`canonical_thread_handoff`) |
|
||||
|
||||
Handoff comments are thread-scoped: they are only read when the request filters
|
||||
by a single `issue` or `pr`. Otherwise the handoff source reports `not run`
|
||||
with a reason — it is never rendered as empty-and-healthy. Each source degrades
|
||||
independently: an unavailable control-plane DB or a failed comment fetch is a
|
||||
`sources[]` entry with `ok:false` and a `reason`, never a dropped timeline.
|
||||
|
||||
### Query parameters
|
||||
|
||||
`issue`, `pr`, `session` (conjunctive filters); `limit` (default 50, max 500)
|
||||
and `offset` for pagination; `remote`, `org`, `repo` to override the default
|
||||
registry-project scope. Events sort ascending by
|
||||
`(timestamp, source_rank, event_key)`; missing timestamps sort last.
|
||||
|
||||
### Filter authority, and refusing what cannot be answered
|
||||
|
||||
A filter dimension is only meaningful for a source whose records carry it.
|
||||
Each source declares its own support in `_SOURCE_FILTER_SUPPORT` and reports it
|
||||
per response as `supported_filters` / `unsupported_filters`:
|
||||
|
||||
| Source | issue | pr | session |
|
||||
|---|---|---|---|
|
||||
| `control_plane` | yes | yes | **no** — the `events` table is `(event_id, work_item_id, event_type, message, created_at)` and records no session |
|
||||
| `gitea_handoff` | yes | yes | yes — a CTH comment declares its own `Session:` field |
|
||||
|
||||
`session_id` is read only from that declared CTH field. It is never inferred
|
||||
from a work item, an actor, or message text, and a value that is
|
||||
redaction-altering or bare-secret-shaped is dropped rather than emitted.
|
||||
|
||||
When **no source that ran** can carry a requested dimension, the request is
|
||||
refused rather than answered: the response is `422` with `ok:false` and a
|
||||
structured `error` naming `unsupported_filters` and the per-source reason. A
|
||||
`200` with zero events would tell an operator that no such activity exists,
|
||||
which is a stronger — and false — claim than "this cannot be answered here".
|
||||
A source that *can* answer the dimension and simply matched nothing still
|
||||
returns `200` with `ok:true` and an empty page.
|
||||
|
||||
### Redaction
|
||||
|
||||
Every free-text field (event messages, decision/proof text, roles, actors) is
|
||||
passed through the console redaction policy (`webui.console_redaction`, backed
|
||||
by `gitea_audit.redact`) before it leaves the module, failing closed to the
|
||||
placeholder. No unredacted tool arguments or secrets are ever emitted, and a
|
||||
generation error never drops raw data to a caller or a log.
|
||||
|
||||
Redaction also runs *before* any structured value is derived from free text.
|
||||
`evidence_refs` are extracted from already-redacted proof/decision text, and a
|
||||
commit reference is recognised only where the text declares one (`commit`,
|
||||
`head`, `base`, `sha`, …). An undeclared 40-character hex run has the exact
|
||||
shape of a Gitea access token, so it is never lifted out of prose into a
|
||||
structured field. Every reference is then independently revalidated against an
|
||||
allowed shape and a second redaction pass immediately before serialization;
|
||||
anything unproven is dropped and the event is flagged `sensitive`.
|
||||
|
||||
### Tests
|
||||
|
||||
```bash
|
||||
pytest tests/test_webui_timeline.py -q
|
||||
```
|
||||
|
||||
## Tests
|
||||
|
||||
```bash
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
# Web Console: Notifications & Human-Attention Routing (#648)
|
||||
|
||||
- **Status:** Phase 3 Live
|
||||
- **Tracking Issue:** [#648](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/648)
|
||||
- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/631)
|
||||
- **Attention Boundary Reference:** [#628](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/628)
|
||||
|
||||
---
|
||||
|
||||
## 1. Overview
|
||||
|
||||
The **Notifications & Human-Attention Console** (`/notifications`, `/api/v1/notifications`) provides intelligent event classification and human-attention routing for autonomous workflow operations.
|
||||
|
||||
To prevent alert fatigue while ensuring critical escalation boundaries are never missed, events are classified into three distinct **Attention Classes**:
|
||||
|
||||
1. **`human-required`** (Urgent Escalation Boundary):
|
||||
- Items requiring immediate human intervention or business decisions.
|
||||
- Triggers: Auth failures, hard stops, irrecoverable state, decision locks, failed report validations, critical probe errors.
|
||||
- Display: Highlighted in red (`badge-blocked`) with a `HUMAN REQUIRED` badge.
|
||||
|
||||
2. **`operator`** (Operational Inbox):
|
||||
- Items requiring controller or operator review/triage during routine execution.
|
||||
- Triggers: Blocked PRs (merge conflicts), stale leases, duplicate PRs on issues, unassigned ready work.
|
||||
- Display: Displayed in orange/yellow (`badge-claimed`).
|
||||
|
||||
3. **`routine`** (Background Workflow Transitions):
|
||||
- Normal, healthy workflow transitions and state progressions.
|
||||
- Triggers: Active PRs/issues in standard state, clean branch creation, routine heartbeats.
|
||||
- Display: Filtered out of default inbox views to eliminate notification spam; viewable on demand via the "Routine" or "All" tab.
|
||||
|
||||
---
|
||||
|
||||
## 2. API Endpoints
|
||||
|
||||
### `GET /api/v1/notifications`
|
||||
*Compatibility Alias:* `GET /api/notifications`
|
||||
|
||||
#### Query Parameters:
|
||||
- `project_id` (optional): Filter notifications by project ID.
|
||||
- `attention_class` (optional): `inbox` (default: human-required + operator), `human-required`, `operator`, `routine`, `all`.
|
||||
|
||||
#### Example JSON Response:
|
||||
```json
|
||||
{
|
||||
"project_id": "gitea-tools",
|
||||
"repo_label": "Scaled-Tech-Consulting/Gitea-Tools",
|
||||
"human_required_count": 0,
|
||||
"operator_count": 2,
|
||||
"routine_count": 5,
|
||||
"total_count": 7,
|
||||
"fetch_error": null,
|
||||
"inbox_items": [
|
||||
{
|
||||
"id": "notif-pr-block-742",
|
||||
"attention_class": "operator",
|
||||
"category": "blocker",
|
||||
"title": "Blocked PR #742",
|
||||
"summary": "PR #742 requires merge conflict resolution.",
|
||||
"work_kind": "pr",
|
||||
"work_number": 742,
|
||||
"project_id": "gitea-tools",
|
||||
"repo_label": "Scaled-Tech-Consulting/Gitea-Tools",
|
||||
"created_at": "2026-07-25T16:39:47Z",
|
||||
"deep_link": "/traffic",
|
||||
"requires_human": false,
|
||||
"extra": {}
|
||||
}
|
||||
],
|
||||
"all_items": [...]
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. UI Navigation
|
||||
|
||||
- Access via the **Traffic** navigation menu: **Traffic → Notifications**.
|
||||
- The main view displays:
|
||||
- **Metrics Summary Bar**: Highlighting counts for Human Required, Operator Inbox, and Routine items.
|
||||
- **Attention Filter Tabs**: Toggle between Inbox (Human + Operator), Human Required, Operator, Routine, and All.
|
||||
- **Structured Event Table**: Displays category, title, summary, work item links, and timestamps.
|
||||
@@ -0,0 +1,160 @@
|
||||
# Web console requests: intent preview and workflow initiation (#643)
|
||||
|
||||
**Phase 2. Preview is always live and always read-only. Initiation is wired but
|
||||
denied until an operator opts in.**
|
||||
|
||||
Before this surface, starting role work meant pasting a prompt into a terminal
|
||||
and trusting the operator to have checked the allocator first. Nothing enforced
|
||||
that check, so two sessions could reach for the same issue and each believe it
|
||||
was theirs. This page replaces the paste with a *request*: a desired role, an
|
||||
issue or PR, and a stated intent, answered by an authorization decision and —
|
||||
on confirmation — an exclusive assignment from the allocator.
|
||||
|
||||
| Concern | Module |
|
||||
|---------|--------|
|
||||
| Request model, preview, initiation | `webui/request_service.py` |
|
||||
| Form and preview rendering | `webui/request_views.py` |
|
||||
| Authorization | `webui/console_authz.py` (`initiate_workflow`) |
|
||||
| Audit | `webui/console_audit.py` |
|
||||
| Ownership substrate | `allocator_service.py` + `control_plane_db.py` |
|
||||
|
||||
## Surfaces
|
||||
|
||||
| Path | Method | Purpose |
|
||||
|------|--------|---------|
|
||||
| `/requests` | GET | Request form |
|
||||
| `/requests` | POST | Render an intent preview. **Never assigns.** |
|
||||
| `/api/v1/requests/preview` | POST | Intent preview as JSON |
|
||||
| `/api/v1/requests/apply` | POST | Initiate — confirmed, audited, allocator-owned |
|
||||
|
||||
The HTML form has no initiate button on purpose. Initiating requires a
|
||||
confirmed POST to `/api/v1/requests/apply`, so a stray form submission cannot
|
||||
reserve work as a side effect.
|
||||
|
||||
## The request
|
||||
|
||||
```json
|
||||
{
|
||||
"desired_role": "author",
|
||||
"work_kind": "issue",
|
||||
"work_number": 643,
|
||||
"intent_summary": "implement request preview and initiation",
|
||||
"remote": "prgs",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"expected_head_sha": null
|
||||
}
|
||||
```
|
||||
|
||||
`desired_role` is one of `author`, `reviewer`, `merger`, `reconciler`,
|
||||
`controller`. `work_kind` is `issue` or `pr`. `remote`/`org`/`repo` default to
|
||||
the first project in the registry when omitted; when neither the request nor
|
||||
the registry resolves them, the request is rejected rather than pointed at some
|
||||
other repository. `intent_summary` is required — it is what the audit record
|
||||
states as the reason — and is truncated to 500 characters.
|
||||
|
||||
Parsing rejects rather than corrects. An unknown role, an unknown work kind, a
|
||||
non-positive number, or a missing intent each return `400` with a `reason_code`
|
||||
and the offending `field`.
|
||||
|
||||
## Preview
|
||||
|
||||
Five checks, each with its own verdict, reason code, and detail:
|
||||
|
||||
| Check | Passes when |
|
||||
|-------|-------------|
|
||||
| `authorization` | The console principal holds `operator` or above |
|
||||
| `capability` | The desired role maps to a declared profile and MCP namespace |
|
||||
| `lease_availability` | No active claim holds the work unit |
|
||||
| `next_safe_action` | The allocator would independently select this exact work unit |
|
||||
| `head_pin` | PR work resolves to a head SHA, and a supplied SHA still matches |
|
||||
|
||||
A preview also returns the role's `allowed_actions` and `prohibited_actions`
|
||||
(from `allocator_service.ROLE_ACTIONS`), the `required_profile` and
|
||||
`required_namespace` the work must run under, and a `correlation_id` that ties
|
||||
the preview to its audit record and to any assignment that follows.
|
||||
|
||||
Preview is read-only in the strict sense: it calls the allocator with
|
||||
`apply=false` and writes nothing but an audit line. An unauthorized principal
|
||||
never reaches the allocator or the control-plane DB at all, so a denial cannot
|
||||
be used to enumerate the queue.
|
||||
|
||||
## Initiation
|
||||
|
||||
`POST /api/v1/requests/apply` refuses in this order, and every refusal returns
|
||||
before any assignment is attempted:
|
||||
|
||||
| Condition | Outcome | Status |
|
||||
|-----------|---------|--------|
|
||||
| Unparseable request | `invalid_request` | 400 |
|
||||
| Not authorized, or execution not wired | `denied` | 403 |
|
||||
| `confirm` not set | `denied` / `confirmation_required` | 409 |
|
||||
| Work unit already claimed | `blocked` / `duplicate_assignment` | 409 |
|
||||
| Allocator would select other work | `wait` / `not_next_safe_work` | 409 |
|
||||
| Allocator declines on apply | `blocked` or `wait` | 409 |
|
||||
| Evidence unavailable | `wait` / `evidence_unavailable` | 503 |
|
||||
| Assigned | `assigned_work` | 201 |
|
||||
|
||||
A success returns the assignment plus a `handoff` block naming the profile, the
|
||||
namespace, and the actions that stay forbidden — enough for the operator to
|
||||
continue in the right MCP namespace without guessing.
|
||||
|
||||
### Why apply runs the allocator twice
|
||||
|
||||
The allocator is the only source of exclusive ownership (#600 / #613), and it
|
||||
selects work; it does not take orders. So `apply` runs a dry-run first and
|
||||
proceeds only when the allocator would independently pick the requested work
|
||||
unit. If it would not, the request reports `wait` and mutates nothing.
|
||||
|
||||
A request is therefore a *confirmation* of the allocator's decision, never an
|
||||
override of it. The apply call carries the dry-run's
|
||||
`candidate_set_fingerprint` as a CAS pin (#776), so a queue that changed
|
||||
between the two calls fails closed rather than assigning against a stale view.
|
||||
The result is checked again on the way out: an assignment naming a different
|
||||
work unit is not read as success.
|
||||
|
||||
### Fail-closed defaults
|
||||
|
||||
- An unreadable control-plane DB denies. It is never treated as "nothing holds
|
||||
this work unit".
|
||||
- An incomplete queue inventory denies (#758). Ranking a partial candidate set
|
||||
can select the wrong work.
|
||||
- An allocator that raises denies.
|
||||
- PR work with no resolvable head SHA denies; a supplied SHA that no longer
|
||||
matches denies with `head_moved`.
|
||||
|
||||
## Enabling initiation
|
||||
|
||||
Execution is wired off. Set `WEBUI_REQUESTS_EXECUTION=1` to enable it for the
|
||||
`initiate_workflow` action only — see
|
||||
[`webui-authz-audit.md`](webui-authz-audit.md) for why this is an
|
||||
action-scoped flag rather than a phase bump. With the variable unset, `apply`
|
||||
returns `403` with `reason_code: unauthorized` no matter who asks.
|
||||
|
||||
Enabling execution does **not** enable approvals or merges. Those are phase 3
|
||||
console actions and remain forbidden in every path here; the console reserves
|
||||
work and hands off, and the MCP role profile enforces what that role may then
|
||||
do.
|
||||
|
||||
## Audit
|
||||
|
||||
Every preview and every apply emits a console audit record (schema in
|
||||
[`webui-authz-audit.md`](webui-authz-audit.md)):
|
||||
|
||||
| Event | `result` |
|
||||
|-------|----------|
|
||||
| Preview | `previewed` |
|
||||
| Refusal at any stage | `denied` |
|
||||
| Assignment created | `succeeded` |
|
||||
|
||||
`correlation.request_id` carries the request's `correlation_id`, and a
|
||||
successful record's `metadata` carries `assignment_id` and `lease_id`, so an
|
||||
assignment can be traced back to the intent that produced it. The operator's
|
||||
`intent_summary` travels in `metadata` and passes through the standard
|
||||
redaction pass before persistence like every other field.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- No browser-initiated approve or merge, in this phase or any other.
|
||||
- No bypass of allocator exclusive ownership; no self-selection of work.
|
||||
- No auto-start from raw monitoring incidents (#612 stays downstream).
|
||||
@@ -0,0 +1,102 @@
|
||||
# Web Console: restart status, impact preview, and approval state (#667)
|
||||
|
||||
Phase 1 of the console restart surface. It consumes the #655 coordinator
|
||||
substrate and displays it. It performs no restart, reload, drain, approval, or
|
||||
process action, and it registers no write endpoint.
|
||||
|
||||
Issue #667's rollout is explicit — *status views first, write approval after the
|
||||
backend gates are green* — and this change delivers only the status half.
|
||||
|
||||
## Surfaces
|
||||
|
||||
| Path | Method | Purpose |
|
||||
|------|--------|---------|
|
||||
| `/runtime/restart` | GET | Restart status page |
|
||||
| `/api/v1/system/restart/status` | GET | Same snapshot as JSON |
|
||||
|
||||
Both accept an optional `restart_class` query parameter (default
|
||||
`full_mcp_restart`). An unrecognised class is not an error: the coordinator
|
||||
resolves it as unknown and fails closed, and the page shows the resulting deny.
|
||||
|
||||
Neither path accepts `POST`; a write attempt returns `405`, and a test asserts
|
||||
it.
|
||||
|
||||
## What it shows
|
||||
|
||||
* **Impact preview (#658)** — verdict, blast radius, affected sessions, leases,
|
||||
critical sections, mutations, and the counts behind them, evaluated
|
||||
`dry_run=True` against live control-plane state.
|
||||
* **Drain proof (#661)** — verification of a supplied proof: valid, clean,
|
||||
expired, tampered, and the reasons behind a refusal.
|
||||
* **Post-restart reconcile (#662)** — the most recent completion proof, its
|
||||
overall status, and which dimensions still require follow-up.
|
||||
* **Restart classes (#663)** — the least-privilege matrix, with *you may
|
||||
request* and *you may execute* computed for the viewing role rather than for a
|
||||
generic operator.
|
||||
* **Approval controls (#633)** — the authorization state of
|
||||
`system.restart_namespace` and `system.reload_namespace`.
|
||||
* **Break-glass (#664)** — declared and marked unavailable; see below.
|
||||
|
||||
## Three rules this surface holds itself to
|
||||
|
||||
A status page that is wrong is worse than one that is missing, because an
|
||||
operator acts on it. Three properties are enforced by tests, and each was
|
||||
verified by reverting the guard and watching a test fail.
|
||||
|
||||
### An unreadable source reports unavailable, never green
|
||||
|
||||
Every source carries its own `SourceStatus`. Nothing substitutes a default,
|
||||
placeholder, or self-comparison for a reading that failed. An unreadable
|
||||
control-plane database yields `inventory_complete: false`, which the coordinator
|
||||
itself turns into a fail-closed verdict, and the page says the blast radius is
|
||||
unknown rather than showing an empty affected-sessions table.
|
||||
|
||||
An absent drain proof is reported as absent — not as a pass. The #661 gate
|
||||
authorizes a restart only against a valid, unexpired, clean proof, so no proof
|
||||
is precisely the state that gate denies on.
|
||||
|
||||
### Authorization is asked the way execution would ask it
|
||||
|
||||
Every probe passes `for_execution=True`.
|
||||
|
||||
Asked without it, an admin is `allowed` for `system.restart_namespace`. On a
|
||||
control surface that reads as a live button. Asked the way an execution attempt
|
||||
would ask, the same principal is refused `phase_not_active`, because the console
|
||||
is in Phase 1 and the action is Phase 2. This surface reports the second answer.
|
||||
|
||||
`execution_enabled` is therefore `false` for every action and every role today,
|
||||
and a test asserts that across the whole role matrix.
|
||||
|
||||
### The control-plane database is opened read-only
|
||||
|
||||
`ControlPlaneDB()` creates directories and runs migrations on construction — a
|
||||
write. This surface never constructs one. It opens the sqlite file with
|
||||
`mode=ro`, exactly as `webui/inventory.py` does, and treats a missing file as
|
||||
missing authority rather than as an empty inventory.
|
||||
|
||||
The test that protects this points at a path inside a directory that already
|
||||
exists, so a read-write `connect` would really create the file. A nested
|
||||
missing-directory path would have passed for the wrong reason.
|
||||
|
||||
## Break-glass is declared, not offered
|
||||
|
||||
The break-glass workflow (#664) is not available on this branch's base. The
|
||||
panel is rendered to operator-class roles as **unavailable**, naming the issue
|
||||
that tracks it. It is not silently omitted, because an operator who has been
|
||||
told a governance path exists needs to see that it is not wired here; and it is
|
||||
not rendered as a control, because there is nothing behind it.
|
||||
|
||||
Unprivileged viewers see only a note that the surface is operator-class.
|
||||
|
||||
## Redaction and escaping
|
||||
|
||||
Every interpolated value passes through `_esc` (`html.escape(..., quote=True)`).
|
||||
Free-form text and anything that can carry a filesystem path additionally passes
|
||||
through `webui.inventory.scrub_text`, which redacts credential-shaped tokens
|
||||
inside a string rather than only at its start. The impact payload is passed
|
||||
through `webui.inventory.scrub` before rendering.
|
||||
|
||||
## Linkage
|
||||
|
||||
Parent #655 · extends #642 · consumes #658, #661, #662, #663 · RBAC #633 ·
|
||||
console #631 · vision #652 · roadmap #653 · break-glass #664.
|
||||
+1015
File diff suppressed because it is too large
Load Diff
+50
-1
@@ -1169,10 +1169,57 @@ def server_command():
|
||||
return python, [os.path.join(root, "mcp_server.py")]
|
||||
|
||||
|
||||
RECOGNIZED_GITEA_ENV_KEYS = frozenset({
|
||||
"GITEA_MCP_CONFIG",
|
||||
"GITEA_MCP_PROFILE",
|
||||
"GITEA_PROFILE_NAME",
|
||||
"GITEA_SERVICE",
|
||||
"GITEA_EXECUTION_ROLE",
|
||||
"GITEA_CLIENT_MANAGED",
|
||||
"GITEA_MCP_CLIENT_MANAGED",
|
||||
"GITEA_SERVER_PROVENANCE",
|
||||
"GITEA_AUTHOR_WORKTREE",
|
||||
"GITEA_ACTIVE_WORKTREE",
|
||||
"GITEA_DISABLE_KEYCHAIN",
|
||||
"GITEA_CONTROL_PLANE_DB",
|
||||
"GITEA_DB_PATH",
|
||||
"GITEA_LOG_LEVEL",
|
||||
"GITEA_DEBUG",
|
||||
"GITEA_HMAC_SECRET",
|
||||
"GITEA_IRRECOVERABLE_HMAC_SECRET",
|
||||
"GITEA_FORCE_MCP_RUNTIME_CHECK",
|
||||
"GITEA_FORCE_CLIENT_MANAGED",
|
||||
})
|
||||
|
||||
RECOGNIZED_GITEA_ENV_PREFIXES = (
|
||||
"GITEA_TOKEN_",
|
||||
"GITEA_PASS_",
|
||||
"GITEA_USER_",
|
||||
"GITEA_URL_",
|
||||
"GITEA_HOST_",
|
||||
"GITEA_REMOTE_",
|
||||
"GITEA_HTTP_HEADER_",
|
||||
)
|
||||
|
||||
|
||||
def get_unconsumed_gitea_env_overrides(env=None) -> dict[str, str]:
|
||||
"""Find unsupported GITEA_* env vars present in *env* (defaults to os.environ)."""
|
||||
target = os.environ if env is None else env
|
||||
unconsumed = {}
|
||||
for key, value in target.items():
|
||||
if key.startswith("GITEA_"):
|
||||
if key in RECOGNIZED_GITEA_ENV_KEYS:
|
||||
continue
|
||||
if any(key.startswith(p) for p in RECOGNIZED_GITEA_ENV_PREFIXES):
|
||||
continue
|
||||
unconsumed[key] = str(value)
|
||||
return unconsumed
|
||||
|
||||
|
||||
def launcher_entry(profile_name, config_path=None):
|
||||
"""Return a thin MCP launcher entry for *profile_name*.
|
||||
|
||||
Contains only command/args and the two GITEA_MCP_* env vars — never a token
|
||||
Contains command/args and the GITEA_MCP_* / GITEA_CLIENT_MANAGED env vars — never a token
|
||||
or password. Suitable for Claude / Gemini / Codex ``mcpServers`` blocks.
|
||||
"""
|
||||
command, args = server_command()
|
||||
@@ -1183,11 +1230,13 @@ def launcher_entry(profile_name, config_path=None):
|
||||
"env": {
|
||||
"GITEA_MCP_CONFIG": config_path or DEFAULT_CONFIG_PATH,
|
||||
"GITEA_MCP_PROFILE": profile_name,
|
||||
"GITEA_CLIENT_MANAGED": "1",
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
def keychain_set(item_id, token, account=None, runner=subprocess.run):
|
||||
"""Store *token* in the macOS keychain under service *item_id*.
|
||||
|
||||
|
||||
+3059
-131
File diff suppressed because it is too large
Load Diff
@@ -16,11 +16,18 @@ ISSUE_LOCK_FILE = os.environ.get("GITEA_ISSUE_LOCK_FILE", "/tmp/gitea_issue_lock
|
||||
SOURCE_LOCK_ISSUE = "gitea_lock_issue"
|
||||
SOURCE_LOCK_ADOPTION = "gitea_lock_issue_adoption"
|
||||
SOURCE_OPERATOR_OVERRIDE = "operator_override"
|
||||
SOURCE_RECOVER_DIRTY_ORPHANED = "gitea_recover_dirty_orphaned_issue_worktree"
|
||||
# #864: dirty-preserving same-claimant author-session rebind (dead owner PID).
|
||||
SOURCE_DIRTY_SAME_CLAIMANT_REBIND = (
|
||||
"gitea_rebind_dirty_same_claimant_author_session"
|
||||
)
|
||||
|
||||
SANCTIONED_LOCK_SOURCES = frozenset({
|
||||
SOURCE_LOCK_ISSUE,
|
||||
SOURCE_LOCK_ADOPTION,
|
||||
SOURCE_OPERATOR_OVERRIDE,
|
||||
SOURCE_RECOVER_DIRTY_ORPHANED,
|
||||
SOURCE_DIRTY_SAME_CLAIMANT_REBIND,
|
||||
})
|
||||
|
||||
_OPERATOR_OVERRIDE_ENV = "GITEA_ISSUE_LOCK_OPERATOR_OVERRIDE"
|
||||
|
||||
+130
-8
@@ -85,6 +85,12 @@ HEAD_RELATION_STRICT_DESCENDANT = "strict_descendant"
|
||||
# #772: an unpublished claim has no recorded head to compare against at all, so
|
||||
# its head is measured against the base the branch was cut from instead.
|
||||
HEAD_RELATION_DESCENDS_FROM_BASE = "descends_from_recorded_base"
|
||||
# #871: the remote/PR head advanced *past* the recorded head via a sanctioned
|
||||
# merge-based branch synchronization (``gitea_update_pr_branch_by_merge``) while
|
||||
# the local worktree stayed at the recorded head. This is the inverse of the
|
||||
# #768 descendant relation — here the *remote* strictly descends the local head,
|
||||
# and only because a base was merged into the branch, proven server-side.
|
||||
HEAD_RELATION_REMOTE_MERGE_SYNCED = "remote_merge_synced"
|
||||
|
||||
# Which body of evidence a recovery was decided on (#772 AC10). These are not
|
||||
# interchangeable: a published claim proves ownership against a remote/PR head,
|
||||
@@ -266,6 +272,70 @@ def _assess_base_descendancy(
|
||||
]
|
||||
|
||||
|
||||
def _assess_remote_merge_synced(
|
||||
sync_provenance: Mapping[str, Any] | None,
|
||||
*,
|
||||
recorded_head: str,
|
||||
remote_head: str,
|
||||
) -> tuple[bool, list[str]]:
|
||||
"""Did ``remote_head`` advance past ``recorded_head`` via a sanctioned
|
||||
merge-based branch sync (#871)?
|
||||
|
||||
``sync_provenance`` is the server-side git observation from
|
||||
``issue_lock_worktree.read_merge_sync_provenance``. Its own
|
||||
``prior_head_sha`` / ``synced_head_sha`` are re-checked against the heads
|
||||
this assessment is actually reasoning about, so an observation taken for some
|
||||
other pair of commits — stale, mismatched, or hand-built — can never
|
||||
authorize recovery. This is the inverse of ``_assess_strict_descendant``: the
|
||||
recorded head is the ancestor and the *remote* head is the descendant, and it
|
||||
is accepted only because the remote head is a base-into-branch merge that
|
||||
preserved the branch mainline back to the recorded head.
|
||||
|
||||
Returns ``(proven, notes)``. Notes name the exact missing element so a
|
||||
refused caller sees why, never a bare "unproven".
|
||||
"""
|
||||
if not isinstance(sync_provenance, Mapping):
|
||||
return False, [
|
||||
"no server-derived merge-sync provenance observation was available; a "
|
||||
"remote head ahead of the recorded head cannot be accepted"
|
||||
]
|
||||
|
||||
probe_prior = _text(sync_provenance.get("prior_head_sha"))
|
||||
probe_synced = _text(sync_provenance.get("synced_head_sha"))
|
||||
if probe_prior != recorded_head or probe_synced != remote_head:
|
||||
return False, [
|
||||
f"merge-sync observation covers {probe_prior or 'unknown'} -> "
|
||||
f"{probe_synced or 'unknown'}, not the heads under assessment "
|
||||
f"({recorded_head} -> {remote_head})"
|
||||
]
|
||||
if not sync_provenance.get("probe_ok"):
|
||||
return False, (
|
||||
list(sync_provenance.get("reasons") or [])
|
||||
or ["merge-sync provenance probe did not complete; provenance unproven"]
|
||||
)
|
||||
if not sync_provenance.get("prior_is_ancestor"):
|
||||
return False, [
|
||||
f"recorded head {recorded_head} is not an ancestor of remote head "
|
||||
f"{remote_head}; a rewritten or force-moved head cannot be recovered"
|
||||
]
|
||||
if not sync_provenance.get("is_merge_sync"):
|
||||
return False, (
|
||||
list(sync_provenance.get("reasons") or [])
|
||||
or [
|
||||
f"remote head {remote_head} is not a sanctioned merge-based sync "
|
||||
f"of the base into the branch above {recorded_head}"
|
||||
]
|
||||
)
|
||||
|
||||
proof = _text(sync_provenance.get("proof")) or (
|
||||
f"{remote_head} merged the base into the branch above {recorded_head}"
|
||||
)
|
||||
return True, [
|
||||
f"remote head {remote_head} advanced past recorded head {recorded_head} "
|
||||
f"via a sanctioned merge-based branch sync ({proof})"
|
||||
]
|
||||
|
||||
|
||||
def assess_dead_session_lock_recovery(
|
||||
existing_lock: Mapping[str, Any] | None,
|
||||
*,
|
||||
@@ -290,6 +360,7 @@ def assess_dead_session_lock_recovery(
|
||||
remote_branch_exists: bool | None = None,
|
||||
recorded_base_sha: str | None = None,
|
||||
base_ancestry: Mapping[str, Any] | None = None,
|
||||
sync_provenance: Mapping[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Decide whether a dead-session author lock may be natively recovered.
|
||||
|
||||
@@ -469,19 +540,39 @@ def assess_dead_session_lock_recovery(
|
||||
head_relation = HEAD_RELATION_STRICT_DESCENDANT
|
||||
ancestry_proof = notes[0] if notes else None
|
||||
else:
|
||||
reasons.append(
|
||||
f"local head {local_head} does not match remote branch head "
|
||||
f"{remote_head}"
|
||||
# #871: the reverse relation — the remote head advanced past
|
||||
# the recorded/local head via a sanctioned merge-based branch
|
||||
# sync while the local worktree stayed put. Accepted only on
|
||||
# server-proven merge-sync provenance, never a caller claim.
|
||||
synced, sync_notes = _assess_remote_merge_synced(
|
||||
sync_provenance,
|
||||
recorded_head=local_head,
|
||||
remote_head=remote_head,
|
||||
)
|
||||
reasons.extend(notes)
|
||||
if synced:
|
||||
head_relation = HEAD_RELATION_REMOTE_MERGE_SYNCED
|
||||
ancestry_proof = sync_notes[0] if sync_notes else None
|
||||
else:
|
||||
reasons.append(
|
||||
f"local head {local_head} does not match remote branch "
|
||||
f"head {remote_head}"
|
||||
)
|
||||
reasons.extend(notes)
|
||||
reasons.extend(sync_notes)
|
||||
evidence["recorded_base"] = recorded_base or None
|
||||
evidence["local_head"] = local_head or None
|
||||
evidence["remote_head"] = remote_head or None
|
||||
# ``recorded_head`` is the head recovery is being measured against;
|
||||
# ``accepted_head`` is the head this recovery actually adopts. They differ
|
||||
# only in the descendant case, and downstream gates need both (#768 AC2/AC7).
|
||||
# #871: in the merge-sync case the branch/PR already carries the synced
|
||||
# remote head, so that is the head recovery adopts; the local worktree stays
|
||||
# at the ancestor recorded head.
|
||||
evidence["recorded_head"] = remote_head or None
|
||||
evidence["accepted_head"] = local_head or None
|
||||
if head_relation == HEAD_RELATION_REMOTE_MERGE_SYNCED:
|
||||
evidence["accepted_head"] = remote_head or None
|
||||
else:
|
||||
evidence["accepted_head"] = local_head or None
|
||||
evidence["head_relation"] = head_relation
|
||||
evidence["ancestry_proof"] = ancestry_proof
|
||||
|
||||
@@ -493,12 +584,17 @@ def assess_dead_session_lock_recovery(
|
||||
# contradictory; re-stating it as a head mismatch would only obscure why.
|
||||
if not unpublished and local_head and pr_head != local_head:
|
||||
# A descendant recovery has not been published yet, so the open PR
|
||||
# legitimately still points at the recorded head. Any other
|
||||
# disagreement is a real mismatch.
|
||||
# legitimately still points at the recorded head. A merge-sync
|
||||
# recovery's PR legitimately sits at the advanced remote head. Any
|
||||
# other disagreement is a real mismatch.
|
||||
if not (
|
||||
head_relation == HEAD_RELATION_STRICT_DESCENDANT
|
||||
and remote_head
|
||||
and pr_head == remote_head
|
||||
) and not (
|
||||
head_relation == HEAD_RELATION_REMOTE_MERGE_SYNCED
|
||||
and remote_head
|
||||
and pr_head == remote_head
|
||||
):
|
||||
reasons.append(
|
||||
f"open PR #{pr_number} head {pr_head} does not match local head "
|
||||
@@ -619,7 +715,11 @@ def assess_dead_session_lock_recovery(
|
||||
)
|
||||
if (
|
||||
head_relation
|
||||
in (HEAD_RELATION_STRICT_DESCENDANT, HEAD_RELATION_DESCENDS_FROM_BASE)
|
||||
in (
|
||||
HEAD_RELATION_STRICT_DESCENDANT,
|
||||
HEAD_RELATION_DESCENDS_FROM_BASE,
|
||||
HEAD_RELATION_REMOTE_MERGE_SYNCED,
|
||||
)
|
||||
and ancestry_proof
|
||||
):
|
||||
proof.append(ancestry_proof)
|
||||
@@ -697,6 +797,16 @@ def owning_pr_recovery_evidence(
|
||||
return None
|
||||
if accepted_head != local_head:
|
||||
return None
|
||||
elif relation == HEAD_RELATION_REMOTE_MERGE_SYNCED:
|
||||
# #871: the PR already sits at the advanced remote head; the local
|
||||
# worktree is the ancestor the merge preserved. The head the open PR
|
||||
# shows and the head recovery adopts are both the synced remote head.
|
||||
if not remote_head or pr_head != remote_head:
|
||||
return None
|
||||
if accepted_head != remote_head:
|
||||
return None
|
||||
if not local_head or local_head == remote_head:
|
||||
return None
|
||||
else:
|
||||
return None
|
||||
try:
|
||||
@@ -765,6 +875,18 @@ def recovered_owning_pr_from_lock(
|
||||
return None
|
||||
if not accepted_head or accepted_head == recorded_head:
|
||||
return None
|
||||
elif relation == HEAD_RELATION_REMOTE_MERGE_SYNCED:
|
||||
# #871: PR sits at the advanced remote head, which is both the recorded
|
||||
# measured-against head and the adopted head; the local worktree is the
|
||||
# ancestor the merge preserved.
|
||||
remote_head = _text(record.get("remote_head"))
|
||||
local_head = _text(record.get("local_head"))
|
||||
if not remote_head or pr_head != remote_head:
|
||||
return None
|
||||
if accepted_head and accepted_head != remote_head:
|
||||
return None
|
||||
if not local_head or local_head == remote_head:
|
||||
return None
|
||||
else:
|
||||
return None
|
||||
try:
|
||||
|
||||
@@ -436,6 +436,117 @@ def owning_pr_renewal_evidence(
|
||||
}
|
||||
|
||||
|
||||
def owning_pr_renewal_from_lock(
|
||||
lock_record: Mapping[str, Any] | None,
|
||||
) -> dict[str, Any] | None:
|
||||
"""Rebuild owning-PR renewal evidence from a persisted lock (#945).
|
||||
|
||||
The renewal mirror of ``issue_lock_recovery.recovered_owning_pr_from_lock``.
|
||||
``owning_pr_renewal_evidence`` supplies the waiver for the duration of the
|
||||
``gitea_lock_issue`` call only. The commit, push, create-PR, and
|
||||
duplicate-assessment gates run later in their own calls and re-derive
|
||||
ownership from the durable lock instead — so without this the open PR that
|
||||
renewal already proved belongs to this author reappears there as competing
|
||||
duplicate work, and the exact owner is refused with
|
||||
``duplicate_commit_prevented`` despite complete matching evidence.
|
||||
|
||||
This reads only the ``lease_renewal`` block that the server itself writes,
|
||||
on a lock the caller must already own. Like the recovery mirror it is a
|
||||
re-read of server-derived state, never a fresh assertion: a caller able to
|
||||
forge it could equally forge the lock file every other ownership gate
|
||||
already treats as authoritative.
|
||||
|
||||
Renewal has no descendant case — the assessor required the local, remote and
|
||||
PR heads to be equal — so that equality is re-checked here, and the record
|
||||
must still name the claimant the lock records.
|
||||
|
||||
**What the claimant check below is, and what it is not.** It compares
|
||||
``lease_renewal.identity``/``profile`` against the claimant recorded on the
|
||||
*same* lock file. Both sides are server-written fields of one document, so
|
||||
this is an internal-consistency check: it rejects a lock whose renewal block
|
||||
and claimant disagree. It does **not** consult the live authenticated caller
|
||||
and therefore does not, on its own, prove that the session invoking a later
|
||||
gate is the session the renewal was granted to.
|
||||
|
||||
The binding that actually keeps one session from using another's renewal is
|
||||
structural, and it lives in the caller rather than here. The enforcement
|
||||
paths load the lock through ``_load_existing_issue_lock()`` with no issue
|
||||
coordinates, which resolves ``issue_lock_store.read_session_issue_lock()``
|
||||
→ the session pointer at ``session-{os.getpid()}.json``. Lock *selection* is
|
||||
scoped to the operating-system process, so a caller cannot aim the recheck
|
||||
at a lock some other process bound. Its limits follow from what that scope
|
||||
is: it is per-process, not per-authenticated-user; it says nothing about a
|
||||
lock reached by explicit issue coordinates rather than the session pointer,
|
||||
and nothing about two roles sharing one process. Live identity and profile
|
||||
are enforced separately, by the mutation-authority and profile gates each
|
||||
mutating path already runs — not by this rebuild.
|
||||
|
||||
This function is therefore strictly a re-read with an added consistency
|
||||
requirement. It narrows what a persisted lock can authorize; it never widens
|
||||
it, and it never substitutes for a caller-identity gate.
|
||||
"""
|
||||
if not isinstance(lock_record, Mapping):
|
||||
return None
|
||||
record = lock_record.get("lease_renewal")
|
||||
if not isinstance(record, Mapping) or not record.get("renewed"):
|
||||
return None
|
||||
|
||||
branch_name = _text(record.get("branch_name")) or _text(
|
||||
lock_record.get("branch_name")
|
||||
)
|
||||
pr_head = _text(record.get("pr_head_sha"))
|
||||
local_head = _text(record.get("head_sha"))
|
||||
remote_head = _text(record.get("remote_head_sha"))
|
||||
raw_pr_number = record.get("pr_number")
|
||||
raw_issue_number = lock_record.get("issue_number")
|
||||
|
||||
if raw_pr_number is None or raw_issue_number is None:
|
||||
return None
|
||||
if not branch_name or not pr_head:
|
||||
return None
|
||||
# The assessor required all three heads to agree before it granted renewal.
|
||||
# Re-check, so a truncated, drifted, or hand-built record cannot widen the
|
||||
# exemption past the single head the renewal disposition actually proved.
|
||||
if not local_head or not remote_head:
|
||||
return None
|
||||
if pr_head != local_head or pr_head != remote_head:
|
||||
return None
|
||||
# Renewal is refused outright unless the durable lock records both a
|
||||
# claimant username and profile, so a sanctioned record always carries them.
|
||||
# Requiring them to still agree rejects a lock whose renewal block and
|
||||
# claimant disagree. Both values are read from this one server-written
|
||||
# document: this is internal consistency, not a check against the live
|
||||
# authenticated caller — see the docstring for the binding that is.
|
||||
claimant = lock_record.get("claimant")
|
||||
if not isinstance(claimant, Mapping):
|
||||
lease = lock_record.get("work_lease")
|
||||
claimant = lease.get("claimant") if isinstance(lease, Mapping) else None
|
||||
if not isinstance(claimant, Mapping):
|
||||
return None
|
||||
identity = _text(record.get("identity"))
|
||||
profile = _text(record.get("profile"))
|
||||
if not identity or identity != _text(claimant.get("username")):
|
||||
return None
|
||||
if not profile or profile != _text(claimant.get("profile")):
|
||||
return None
|
||||
|
||||
try:
|
||||
pr_number = int(raw_pr_number)
|
||||
issue_number = int(raw_issue_number)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
return {
|
||||
"issue_number": issue_number,
|
||||
"pr_number": pr_number,
|
||||
"branch_name": branch_name,
|
||||
"head_sha": pr_head,
|
||||
"recorded_head": pr_head,
|
||||
"accepted_head": pr_head,
|
||||
"head_relation": "equal",
|
||||
}
|
||||
|
||||
|
||||
def build_renewal_record(
|
||||
assessment: Mapping[str, Any] | None,
|
||||
*,
|
||||
|
||||
+944
-36
File diff suppressed because it is too large
Load Diff
@@ -145,6 +145,184 @@ def read_head_ancestry(
|
||||
return result
|
||||
|
||||
|
||||
def read_merge_sync_provenance(
|
||||
worktree_path: str,
|
||||
*,
|
||||
prior_head_sha: str | None,
|
||||
synced_head_sha: str | None,
|
||||
remote: str | None = None,
|
||||
) -> dict:
|
||||
"""Observe whether ``synced_head_sha`` is a sanctioned merge-sync of a base
|
||||
into the branch above ``prior_head_sha`` (#871/#872).
|
||||
|
||||
Reports server-derived git facts only; the recovery disposition lives in
|
||||
``issue_lock_recovery``. All comparisons are executed locally in the
|
||||
declared worktree -- nothing is taken from caller parameters.
|
||||
|
||||
Provenance is proven only when ALL hold:
|
||||
|
||||
* both commits are present (a rewritten/force-moved prior head leaves the
|
||||
object graph and fails closed);
|
||||
* ``prior_head_sha`` is a strict ancestor of ``synced_head_sha`` (the branch
|
||||
history is preserved, never replaced);
|
||||
* ``synced_head_sha`` is a merge commit (two or more parents), i.e. a base
|
||||
merged in — a plain fast-forward of new direct commits is not a sync;
|
||||
* ``prior_head_sha`` is an ancestor of the merge's **first** parent, so the
|
||||
branch mainline (first-parent lineage) still reaches the prior head — a
|
||||
rebase/force-push that re-authored the branch side fails this.
|
||||
"""
|
||||
path = (worktree_path or "").strip()
|
||||
prior = (prior_head_sha or "").strip()
|
||||
synced = (synced_head_sha or "").strip()
|
||||
target_remote = (remote or "").strip() or None
|
||||
result: dict = {
|
||||
"prior_head_sha": prior or None,
|
||||
"synced_head_sha": synced or None,
|
||||
"probe_ok": False,
|
||||
"prior_present": False,
|
||||
"synced_present": False,
|
||||
"prior_is_ancestor": False,
|
||||
"synced_is_merge": False,
|
||||
"first_parent_reaches_prior": False,
|
||||
"is_merge_sync": False,
|
||||
"first_parent_sha": None,
|
||||
"parent_count": None,
|
||||
"proof": None,
|
||||
"reasons": [],
|
||||
}
|
||||
if not path or not prior or not synced:
|
||||
result["reasons"].append(
|
||||
"merge-sync provenance probe requires a worktree path and both "
|
||||
"commit SHAs"
|
||||
)
|
||||
return result
|
||||
if prior == synced:
|
||||
result["reasons"].append(
|
||||
"prior and synced heads are identical; no branch sync occurred"
|
||||
)
|
||||
return result
|
||||
|
||||
def _present(sha: str) -> bool:
|
||||
res = subprocess.run(
|
||||
["git", "-C", path, "rev-parse", "--verify", "--quiet", f"{sha}^{{commit}}"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
return res.returncode == 0
|
||||
|
||||
def _is_ancestor(ancestor: str, descendant: str) -> bool | None:
|
||||
res = subprocess.run(
|
||||
["git", "-C", path, "merge-base", "--is-ancestor", ancestor, descendant],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
if res.returncode == 0:
|
||||
return True
|
||||
if res.returncode == 1:
|
||||
return False
|
||||
return None # failed probe — never a silent "no"
|
||||
|
||||
try:
|
||||
result["prior_present"] = _present(prior)
|
||||
result["synced_present"] = _present(synced)
|
||||
if not result["synced_present"] and path and os.path.isdir(path):
|
||||
if target_remote:
|
||||
subprocess.run(
|
||||
["git", "-C", path, "fetch", target_remote, "--quiet"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
result["synced_present"] = _present(synced)
|
||||
if not result["synced_present"]:
|
||||
subprocess.run(
|
||||
["git", "-C", path, "fetch", "--quiet"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
result["synced_present"] = _present(synced)
|
||||
except OSError as exc: # git unavailable — fail closed, never assume
|
||||
result["reasons"].append(f"merge-sync provenance probe could not run: {exc}")
|
||||
return result
|
||||
|
||||
if not result["prior_present"]:
|
||||
result["reasons"].append(
|
||||
f"prior head {prior} is not reachable in '{path}'; history may have "
|
||||
"been rewritten or force-moved"
|
||||
)
|
||||
if not result["synced_present"]:
|
||||
result["reasons"].append(
|
||||
f"synced head {synced} is not reachable in '{path}'"
|
||||
)
|
||||
if not (result["prior_present"] and result["synced_present"]):
|
||||
return result
|
||||
|
||||
ancestor = _is_ancestor(prior, synced)
|
||||
if ancestor is None:
|
||||
result["reasons"].append(
|
||||
"ancestry probe failed; merge-sync provenance unproven"
|
||||
)
|
||||
return result
|
||||
result["prior_is_ancestor"] = bool(ancestor)
|
||||
if not ancestor:
|
||||
result["reasons"].append(
|
||||
f"prior head {prior} is not an ancestor of synced head {synced}; "
|
||||
"the branch history was not preserved (not a merge-based sync)"
|
||||
)
|
||||
return result
|
||||
|
||||
parents_res = subprocess.run(
|
||||
["git", "-C", path, "rev-list", "--parents", "-n", "1", synced],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
if parents_res.returncode != 0:
|
||||
result["reasons"].append(
|
||||
f"could not read parents of {synced}; merge-sync provenance unproven"
|
||||
)
|
||||
return result
|
||||
tokens = (parents_res.stdout or "").split()
|
||||
# tokens[0] is the commit itself; the rest are its parents.
|
||||
parents = tokens[1:]
|
||||
result["parent_count"] = len(parents)
|
||||
result["synced_is_merge"] = len(parents) >= 2
|
||||
if not result["synced_is_merge"]:
|
||||
result["probe_ok"] = True
|
||||
result["reasons"].append(
|
||||
f"synced head {synced} has {len(parents)} parent(s); a merge-based "
|
||||
"branch sync produces a merge commit (two or more parents)"
|
||||
)
|
||||
return result
|
||||
first_parent = parents[0]
|
||||
result["first_parent_sha"] = first_parent
|
||||
|
||||
fp_reaches = _is_ancestor(prior, first_parent) if prior != first_parent else True
|
||||
if fp_reaches is None:
|
||||
result["reasons"].append(
|
||||
"first-parent ancestry probe failed; merge-sync provenance unproven"
|
||||
)
|
||||
return result
|
||||
result["first_parent_reaches_prior"] = bool(fp_reaches)
|
||||
result["probe_ok"] = True
|
||||
if not fp_reaches:
|
||||
result["reasons"].append(
|
||||
f"merge first parent {first_parent} does not reach prior head "
|
||||
f"{prior}; the branch mainline was re-authored (not a sanctioned sync)"
|
||||
)
|
||||
return result
|
||||
|
||||
result["is_merge_sync"] = True
|
||||
result["proof"] = (
|
||||
f"synced head {synced} is a merge commit (parents={len(parents)}) whose "
|
||||
f"first-parent lineage reaches prior head {prior}; base merged into branch"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def read_recorded_base(
|
||||
worktree_path: str,
|
||||
*,
|
||||
|
||||
+209
-13
@@ -39,6 +39,7 @@ SAFE_RELEASE_OWNED = "release_owned"
|
||||
SAFE_STALE_PROMPT = "stale_prompt_lease"
|
||||
SAFE_UNKNOWN = "inspect_only"
|
||||
SAFE_NO_AUTHORITY = "file_or_comment_not_authoritative"
|
||||
SAFE_CONSUME_CROSS_ROLE = "consume_cross_role_handoff"
|
||||
|
||||
LEASE_STATUS_ACTIVE = "active"
|
||||
LEASE_STATUS_RELEASED = "released"
|
||||
@@ -250,6 +251,23 @@ def decide_safe_next_action(
|
||||
"same_owner": True,
|
||||
"also_allowed": [SAFE_ABANDON_ALLOWED, SAFE_RELEASE_OWNED],
|
||||
}
|
||||
handoff = is_pending_cross_role_handoff({"lease": lease})
|
||||
if handoff:
|
||||
return {
|
||||
"safe_next_action": SAFE_CONSUME_CROSS_ROLE,
|
||||
"reasons": [
|
||||
f"controller allocation pending handoff (freshness={status}); "
|
||||
"required-role worker may consume without abandon/reassign; "
|
||||
f"required_role={handoff['required_role']}"
|
||||
],
|
||||
"block": False,
|
||||
"same_owner": False,
|
||||
"owner_session_id": owner,
|
||||
"required_role": handoff["required_role"],
|
||||
"cross_role_handoff": True,
|
||||
"handoff_status": "pending",
|
||||
"also_allowed": [SAFE_ABANDON_ALLOWED],
|
||||
}
|
||||
return {
|
||||
"safe_next_action": SAFE_ABANDON_ALLOWED,
|
||||
"reasons": [
|
||||
@@ -272,6 +290,24 @@ def decide_safe_next_action(
|
||||
}
|
||||
|
||||
if not same_owner and status == "active":
|
||||
# #843: pending cross-role handoff is consumable by required role
|
||||
handoff = is_pending_cross_role_handoff({"lease": lease})
|
||||
if handoff:
|
||||
return {
|
||||
"safe_next_action": SAFE_CONSUME_CROSS_ROLE,
|
||||
"reasons": [
|
||||
"controller cross-role allocation pending handoff; "
|
||||
f"required_role={handoff['required_role']}; "
|
||||
"consume via gitea_adopt_workflow_lease without "
|
||||
"abandonment or sharing the controller session"
|
||||
],
|
||||
"block": False,
|
||||
"same_owner": False,
|
||||
"owner_session_id": owner,
|
||||
"required_role": handoff["required_role"],
|
||||
"cross_role_handoff": True,
|
||||
"handoff_status": "pending",
|
||||
}
|
||||
return {
|
||||
"safe_next_action": SAFE_WAIT_FOREIGN,
|
||||
"reasons": [
|
||||
@@ -440,6 +476,84 @@ def list_active_leases(
|
||||
}
|
||||
|
||||
|
||||
|
||||
def parse_lease_provenance(lease_or_state: Mapping[str, Any] | None) -> dict[str, Any]:
|
||||
"""Return durable lease provenance dict (empty when absent/unparseable)."""
|
||||
if not lease_or_state:
|
||||
return {}
|
||||
if "provenance" in lease_or_state and isinstance(lease_or_state.get("provenance"), dict):
|
||||
return dict(lease_or_state["provenance"])
|
||||
raw = None
|
||||
if "provenance_json" in lease_or_state:
|
||||
raw = lease_or_state.get("provenance_json")
|
||||
elif "lease" in lease_or_state and isinstance(lease_or_state.get("lease"), Mapping):
|
||||
raw = lease_or_state["lease"].get("provenance_json")
|
||||
if not raw:
|
||||
return {}
|
||||
if isinstance(raw, dict):
|
||||
return dict(raw)
|
||||
try:
|
||||
loaded = json.loads(raw)
|
||||
except (TypeError, json.JSONDecodeError):
|
||||
return {}
|
||||
return dict(loaded) if isinstance(loaded, dict) else {}
|
||||
|
||||
|
||||
def is_pending_cross_role_handoff(
|
||||
state: Mapping[str, Any] | None,
|
||||
) -> dict[str, Any] | None:
|
||||
"""Return handoff evidence when a controller allocation awaits consume (#843).
|
||||
|
||||
A pending handoff is identified by durable provenance written at
|
||||
cross-role apply time — not by title heuristics or session-id guessing.
|
||||
"""
|
||||
if not state:
|
||||
return None
|
||||
lease = state.get("lease") if isinstance(state.get("lease"), Mapping) else state
|
||||
if not isinstance(lease, Mapping):
|
||||
return None
|
||||
status = str(lease.get("status") or "").strip().lower()
|
||||
if status in (LEASE_STATUS_ABANDONED, LEASE_STATUS_RELEASED, LEASE_STATUS_EXPIRED):
|
||||
return None
|
||||
prov = parse_lease_provenance(state)
|
||||
if not prov and isinstance(lease, Mapping):
|
||||
prov = parse_lease_provenance(lease)
|
||||
if not prov.get("cross_role_handoff"):
|
||||
return None
|
||||
handoff_status = str(prov.get("handoff_status") or "pending").strip().lower()
|
||||
if handoff_status != "pending":
|
||||
return None
|
||||
adopted_by = (
|
||||
lease.get("adopted_by_session_id")
|
||||
or prov.get("adopted_by_session_id")
|
||||
or ""
|
||||
)
|
||||
if str(adopted_by).strip():
|
||||
return None
|
||||
required_role = str(
|
||||
prov.get("required_role") or lease.get("role") or ""
|
||||
).strip().lower()
|
||||
if not required_role:
|
||||
return None
|
||||
return {
|
||||
"cross_role_handoff": True,
|
||||
"handoff_status": "pending",
|
||||
"required_role": required_role,
|
||||
"allocating_session_id": str(
|
||||
prov.get("allocating_session_id") or lease.get("session_id") or ""
|
||||
),
|
||||
"allocating_role": str(prov.get("allocating_role") or "controller"),
|
||||
"lease_id": str(lease.get("lease_id") or ""),
|
||||
"assignment_id": (
|
||||
str(state["assignment"]["assignment_id"])
|
||||
if isinstance(state.get("assignment"), Mapping)
|
||||
and state["assignment"].get("assignment_id")
|
||||
else None
|
||||
),
|
||||
"provenance": prov,
|
||||
}
|
||||
|
||||
|
||||
def adopt_lease(
|
||||
db: cpd.ControlPlaneDB,
|
||||
*,
|
||||
@@ -450,8 +564,17 @@ def adopt_lease(
|
||||
expected_head_sha: str | None = None,
|
||||
owner_pid: int | None = None,
|
||||
operator_authorized: bool = False,
|
||||
adopter_profile_name: str | None = None,
|
||||
adopter_namespace: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Sanctioned adopt path with provenance; never silent foreign steal."""
|
||||
"""Sanctioned adopt path with provenance; never silent foreign steal.
|
||||
|
||||
#843 F1: for a pending cross-role handoff, ``role`` must be the
|
||||
authoritative profile-derived role supplied by the MCP boundary — never
|
||||
caller-asserted authority. When the handoff provenance declares
|
||||
``required_profile`` / ``required_namespace`` and the caller context is
|
||||
provided, both are validated exactly; a mismatch fails closed.
|
||||
"""
|
||||
state = db.get_lease_workflow_state(lease_id)
|
||||
if not state:
|
||||
raise LeaseLifecycleError(
|
||||
@@ -463,11 +586,8 @@ def adopt_lease(
|
||||
owner = str(lease.get("session_id") or "")
|
||||
same_owner = owner == str(adopter_session_id)
|
||||
|
||||
if freshness["freshness"] == "active" and not same_owner:
|
||||
raise LeaseLifecycleError(
|
||||
f"refusing to steal active foreign lease {lease_id} owned by "
|
||||
f"{owner} (fail closed)"
|
||||
)
|
||||
handoff = is_pending_cross_role_handoff(state)
|
||||
adopter_role = (role or "").strip().lower()
|
||||
|
||||
if freshness["freshness"] in ("abandoned", "released"):
|
||||
raise LeaseLifecycleError(
|
||||
@@ -475,13 +595,64 @@ def adopt_lease(
|
||||
"(fail closed)"
|
||||
)
|
||||
|
||||
# Expired or stale: require abandon-style safety before ownership transfer
|
||||
# when not same owner; same owner may reclaim.
|
||||
if not same_owner and freshness["freshness"] in (
|
||||
if handoff and not same_owner:
|
||||
# Terminal statuses already rejected above. Freshness may be
|
||||
# active OR stale_dead_process (controller exited) — both are
|
||||
# consumable without abandonment when handoff is still pending.
|
||||
if freshness["freshness"] not in (
|
||||
"active",
|
||||
"stale_dead_process",
|
||||
"stale_missing_worktree",
|
||||
):
|
||||
raise LeaseLifecycleError(
|
||||
f"lease {lease_id} freshness={freshness['freshness']}; "
|
||||
"terminal or non-active allocation cannot be handoff-consumed "
|
||||
"(fail closed)"
|
||||
)
|
||||
required = handoff["required_role"]
|
||||
if adopter_role != required:
|
||||
raise LeaseLifecycleError(
|
||||
f"wrong role for cross-role handoff consume of {lease_id}: "
|
||||
f"required={required} adopter={adopter_role or 'none'} "
|
||||
"(fail closed)"
|
||||
)
|
||||
# #843 F1: provenance profile/namespace restrictions are validated
|
||||
# against the authoritative caller context when declared. Caller
|
||||
# input can never widen authority; a mismatch fails closed.
|
||||
handoff_prov = handoff.get("provenance") or {}
|
||||
required_profile = str(
|
||||
handoff_prov.get("required_profile") or ""
|
||||
).strip()
|
||||
if required_profile and adopter_profile_name is not None:
|
||||
if str(adopter_profile_name).strip() != required_profile:
|
||||
raise LeaseLifecycleError(
|
||||
f"wrong profile for cross-role handoff consume of "
|
||||
f"{lease_id}: required_profile={required_profile} "
|
||||
f"adopter_profile={adopter_profile_name} (fail closed)"
|
||||
)
|
||||
required_namespace = str(
|
||||
handoff_prov.get("required_namespace") or ""
|
||||
).strip()
|
||||
if required_namespace and adopter_namespace is not None:
|
||||
if str(adopter_namespace).strip() != required_namespace:
|
||||
raise LeaseLifecycleError(
|
||||
f"wrong namespace for cross-role handoff consume of "
|
||||
f"{lease_id}: required_namespace={required_namespace} "
|
||||
f"adopter_namespace={adopter_namespace} (fail closed)"
|
||||
)
|
||||
reason = "cross-role-handoff-consume"
|
||||
elif freshness["freshness"] == "active" and not same_owner:
|
||||
raise LeaseLifecycleError(
|
||||
f"refusing to steal active foreign lease {lease_id} owned by "
|
||||
f"{owner} (fail closed)"
|
||||
)
|
||||
elif not same_owner and freshness["freshness"] in (
|
||||
"expired",
|
||||
"stale_dead_process",
|
||||
"stale_missing_worktree",
|
||||
):
|
||||
# Expired or stale (non-handoff): require abandon-style safety before
|
||||
# ownership transfer when not same owner; same owner may reclaim.
|
||||
if not operator_authorized and freshness["freshness"] == "expired":
|
||||
# Deterministic reclaim of expired foreign lease is allowed
|
||||
# without operator flag (sanctioned expire reclaim).
|
||||
@@ -492,6 +663,9 @@ def adopt_lease(
|
||||
f"lease {lease_id} freshness={freshness['freshness']}; "
|
||||
"use abandon with proof before foreign adopt (fail closed)"
|
||||
)
|
||||
reason = "sanctioned-reclaim-adopt"
|
||||
else:
|
||||
reason = "owner-resume-adopt" if same_owner else "sanctioned-reclaim-adopt"
|
||||
|
||||
provenance = build_adopt_provenance(
|
||||
adopted_from_session_id=owner,
|
||||
@@ -504,10 +678,14 @@ def adopt_lease(
|
||||
worktree_path=worktree_path,
|
||||
expected_head_sha=expected_head_sha or lease.get("expected_head_sha"),
|
||||
prior_lease_id=lease_id,
|
||||
reason=(
|
||||
"owner-resume-adopt" if same_owner else "sanctioned-reclaim-adopt"
|
||||
),
|
||||
reason=reason,
|
||||
)
|
||||
if handoff and not same_owner:
|
||||
provenance["cross_role_handoff"] = True
|
||||
provenance["handoff_status"] = "adopted"
|
||||
provenance["required_role"] = handoff["required_role"]
|
||||
provenance["allocating_session_id"] = handoff["allocating_session_id"]
|
||||
provenance["allocating_role"] = handoff["allocating_role"]
|
||||
|
||||
result = db.adopt_lease(
|
||||
lease_id=lease_id,
|
||||
@@ -518,7 +696,7 @@ def adopt_lease(
|
||||
owner_pid=owner_pid if owner_pid is not None else os.getpid(),
|
||||
provenance=provenance,
|
||||
)
|
||||
return {
|
||||
out = {
|
||||
"success": True,
|
||||
"outcome": result.get("outcome"),
|
||||
"same_owner": same_owner,
|
||||
@@ -531,6 +709,24 @@ def adopt_lease(
|
||||
"comment_lease_only": False,
|
||||
"reasons": result.get("reasons") or [],
|
||||
}
|
||||
if handoff and not same_owner:
|
||||
out["cross_role_handoff"] = True
|
||||
out["handoff_status"] = "adopted"
|
||||
out["required_role"] = handoff["required_role"]
|
||||
out["adopted_by_session_id"] = adopter_session_id
|
||||
out["adopted_from_session_id"] = owner
|
||||
lease_row = result.get("lease") or {}
|
||||
if isinstance(lease_row, Mapping):
|
||||
out["read_after_write"] = {
|
||||
"lease_id": lease_row.get("lease_id"),
|
||||
"session_id": lease_row.get("session_id"),
|
||||
"role": lease_row.get("role"),
|
||||
"status": lease_row.get("status"),
|
||||
"adopted_by_session_id": lease_row.get("adopted_by_session_id"),
|
||||
"adopted_from_session_id": lease_row.get("adopted_from_session_id"),
|
||||
"phase": lease_row.get("phase"),
|
||||
}
|
||||
return out
|
||||
|
||||
|
||||
def release_lease(
|
||||
|
||||
+212
@@ -0,0 +1,212 @@
|
||||
"""Central lease policy configuration (#790 Slice A, AC-N7).
|
||||
|
||||
The single authoritative source for every lease duration in the project. Before
|
||||
this module the numbers were scattered: a four-hour author TTL was declared
|
||||
twice (``issue_lock_store`` and ``gitea_mcp_server``), the reviewer/merger
|
||||
sliding window lived in ``reviewer_pr_lease``, the conflict-fix window in
|
||||
``pr_work_lease``, and the control-plane default in ``control_plane_db``.
|
||||
Nothing tied them together, so tuning one class silently diverged from the
|
||||
others and no reader could answer "how long does a lease live?" without
|
||||
grepping four files.
|
||||
|
||||
AC-N7 requires that this configuration exist *before* the first heartbeat and
|
||||
TTL behavior that reads from it, so it ships in Slice A rather than trailing the
|
||||
code it governs.
|
||||
|
||||
Deliberate boundaries:
|
||||
|
||||
* **Declaration is not rewiring.** Every task class is declared here, but only
|
||||
those with ``heartbeat_lifecycle_active`` were migrated onto the shared
|
||||
heartbeat lifecycle in Slice A — currently ``author_issue_work`` alone.
|
||||
Reviewer, merger, and conflict-fix leases keep their own existing behavior
|
||||
until Slice C moves them; their numbers are recorded here so the two cannot
|
||||
drift apart unnoticed, and ``tests/test_issue_790_lease_policy.py`` asserts
|
||||
the recorded values still equal the constants those modules use.
|
||||
* **No policy decision lives here.** This module answers "how long", never "may
|
||||
this session proceed". Freshness, reclaim, and renewal dispositions stay in
|
||||
``issue_lock_store``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
# Task classes. Only the first is migrated onto the shared lifecycle in Slice A.
|
||||
TASK_CLASS_AUTHOR_ISSUE_WORK = "author_issue_work"
|
||||
TASK_CLASS_REVIEWER_PR = "reviewer_pr"
|
||||
TASK_CLASS_MERGER_PR = "merger_pr"
|
||||
TASK_CLASS_CONFLICT_FIX = "conflict_fix"
|
||||
|
||||
# Durable marker for a lease minted under the shared heartbeat lifecycle.
|
||||
#
|
||||
# #790 AC-N8: this explicit marker — never a timestamp comparison — is what
|
||||
# distinguishes a heartbeat-lifecycle lease from a legacy one. A lock written
|
||||
# before this lifecycle existed carries no marker and reads as
|
||||
# ``LIFECYCLE_LEGACY``.
|
||||
LIFECYCLE_HEARTBEAT_V1 = "heartbeat-v1"
|
||||
LIFECYCLE_LEGACY = "legacy"
|
||||
|
||||
_ENV_PREFIX = "GITEA_LEASE_POLICY"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LeasePolicy:
|
||||
"""Durations governing one task class.
|
||||
|
||||
All intervals are minutes except ``absolute_cap_hours``. ``None`` for the
|
||||
cap means the class has no maximum continuous duration.
|
||||
"""
|
||||
|
||||
task_class: str
|
||||
initial_ttl_minutes: float
|
||||
heartbeat_cadence_minutes: float
|
||||
stale_warning_minutes: float
|
||||
missed_heartbeat_grace_minutes: float
|
||||
absolute_cap_hours: float | None
|
||||
recovery_grace_minutes: float
|
||||
terminal_race_drain_minutes: float
|
||||
terminal_retirement_eligible: bool
|
||||
heartbeat_lifecycle_active: bool
|
||||
|
||||
|
||||
# Defaults. ``author_issue_work`` adopts the reviewer window proven by #747
|
||||
# rather than inventing new numbers: a lease expires 10 minutes after its last
|
||||
# valid heartbeat, warns at half that, and an actively heartbeating session is
|
||||
# never evicted. The prior value was a fixed four hours (240 minutes) that no
|
||||
# heartbeat could shorten — the defect this issue exists to correct.
|
||||
_DEFAULTS: dict[str, LeasePolicy] = {
|
||||
TASK_CLASS_AUTHOR_ISSUE_WORK: LeasePolicy(
|
||||
task_class=TASK_CLASS_AUTHOR_ISSUE_WORK,
|
||||
initial_ttl_minutes=10.0,
|
||||
heartbeat_cadence_minutes=2.0,
|
||||
stale_warning_minutes=5.0,
|
||||
missed_heartbeat_grace_minutes=10.0,
|
||||
absolute_cap_hours=8.0,
|
||||
recovery_grace_minutes=10.0,
|
||||
terminal_race_drain_minutes=2.0,
|
||||
terminal_retirement_eligible=True,
|
||||
heartbeat_lifecycle_active=True,
|
||||
),
|
||||
# Declared, not rewired. These mirror reviewer_pr_lease.LEASE_TTL_MINUTES
|
||||
# and STALE_WARNING_MINUTES; Slice C migrates the call sites.
|
||||
TASK_CLASS_REVIEWER_PR: LeasePolicy(
|
||||
task_class=TASK_CLASS_REVIEWER_PR,
|
||||
initial_ttl_minutes=10.0,
|
||||
heartbeat_cadence_minutes=2.0,
|
||||
stale_warning_minutes=5.0,
|
||||
missed_heartbeat_grace_minutes=10.0,
|
||||
absolute_cap_hours=None,
|
||||
recovery_grace_minutes=10.0,
|
||||
terminal_race_drain_minutes=2.0,
|
||||
terminal_retirement_eligible=False,
|
||||
heartbeat_lifecycle_active=False,
|
||||
),
|
||||
TASK_CLASS_MERGER_PR: LeasePolicy(
|
||||
task_class=TASK_CLASS_MERGER_PR,
|
||||
initial_ttl_minutes=10.0,
|
||||
heartbeat_cadence_minutes=2.0,
|
||||
stale_warning_minutes=5.0,
|
||||
missed_heartbeat_grace_minutes=10.0,
|
||||
absolute_cap_hours=None,
|
||||
recovery_grace_minutes=10.0,
|
||||
terminal_race_drain_minutes=2.0,
|
||||
terminal_retirement_eligible=False,
|
||||
heartbeat_lifecycle_active=False,
|
||||
),
|
||||
# Mirrors pr_work_lease.DEFAULT_CONFLICT_FIX_TTL_MINUTES. Deliberately left
|
||||
# at its current window; shortening it is Slice C's call, not this slice's.
|
||||
TASK_CLASS_CONFLICT_FIX: LeasePolicy(
|
||||
task_class=TASK_CLASS_CONFLICT_FIX,
|
||||
initial_ttl_minutes=120.0,
|
||||
heartbeat_cadence_minutes=2.0,
|
||||
stale_warning_minutes=5.0,
|
||||
missed_heartbeat_grace_minutes=10.0,
|
||||
absolute_cap_hours=None,
|
||||
recovery_grace_minutes=10.0,
|
||||
terminal_race_drain_minutes=2.0,
|
||||
terminal_retirement_eligible=False,
|
||||
heartbeat_lifecycle_active=False,
|
||||
),
|
||||
}
|
||||
|
||||
_NUMERIC_FIELDS = (
|
||||
"initial_ttl_minutes",
|
||||
"heartbeat_cadence_minutes",
|
||||
"stale_warning_minutes",
|
||||
"missed_heartbeat_grace_minutes",
|
||||
"absolute_cap_hours",
|
||||
"recovery_grace_minutes",
|
||||
"terminal_race_drain_minutes",
|
||||
)
|
||||
|
||||
|
||||
def env_var_name(task_class: str, field: str) -> str:
|
||||
"""Environment variable that overrides one field of one task class."""
|
||||
return f"{_ENV_PREFIX}_{task_class.upper()}_{field.upper()}"
|
||||
|
||||
|
||||
def _override(task_class: str, field: str, default: float | None) -> float | None:
|
||||
"""Read one override, falling back to *default* on anything unusable.
|
||||
|
||||
A malformed or non-positive override is ignored rather than raised: a typo
|
||||
in an environment variable must not be able to mint a zero-length lease that
|
||||
makes every claim instantly reclaimable, nor crash the server at import.
|
||||
"""
|
||||
raw = (os.environ.get(env_var_name(task_class, field)) or "").strip()
|
||||
if not raw:
|
||||
return default
|
||||
try:
|
||||
value = float(raw)
|
||||
except (TypeError, ValueError):
|
||||
return default
|
||||
if value <= 0:
|
||||
return default
|
||||
return value
|
||||
|
||||
|
||||
def policy_for(task_class: str) -> LeasePolicy:
|
||||
"""Return the effective policy for *task_class*.
|
||||
|
||||
Unknown task classes fall back to the author policy, which is the most
|
||||
conservative migrated class, rather than raising — a new caller must never
|
||||
be able to crash a lock write by naming a class this table has not learned.
|
||||
"""
|
||||
key = str(task_class or "").strip() or TASK_CLASS_AUTHOR_ISSUE_WORK
|
||||
base = _DEFAULTS.get(key) or _DEFAULTS[TASK_CLASS_AUTHOR_ISSUE_WORK]
|
||||
resolved = {
|
||||
field: _override(base.task_class, field, getattr(base, field))
|
||||
for field in _NUMERIC_FIELDS
|
||||
}
|
||||
if all(resolved[field] == getattr(base, field) for field in _NUMERIC_FIELDS):
|
||||
return base
|
||||
return LeasePolicy(
|
||||
task_class=base.task_class,
|
||||
terminal_retirement_eligible=base.terminal_retirement_eligible,
|
||||
heartbeat_lifecycle_active=base.heartbeat_lifecycle_active,
|
||||
**resolved,
|
||||
)
|
||||
|
||||
|
||||
def known_task_classes() -> tuple[str, ...]:
|
||||
"""Every declared task class, migrated or not."""
|
||||
return tuple(_DEFAULTS)
|
||||
|
||||
|
||||
def describe(task_class: str) -> dict[str, Any]:
|
||||
"""Serializable view of a policy, for audit records and tool payloads."""
|
||||
policy = policy_for(task_class)
|
||||
return {
|
||||
"task_class": policy.task_class,
|
||||
"initial_ttl_minutes": policy.initial_ttl_minutes,
|
||||
"heartbeat_cadence_minutes": policy.heartbeat_cadence_minutes,
|
||||
"stale_warning_minutes": policy.stale_warning_minutes,
|
||||
"missed_heartbeat_grace_minutes": policy.missed_heartbeat_grace_minutes,
|
||||
"absolute_cap_hours": policy.absolute_cap_hours,
|
||||
"recovery_grace_minutes": policy.recovery_grace_minutes,
|
||||
"terminal_race_drain_minutes": policy.terminal_race_drain_minutes,
|
||||
"terminal_retirement_eligible": policy.terminal_retirement_eligible,
|
||||
"heartbeat_lifecycle_active": policy.heartbeat_lifecycle_active,
|
||||
"lifecycle_version": LIFECYCLE_HEARTBEAT_V1,
|
||||
}
|
||||
@@ -0,0 +1,328 @@
|
||||
"""Sanctioned MCP client reconnect request surface for Codex/LLM sessions (#678).
|
||||
|
||||
Codex and other agent hosts can detect stale or closed Gitea MCP runtimes, but
|
||||
the host owns the transport. This module never restarts, kills, or reloads a
|
||||
daemon. It builds:
|
||||
|
||||
1. A **callable reconnect request** result agents can invoke via
|
||||
``gitea_request_mcp_reconnect`` (report-only, side-effect free).
|
||||
2. A **typed blocker** with exact operator UI steps when recovery must be
|
||||
performed by the host/operator.
|
||||
|
||||
Forbidden recovery paths (must never be recommended):
|
||||
|
||||
* ``pkill`` / ``kill`` / ``killall`` of MCP daemons
|
||||
* ``touch`` / mtime config reload hacks
|
||||
* ``.env`` or MCP config edits as recovery
|
||||
* session-state file edits
|
||||
* raw Gitea API / direct server-import fallbacks
|
||||
|
||||
After the operator reconnects, workflows restart from identity / runtime /
|
||||
capability preflight (``gitea_whoami`` → ``gitea_resolve_task_capability`` →
|
||||
task).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Mapping
|
||||
|
||||
# --- Reason vocabulary -------------------------------------------------------
|
||||
|
||||
REASON_STALE_RUNTIME = "stale-runtime"
|
||||
REASON_TRANSPORT_EOF = "transport_eof"
|
||||
REASON_MISSING_NAMESPACE = "missing_namespace"
|
||||
REASON_NOT_REQUIRED = "not_required"
|
||||
REASON_UNSPECIFIED = "unspecified"
|
||||
|
||||
VALID_REASONS = frozenset(
|
||||
{
|
||||
REASON_STALE_RUNTIME,
|
||||
REASON_TRANSPORT_EOF,
|
||||
REASON_MISSING_NAMESPACE,
|
||||
REASON_NOT_REQUIRED,
|
||||
REASON_UNSPECIFIED,
|
||||
}
|
||||
)
|
||||
|
||||
# Boundary statuses reported to callers (match review_workflow_boundary style).
|
||||
BOUNDARY_CLEAN = "clean"
|
||||
BOUNDARY_MISMATCH = "mismatch"
|
||||
BOUNDARY_STALE = "stale"
|
||||
BOUNDARY_UNKNOWN = "unknown"
|
||||
|
||||
# Typed blocker kinds
|
||||
BLOCKER_OPERATOR_RECONNECT = "operator_mcp_reconnect_required"
|
||||
BLOCKER_NONE = "none"
|
||||
|
||||
FORBIDDEN_RECOVERY_PATHS: tuple[str, ...] = (
|
||||
"pkill / kill / killall of mcp_server.py, gitea_mcp_server, or broad python sweeps",
|
||||
"touch / mtime-based MCP config reload hacks",
|
||||
".env edits as recovery",
|
||||
"MCP config file edits as recovery",
|
||||
"session-state file edits as recovery",
|
||||
"raw Gitea API or direct MCP server-import fallbacks",
|
||||
)
|
||||
|
||||
# Client-specific operator UI steps. Keep Codex first (issue title surface).
|
||||
OPERATOR_UI_STEPS: dict[str, tuple[str, ...]] = {
|
||||
"codex": (
|
||||
"In Codex, open the MCP / Developer tools panel for this workspace.",
|
||||
"Locate the named Gitea MCP server entry (namespace) that needs reconnect "
|
||||
"(e.g. gitea-author, gitea-reviewer, gitea-merger, gitea-tools, "
|
||||
"gitea-controller, gitea-reconciler).",
|
||||
"Click 'Reload Developer Tools' or the server reconnect/reload control "
|
||||
"for that entry so the client spawns a fresh MCP subprocess.",
|
||||
"If per-server reconnect is unavailable, fully restart the Codex client "
|
||||
"(quit and relaunch) so all MCP namespaces reattach.",
|
||||
"After reconnect, rerun the blocked workflow from preflight: "
|
||||
"gitea_whoami → gitea_resolve_task_capability → the original task. "
|
||||
"Do not resume mid-mutation.",
|
||||
),
|
||||
"claude_code": (
|
||||
"Run `/mcp` (or open the MCP servers UI) in Claude Code.",
|
||||
"Reconnect the affected gitea-* server entry so the client reopens stdio.",
|
||||
"If reconnect fails, relaunch the Claude Code session entirely.",
|
||||
"After reconnect, restart the workflow from gitea_whoami → "
|
||||
"gitea_resolve_task_capability → task.",
|
||||
),
|
||||
"generic": (
|
||||
"Use the host/IDE MCP reconnect or reload control for the named namespace.",
|
||||
"If no per-namespace control exists, restart the MCP client/editor.",
|
||||
"After reconnect, restart the workflow from identity/capability preflight.",
|
||||
),
|
||||
}
|
||||
|
||||
DEFAULT_CLIENT = "codex"
|
||||
|
||||
|
||||
def normalize_reason(reason: str | None) -> str:
|
||||
"""Map free-form reason strings onto the closed vocabulary."""
|
||||
raw = (reason or "").strip().lower()
|
||||
if not raw:
|
||||
return REASON_UNSPECIFIED
|
||||
if raw in VALID_REASONS:
|
||||
return raw
|
||||
text = raw.replace(" ", "_").replace("-", "_")
|
||||
aliases = {
|
||||
"stale_runtime": REASON_STALE_RUNTIME,
|
||||
"staleruntime": REASON_STALE_RUNTIME,
|
||||
"runtime_stale": REASON_STALE_RUNTIME,
|
||||
"stale": REASON_STALE_RUNTIME,
|
||||
"transport_eof": REASON_TRANSPORT_EOF,
|
||||
"transport_closed": REASON_TRANSPORT_EOF,
|
||||
"eof": REASON_TRANSPORT_EOF,
|
||||
"client_is_closing": REASON_TRANSPORT_EOF,
|
||||
"missing_namespace": REASON_MISSING_NAMESPACE,
|
||||
"namespace_missing": REASON_MISSING_NAMESPACE,
|
||||
"not_required": REASON_NOT_REQUIRED,
|
||||
"healthy": REASON_NOT_REQUIRED,
|
||||
"ok": REASON_NOT_REQUIRED,
|
||||
"unspecified": REASON_UNSPECIFIED,
|
||||
}
|
||||
if text in aliases:
|
||||
return aliases[text]
|
||||
hyphenated = text.replace("_", "-")
|
||||
if hyphenated in VALID_REASONS:
|
||||
return hyphenated
|
||||
return REASON_UNSPECIFIED
|
||||
|
||||
|
||||
def normalize_client(client: str | None) -> str:
|
||||
"""Return a known client key for operator UI steps."""
|
||||
text = (client or "").strip().lower().replace(" ", "_").replace("-", "_")
|
||||
if text in ("codex", "openai_codex", "openai"):
|
||||
return "codex"
|
||||
if text in ("claude", "claude_code", "claude_desktop", "anthropic"):
|
||||
return "claude_code"
|
||||
if text in OPERATOR_UI_STEPS:
|
||||
return text
|
||||
return DEFAULT_CLIENT
|
||||
|
||||
|
||||
def classify_boundary_status(
|
||||
*,
|
||||
startup_sha: str | None,
|
||||
current_master_sha: str | None,
|
||||
live_stale: bool | None = None,
|
||||
in_parity: bool | None = None,
|
||||
) -> str:
|
||||
"""Derive boundary_status from parity evidence."""
|
||||
if live_stale is True or in_parity is False:
|
||||
return BOUNDARY_STALE
|
||||
start = (startup_sha or "").strip().lower()
|
||||
current = (current_master_sha or "").strip().lower()
|
||||
if start and current and start != current:
|
||||
return BOUNDARY_MISMATCH
|
||||
if start and current and start == current:
|
||||
return BOUNDARY_CLEAN
|
||||
if in_parity is True:
|
||||
return BOUNDARY_CLEAN
|
||||
return BOUNDARY_UNKNOWN
|
||||
|
||||
|
||||
def operator_ui_steps(client: str | None, *, namespace: str | None = None) -> list[str]:
|
||||
"""Exact operator UI steps for the named client."""
|
||||
key = normalize_client(client)
|
||||
steps = list(OPERATOR_UI_STEPS.get(key) or OPERATOR_UI_STEPS[DEFAULT_CLIENT])
|
||||
ns = (namespace or "").strip()
|
||||
if ns:
|
||||
steps = [
|
||||
s.replace("named Gitea MCP server entry (namespace)", f"namespace '{ns}'")
|
||||
.replace("affected gitea-* server entry", f"server entry '{ns}'")
|
||||
.replace("named namespace", f"namespace '{ns}'")
|
||||
for s in steps
|
||||
]
|
||||
return steps
|
||||
|
||||
|
||||
def build_reconnect_request(
|
||||
*,
|
||||
namespace: str,
|
||||
profile: str | None = None,
|
||||
pid: int | str | None = None,
|
||||
session_id: str | None = None,
|
||||
startup_sha: str | None = None,
|
||||
current_master_sha: str | None = None,
|
||||
boundary_status: str | None = None,
|
||||
reason: str | None = None,
|
||||
client: str | None = DEFAULT_CLIENT,
|
||||
live_stale: bool | None = None,
|
||||
in_parity: bool | None = None,
|
||||
restart_required: bool | None = None,
|
||||
stop_required: bool | None = None,
|
||||
extra: Mapping[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build the structured reconnect-request / typed-blocker payload (#678).
|
||||
|
||||
Never mutates process, config, or session state. Always side-effect free.
|
||||
"""
|
||||
ns = (namespace or "").strip() or "unknown"
|
||||
normalized_reason = normalize_reason(reason)
|
||||
boundary = (boundary_status or "").strip() or classify_boundary_status(
|
||||
startup_sha=startup_sha,
|
||||
current_master_sha=current_master_sha,
|
||||
live_stale=live_stale,
|
||||
in_parity=in_parity,
|
||||
)
|
||||
|
||||
reconnect_needed = True
|
||||
if normalized_reason == REASON_NOT_REQUIRED and boundary == BOUNDARY_CLEAN:
|
||||
reconnect_needed = False
|
||||
if restart_required is False and stop_required is False and boundary == BOUNDARY_CLEAN:
|
||||
# Explicit healthy probe
|
||||
if normalized_reason in (REASON_NOT_REQUIRED, REASON_UNSPECIFIED):
|
||||
reconnect_needed = False
|
||||
normalized_reason = REASON_NOT_REQUIRED
|
||||
|
||||
if restart_required is True or stop_required is True:
|
||||
reconnect_needed = True
|
||||
if normalized_reason in (REASON_NOT_REQUIRED, REASON_UNSPECIFIED):
|
||||
normalized_reason = REASON_STALE_RUNTIME
|
||||
|
||||
client_key = normalize_client(client)
|
||||
steps = operator_ui_steps(client_key, namespace=ns)
|
||||
|
||||
result: dict[str, Any] = {
|
||||
"success": True,
|
||||
"read_only": True,
|
||||
"reconnect_performed": False,
|
||||
"mutation_performed": False,
|
||||
"reconnect_needed": reconnect_needed,
|
||||
"namespace": ns,
|
||||
"profile": (profile or "").strip() or None,
|
||||
"pid": pid,
|
||||
"session_id": (session_id or "").strip() or None,
|
||||
"startup_sha": (startup_sha or "").strip() or None,
|
||||
"current_master_sha": (current_master_sha or "").strip() or None,
|
||||
"boundary_status": boundary,
|
||||
"reason": normalized_reason,
|
||||
"client": client_key,
|
||||
"forbidden_recovery_paths": list(FORBIDDEN_RECOVERY_PATHS),
|
||||
"post_reconnect_preflight": [
|
||||
"gitea_whoami",
|
||||
"gitea_resolve_task_capability",
|
||||
"original_task",
|
||||
],
|
||||
"exact_safe_next_action": None,
|
||||
"blocker_kind": BLOCKER_NONE,
|
||||
"operator_ui_steps": steps,
|
||||
"typed_blocker": None,
|
||||
}
|
||||
|
||||
if reconnect_needed:
|
||||
result["blocker_kind"] = BLOCKER_OPERATOR_RECONNECT
|
||||
result["stop_required"] = True
|
||||
result["restart_required"] = True
|
||||
result["exact_safe_next_action"] = (
|
||||
f"blocker_kind={BLOCKER_OPERATOR_RECONNECT}: operator must reconnect "
|
||||
f"MCP namespace '{ns}' via the host UI (client={client_key}). "
|
||||
"Do not pkill, touch configs, edit session state, or use raw API. "
|
||||
"After reconnect, restart from gitea_whoami → "
|
||||
"gitea_resolve_task_capability → task."
|
||||
)
|
||||
result["typed_blocker"] = {
|
||||
"blocker_kind": BLOCKER_OPERATOR_RECONNECT,
|
||||
"namespaces": [ns],
|
||||
"why_reconnect_required": normalized_reason,
|
||||
"operator_ui_steps": steps,
|
||||
"client": client_key,
|
||||
"forbidden_recovery_paths": list(FORBIDDEN_RECOVERY_PATHS),
|
||||
"instruction_after_reconnect": (
|
||||
"Rerun the blocked workflow from preflight "
|
||||
"(gitea_whoami → gitea_resolve_task_capability → task). "
|
||||
"Do not continue mid-mutation from pre-reconnect state."
|
||||
),
|
||||
}
|
||||
else:
|
||||
result["stop_required"] = False
|
||||
result["restart_required"] = False
|
||||
result["exact_safe_next_action"] = (
|
||||
f"Reconnect not required for namespace '{ns}' "
|
||||
f"(boundary_status={boundary}). Proceed with the original task."
|
||||
)
|
||||
|
||||
if extra:
|
||||
for key, value in extra.items():
|
||||
if key not in result:
|
||||
result[key] = value
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def reasons_never_suggest_forbidden(text: str) -> bool:
|
||||
"""Return True when *text* does not recommend a forbidden recovery path.
|
||||
|
||||
Mentions that *ban* a path (e.g. ``Do not pkill`` / ``never edit session
|
||||
state``) are allowed. Positive recommendations such as ``use pkill`` or
|
||||
``run killall`` fail.
|
||||
"""
|
||||
import re
|
||||
|
||||
lowered = (text or "").lower()
|
||||
# Strip common ban prefixes so "do not pkill" does not trip positive checks.
|
||||
scrubbed = re.sub(
|
||||
r"\b(?:do not|don't|never|must not|forbid(?:den)?|ban(?:ned)?)\b"
|
||||
r"[^.!;\n]{0,80}",
|
||||
" ",
|
||||
lowered,
|
||||
)
|
||||
# Positive imperative / advisory forms that would tell an agent to do harm.
|
||||
positive_suggestions = (
|
||||
"use pkill",
|
||||
"run pkill",
|
||||
"try pkill",
|
||||
"pkill -f",
|
||||
"use killall",
|
||||
"run killall",
|
||||
"killall mcp",
|
||||
"use kill ",
|
||||
"run kill ",
|
||||
"touch the mcp",
|
||||
"touch mcp config",
|
||||
"utime(",
|
||||
"edit the mcp config to recover",
|
||||
"edit .env to recover",
|
||||
"import gitea_mcp_server",
|
||||
"python -c 'import gitea_mcp",
|
||||
)
|
||||
return not any(frag in scrubbed for frag in positive_suggestions)
|
||||
@@ -0,0 +1,237 @@
|
||||
"""Antigravity IDE vs Global MCP Config Drift Diagnostic (#672).
|
||||
|
||||
Diagnoses config drift between the active IDE MCP configuration
|
||||
(e.g. ``~/.gemini/antigravity-ide/mcp_config.json``) and the offline/global
|
||||
canonical configuration (e.g. ``~/.gemini/config/mcp_config.json``).
|
||||
|
||||
Hard rules (#672 / #630 / #655):
|
||||
* Distinguish offline/global success from active IDE namespace availability.
|
||||
* Never print tokens, DSNs, Authorization headers, or secret-bearing env vars.
|
||||
* Sanctioned repair path is: backup active config -> patch active config from canonical
|
||||
-> reconnect through IDE/client -> verify with live ``gitea_whoami``.
|
||||
* FORBIDDEN: ``pkill``, mtime edits, source edits, or session-state edits for repair.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from webui import console_redaction
|
||||
|
||||
DEFAULT_ACTIVE_IDE_CONFIG = "~/.gemini/antigravity-ide/mcp_config.json"
|
||||
DEFAULT_GLOBAL_CONFIG = "~/.gemini/config/mcp_config.json"
|
||||
|
||||
REQUIRED_GITEA_ROLE_SERVERS = (
|
||||
"gitea-author",
|
||||
"gitea-reviewer",
|
||||
"gitea-merger",
|
||||
"gitea-reconciler",
|
||||
"gitea-controller",
|
||||
"gitea-tools",
|
||||
)
|
||||
|
||||
SANCTIONED_REPAIR_RUNBOOK: tuple[str, ...] = (
|
||||
"1. Backup active IDE config: cp ~/.gemini/antigravity-ide/mcp_config.json ~/.gemini/antigravity-ide/mcp_config.json.bak",
|
||||
"2. Patch active IDE config: copy required missing Gitea role server entries from global config (~/.gemini/config/mcp_config.json) into active IDE config.",
|
||||
"3. Reconnect via IDE/client UI or client restart (do NOT use host process kill).",
|
||||
"4. Verify active namespace health using live gitea_whoami and gitea_resolve_task_capability on each role namespace.",
|
||||
"FORBIDDEN REPAIR PATHS: pkill / host process kill, mtime touch edits, source code edits, or session-state edits.",
|
||||
)
|
||||
|
||||
|
||||
def resolve_config_path(path_str: str) -> Path:
|
||||
"""Expand user and resolve absolute path."""
|
||||
return Path(os.path.expanduser(path_str)).resolve()
|
||||
|
||||
|
||||
def load_mcp_config(config_path: str | Path) -> tuple[dict[str, Any] | None, str | None]:
|
||||
"""Load and parse JSON MCP configuration from file.
|
||||
|
||||
Returns (config_dict, error_message).
|
||||
"""
|
||||
resolved = resolve_config_path(str(config_path))
|
||||
if not resolved.exists():
|
||||
return None, f"file_not_found: {resolved}"
|
||||
try:
|
||||
with open(resolved, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
if not isinstance(data, dict):
|
||||
return None, f"invalid_schema: root is not a JSON object in {resolved}"
|
||||
return data, None
|
||||
except Exception as exc:
|
||||
return None, f"unreadable_json: {exc} in {resolved}"
|
||||
|
||||
|
||||
def extract_mcp_servers(config: dict[str, Any] | None) -> dict[str, dict[str, Any]]:
|
||||
"""Extract the mcpServers or mcp_servers mapping safely."""
|
||||
if not config:
|
||||
return {}
|
||||
servers = config.get("mcpServers") or config.get("mcp_servers") or {}
|
||||
if isinstance(servers, dict):
|
||||
return {str(k): v for k, v in servers.items() if isinstance(v, dict)}
|
||||
return {}
|
||||
|
||||
|
||||
def _safe_redact_server_config(srv_cfg: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Redact secrets from environment variables and command line args."""
|
||||
safe = {}
|
||||
if "command" in srv_cfg:
|
||||
safe["command"] = str(srv_cfg["command"])
|
||||
if "args" in srv_cfg and isinstance(srv_cfg["args"], list):
|
||||
safe["args"] = [console_redaction.redact_text(str(a)) for a in srv_cfg["args"]]
|
||||
if "env" in srv_cfg and isinstance(srv_cfg["env"], dict):
|
||||
safe_env = {}
|
||||
for k, v in srv_cfg["env"].items():
|
||||
if any(secret_kw in k.lower() for secret_kw in ("token", "secret", "pass", "key", "auth")):
|
||||
safe_env[k] = "[REDACTED]"
|
||||
else:
|
||||
safe_env[k] = console_redaction.redact_text(str(v))
|
||||
safe["env"] = safe_env
|
||||
return safe
|
||||
|
||||
|
||||
def analyze_config_drift(
|
||||
active_config_path: str = DEFAULT_ACTIVE_IDE_CONFIG,
|
||||
global_config_path: str = DEFAULT_GLOBAL_CONFIG,
|
||||
) -> dict[str, Any]:
|
||||
"""Analyze MCP configuration drift between active IDE config and global config.
|
||||
|
||||
Returns structured diagnostic output.
|
||||
"""
|
||||
active_resolved = resolve_config_path(active_config_path)
|
||||
global_resolved = resolve_config_path(global_config_path)
|
||||
|
||||
active_cfg, active_err = load_mcp_config(active_resolved)
|
||||
global_cfg, global_err = load_mcp_config(global_resolved)
|
||||
|
||||
active_servers = extract_mcp_servers(active_cfg)
|
||||
global_servers = extract_mcp_servers(global_cfg)
|
||||
|
||||
missing_role_servers: list[str] = []
|
||||
present_role_servers: list[str] = []
|
||||
profile_mismatches: list[dict[str, Any]] = []
|
||||
reasons: list[str] = []
|
||||
|
||||
if active_err:
|
||||
reasons.append(f"Active IDE config error: {active_err}")
|
||||
if global_err:
|
||||
reasons.append(f"Global canonical config error: {global_err}")
|
||||
|
||||
# Check Gitea role servers
|
||||
for srv_name in REQUIRED_GITEA_ROLE_SERVERS:
|
||||
in_active = srv_name in active_servers
|
||||
in_global = srv_name in global_servers
|
||||
|
||||
if in_active:
|
||||
present_role_servers.append(srv_name)
|
||||
elif in_global:
|
||||
missing_role_servers.append(srv_name)
|
||||
reasons.append(
|
||||
f"Missing Gitea role server '{srv_name}' in active IDE config ({active_resolved})"
|
||||
)
|
||||
|
||||
if in_active and in_global:
|
||||
# Compare profiles & environments
|
||||
act_env = active_servers[srv_name].get("env", {}) if isinstance(active_servers[srv_name], dict) else {}
|
||||
glo_env = global_servers[srv_name].get("env", {}) if isinstance(global_servers[srv_name], dict) else {}
|
||||
|
||||
act_prof = act_env.get("GITEA_MCP_PROFILE") or act_env.get("GITEA_PROFILE_NAME")
|
||||
glo_prof = glo_env.get("GITEA_MCP_PROFILE") or glo_env.get("GITEA_PROFILE_NAME")
|
||||
|
||||
if act_prof != glo_prof:
|
||||
mismatch_item = {
|
||||
"server": srv_name,
|
||||
"active_profile": act_prof,
|
||||
"global_profile": glo_prof,
|
||||
}
|
||||
profile_mismatches.append(mismatch_item)
|
||||
reasons.append(
|
||||
f"Profile mismatch for '{srv_name}': active='{act_prof}' != global='{glo_prof}'"
|
||||
)
|
||||
|
||||
in_sync = bool(
|
||||
not active_err
|
||||
and not global_err
|
||||
and not missing_role_servers
|
||||
and not profile_mismatches
|
||||
)
|
||||
|
||||
report = {
|
||||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||
"in_sync": in_sync,
|
||||
"active_config_path": str(active_resolved),
|
||||
"active_config_exists": active_cfg is not None,
|
||||
"global_config_path": str(global_resolved),
|
||||
"global_config_exists": global_cfg is not None,
|
||||
"required_role_servers": list(REQUIRED_GITEA_ROLE_SERVERS),
|
||||
"present_role_servers": present_role_servers,
|
||||
"missing_role_servers": missing_role_servers,
|
||||
"profile_mismatches": profile_mismatches,
|
||||
"reasons": reasons,
|
||||
"sanctioned_repair_runbook": list(SANCTIONED_REPAIR_RUNBOOK),
|
||||
"forbidden_repair_methods": [
|
||||
"pkill / host process kill",
|
||||
"mtime touch edits",
|
||||
"source code edits",
|
||||
"session-state edits",
|
||||
],
|
||||
}
|
||||
|
||||
return console_redaction.redact_payload(report)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Diagnose Gitea MCP role server config drift between active IDE and global config."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--active-config",
|
||||
default=DEFAULT_ACTIVE_IDE_CONFIG,
|
||||
help="Path to active IDE MCP config JSON",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--global-config",
|
||||
default=DEFAULT_GLOBAL_CONFIG,
|
||||
help="Path to global/canonical MCP config JSON",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--json", action="store_true", help="Print raw JSON report"
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
report = analyze_config_drift(args.active_config, args.global_config)
|
||||
|
||||
if args.json:
|
||||
print(json.dumps(report, indent=2))
|
||||
else:
|
||||
print("=== MCP Config Drift Diagnostic Report ===")
|
||||
print(f"Timestamp: {report['timestamp']}")
|
||||
print(f"In Sync: {report['in_sync']}")
|
||||
print(f"Active IDE Config: {report['active_config_path']} (exists={report['active_config_exists']})")
|
||||
print(f"Global Config: {report['global_config_path']} (exists={report['global_config_exists']})")
|
||||
print(f"Present Role Servers: {', '.join(report['present_role_servers']) if report['present_role_servers'] else 'None'}")
|
||||
print(f"Missing Role Servers: {', '.join(report['missing_role_servers']) if report['missing_role_servers'] else 'None'}")
|
||||
if report['profile_mismatches']:
|
||||
print("Profile Mismatches:")
|
||||
for m in report['profile_mismatches']:
|
||||
print(f" - {m['server']}: active={m['active_profile']} vs global={m['global_profile']}")
|
||||
if report['reasons']:
|
||||
print("Drift Reasons:")
|
||||
for r in report['reasons']:
|
||||
print(f" - {r}")
|
||||
print("\nSanctioned Repair Runbook:")
|
||||
for step in report['sanctioned_repair_runbook']:
|
||||
print(f" {step}")
|
||||
|
||||
sys.exit(0 if report["in_sync"] else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,785 @@
|
||||
"""Authoritative, read-only PRGS MCP fleet inventory (#949).
|
||||
|
||||
Every pre-existing runtime surface is *per process*. ``gitea_get_runtime_context``
|
||||
and ``gitea_assess_master_parity`` describe only the server answering the call.
|
||||
``gitea_assess_mcp_namespace_health`` accepts ``process``, ``probe_result`` and
|
||||
``registered_tools`` **from the caller**, so it cannot constrain the caller. The
|
||||
control-plane ``sessions`` table records allocator *task* sessions, not server
|
||||
processes. Five independent self-reports of the same revision therefore never
|
||||
proved that exactly five processes exist, that no sixth exists, or that all five
|
||||
belong to one client cohort.
|
||||
|
||||
Evidence model
|
||||
--------------
|
||||
Two independent sources must agree before a fleet member counts as running:
|
||||
|
||||
``control-plane runtime registry``
|
||||
A row each server writes **about itself** at native transport bind
|
||||
(:func:`build_process_runtime_record`). No caller can supply it. It is
|
||||
authoritative for identity: namespace, profile, role, repository binding,
|
||||
cohort, and the revision the process started at.
|
||||
|
||||
``server-side process observation``
|
||||
A process listing performed by the server answering the inventory call
|
||||
(:func:`scan_mcp_server_processes`), never by the caller. It is
|
||||
authoritative for existence and liveness, and it is the only source that
|
||||
can show a process the registry does not know about.
|
||||
|
||||
A member is ``live`` only when a registry row has a matching, still-running
|
||||
process whose start time precedes the registration (so a recycled PID cannot
|
||||
impersonate a dead server). Anything the two sources cannot jointly establish
|
||||
is reported as unknown and fails the mutation gate closed — configuration alone
|
||||
never counts as a running member, and matching Git revisions never establish a
|
||||
single cohort.
|
||||
|
||||
This module performs no restart, reconnect, lease mutation, issue mutation, or
|
||||
process termination. The only signal it ever sends is ``signal 0`` liveness
|
||||
probing, which delivers nothing to the target process.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import secrets
|
||||
import subprocess
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Iterable, Mapping, Sequence
|
||||
|
||||
# ── expected fleet ────────────────────────────────────────────────────────────
|
||||
|
||||
# The configured PRGS fleet. Each entry is one expected member; the roster is
|
||||
# the definition of "expected" for missing/unexpected classification.
|
||||
EXPECTED_PRGS_FLEET: tuple[dict[str, str], ...] = (
|
||||
{"namespace": "gitea-author", "profile": "prgs-author", "role": "author"},
|
||||
{
|
||||
"namespace": "gitea-controller",
|
||||
"profile": "prgs-controller",
|
||||
"role": "controller",
|
||||
},
|
||||
{"namespace": "gitea-reviewer", "profile": "prgs-reviewer", "role": "reviewer"},
|
||||
{"namespace": "gitea-merger", "profile": "prgs-merger", "role": "merger"},
|
||||
{
|
||||
"namespace": "gitea-reconciler",
|
||||
"profile": "prgs-reconciler",
|
||||
"role": "reconciler",
|
||||
},
|
||||
)
|
||||
|
||||
# Cohort identity supplied by a client that manages the whole fleet.
|
||||
COHORT_ID_ENV = "GITEA_MCP_CLIENT_COHORT_ID"
|
||||
|
||||
# Registry rows older than this are pruned at *registration* time (a startup
|
||||
# write), never on the read path. Dead rows inside the window are still reported
|
||||
# as stale evidence rather than silently dropped.
|
||||
RUNTIME_RETENTION_SECONDS = 7 * 24 * 3600
|
||||
|
||||
# Liveness classifications for a registry row.
|
||||
LIVENESS_LIVE = "live"
|
||||
LIVENESS_DEAD = "dead"
|
||||
LIVENESS_PID_RECYCLED = "pid_recycled"
|
||||
LIVENESS_UNOBSERVED = "unobserved"
|
||||
LIVENESS_UNKNOWN = "unknown"
|
||||
|
||||
# Per-member health classifications.
|
||||
HEALTH_RUNNING = "running"
|
||||
HEALTH_MISSING = "missing"
|
||||
HEALTH_DUPLICATE = "duplicate"
|
||||
HEALTH_UNEXPECTED = "unexpected"
|
||||
HEALTH_STALE = "stale"
|
||||
HEALTH_UNKNOWN = "unknown"
|
||||
|
||||
EVIDENCE_AUTHORITY = "control_plane_runtime_registry+server_process_observation"
|
||||
|
||||
_MCP_PROCESS_MARKER = "mcp_server.py"
|
||||
_LSTART_FORMAT = "%a %b %d %H:%M:%S %Y"
|
||||
_ISO_FORMAT = "%Y-%m-%dT%H:%M:%SZ"
|
||||
|
||||
|
||||
# ── time helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _utcnow() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def iso_now() -> str:
|
||||
return _utcnow().strftime(_ISO_FORMAT)
|
||||
|
||||
|
||||
def _parse_iso(value: Any) -> datetime | None:
|
||||
text = (str(value) if value is not None else "").strip()
|
||||
if not text:
|
||||
return None
|
||||
if text.endswith("Z"):
|
||||
text = text[:-1] + "+00:00"
|
||||
try:
|
||||
parsed = datetime.fromisoformat(text)
|
||||
except ValueError:
|
||||
return None
|
||||
if parsed.tzinfo is None:
|
||||
parsed = parsed.replace(tzinfo=timezone.utc)
|
||||
return parsed.astimezone(timezone.utc)
|
||||
|
||||
|
||||
def _int_or_none(value: Any) -> int | None:
|
||||
try:
|
||||
return int(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _clean(value: Any) -> str | None:
|
||||
text = (str(value) if value is not None else "").strip()
|
||||
return text or None
|
||||
|
||||
|
||||
# ── process-level evidence (server side only) ─────────────────────────────────
|
||||
|
||||
|
||||
def probe_pid_alive(pid: int | None) -> bool | None:
|
||||
"""Return whether *pid* exists. ``None`` when it cannot be determined.
|
||||
|
||||
Uses ``signal 0``, which performs a permission/existence check and delivers
|
||||
nothing to the target. This module never sends a terminating signal.
|
||||
"""
|
||||
resolved = _int_or_none(pid)
|
||||
if resolved is None or resolved <= 0:
|
||||
return None
|
||||
try:
|
||||
os.kill(resolved, 0)
|
||||
except ProcessLookupError:
|
||||
return False
|
||||
except PermissionError:
|
||||
# The process exists but belongs to another user.
|
||||
return True
|
||||
except OSError:
|
||||
return None
|
||||
return True
|
||||
|
||||
|
||||
def scan_mcp_server_processes(*, runner=subprocess.run) -> dict[str, Any]:
|
||||
"""Observe running Gitea MCP server processes from this server process.
|
||||
|
||||
This is deliberately performed by the answering server, never by the caller:
|
||||
a caller-supplied process list is exactly the input a fleet gate must not
|
||||
trust. When the listing cannot be obtained the result reports
|
||||
``available=False`` so the inventory fails closed instead of assuming that
|
||||
no unregistered process exists.
|
||||
"""
|
||||
try:
|
||||
proc = runner(
|
||||
["ps", "-o", "pid,lstart,command", "-ax"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 - any failure means unknown evidence
|
||||
return {
|
||||
"available": False,
|
||||
"processes": [],
|
||||
"reason": f"process listing unavailable: {exc}",
|
||||
}
|
||||
|
||||
processes: list[dict[str, Any]] = []
|
||||
for raw_line in (proc.stdout or "").splitlines()[1:]:
|
||||
line = raw_line.strip()
|
||||
if not line or _MCP_PROCESS_MARKER not in line:
|
||||
continue
|
||||
parts = line.split(None, 6)
|
||||
if len(parts) < 7:
|
||||
continue
|
||||
pid = _int_or_none(parts[0])
|
||||
if pid is None:
|
||||
continue
|
||||
try:
|
||||
naive = datetime.strptime(" ".join(parts[1:6]), _LSTART_FORMAT)
|
||||
started_at = naive.astimezone(timezone.utc)
|
||||
except (ValueError, OSError):
|
||||
started_at = None
|
||||
processes.append(
|
||||
{
|
||||
"pid": pid,
|
||||
"started_at": started_at.strftime(_ISO_FORMAT) if started_at else None,
|
||||
"command": parts[6],
|
||||
}
|
||||
)
|
||||
|
||||
processes.sort(key=lambda item: item["pid"])
|
||||
return {"available": True, "processes": processes, "reason": None}
|
||||
|
||||
|
||||
# ── cohort / registration record ──────────────────────────────────────────────
|
||||
|
||||
|
||||
def namespace_for_profile(profile: str | None, *, default: str | None = None) -> str | None:
|
||||
"""Map a configured profile to its fleet namespace.
|
||||
|
||||
Resolved from the expected roster rather than from name-shape heuristics.
|
||||
``role_namespace_gate.infer_mcp_namespace`` only recognises author and
|
||||
reviewer, so it returns the *profile* name for controller, merger, and
|
||||
reconciler — which would label three of the five members with a namespace
|
||||
that does not exist. Widening that helper is controller role-metadata work
|
||||
and belongs to #950; the roster already carries the mapping this inventory
|
||||
needs, so it is read from there.
|
||||
|
||||
A profile outside the roster falls back to *default* (typically the
|
||||
caller's existing inference), so an unexpected member is still described
|
||||
rather than dropped.
|
||||
"""
|
||||
cleaned = _clean(profile)
|
||||
for entry in EXPECTED_PRGS_FLEET:
|
||||
if entry["profile"] == cleaned:
|
||||
return entry["namespace"]
|
||||
return default if default is not None else cleaned
|
||||
|
||||
|
||||
def derive_cohort_identity(env: Mapping[str, str] | None = None) -> dict[str, Any]:
|
||||
"""Derive the client/cohort identity of this server process.
|
||||
|
||||
A cohort is the set of servers a single client launched together. The
|
||||
parent process is the durable expression of that: an IDE/CLI client spawns
|
||||
every namespace as its own child. A process whose parent has gone away
|
||||
(reparented to init) cannot prove which cohort it belongs to, and says so
|
||||
rather than guessing.
|
||||
|
||||
Revisions are deliberately not consulted here. Two servers built from the
|
||||
same commit are not thereby one cohort (#949 AC7).
|
||||
"""
|
||||
source_env = os.environ if env is None else env
|
||||
explicit = _clean(source_env.get(COHORT_ID_ENV))
|
||||
if explicit:
|
||||
return {"cohort_id": explicit, "cohort_source": "explicit_env"}
|
||||
try:
|
||||
ppid = os.getppid()
|
||||
except OSError:
|
||||
ppid = 0
|
||||
if ppid and ppid > 1:
|
||||
return {"cohort_id": f"ppid:{ppid}", "cohort_source": "parent_process"}
|
||||
return {
|
||||
"cohort_id": None,
|
||||
"cohort_source": "unknown",
|
||||
"cohort_reason": (
|
||||
"parent process is unavailable or reparented to init; this server "
|
||||
"cannot prove which client cohort launched it"
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def build_process_runtime_record(
|
||||
*,
|
||||
namespace: str,
|
||||
profile: str | None,
|
||||
role: str | None,
|
||||
remote: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
repository_root: str | None = None,
|
||||
pid: int | None = None,
|
||||
startup_head: str | None = None,
|
||||
daemon_start_head: str | None = None,
|
||||
transport: str | None = None,
|
||||
client_provenance: str | None = None,
|
||||
env: Mapping[str, str] | None = None,
|
||||
boot_id: str | None = None,
|
||||
registered_at: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build the row a server writes about itself at native transport bind.
|
||||
|
||||
Every field describes the *calling* process. Nothing here is caller-supplied
|
||||
in the MCP sense: the only code that reaches this function is the official
|
||||
entrypoint of the process being described.
|
||||
"""
|
||||
cohort = derive_cohort_identity(env)
|
||||
resolved_pid = _int_or_none(pid)
|
||||
if resolved_pid is None:
|
||||
resolved_pid = os.getpid()
|
||||
token = boot_id or secrets.token_hex(8)
|
||||
return {
|
||||
"runtime_id": f"{namespace}:{resolved_pid}:{token}",
|
||||
"namespace": namespace,
|
||||
"profile": _clean(profile),
|
||||
"role": _clean(role),
|
||||
"remote": _clean(remote),
|
||||
"org": _clean(org),
|
||||
"repo": _clean(repo),
|
||||
"repository_root": _clean(repository_root),
|
||||
"pid": resolved_pid,
|
||||
"cohort_id": cohort["cohort_id"],
|
||||
"cohort_source": cohort["cohort_source"],
|
||||
"client_provenance": _clean(client_provenance) or "unknown",
|
||||
"boot_id": token,
|
||||
"startup_head": _clean(startup_head),
|
||||
"daemon_start_head": _clean(daemon_start_head),
|
||||
"transport": _clean(transport),
|
||||
"registered_at": registered_at or iso_now(),
|
||||
"status": "running",
|
||||
}
|
||||
|
||||
|
||||
# ── classification ────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _expected_index(
|
||||
expected_fleet: Sequence[Mapping[str, str]],
|
||||
) -> dict[str, dict[str, str]]:
|
||||
index: dict[str, dict[str, str]] = {}
|
||||
for entry in expected_fleet:
|
||||
profile = _clean(entry.get("profile"))
|
||||
if profile:
|
||||
index[profile] = dict(entry)
|
||||
return index
|
||||
|
||||
|
||||
def _sort_key(member: Mapping[str, Any]) -> tuple:
|
||||
return (
|
||||
str(member.get("namespace") or ""),
|
||||
str(member.get("profile") or ""),
|
||||
_int_or_none(member.get("pid")) or 0,
|
||||
str(member.get("runtime_id") or ""),
|
||||
)
|
||||
|
||||
|
||||
def _normalize_row(
|
||||
row: Mapping[str, Any],
|
||||
*,
|
||||
observed_by_pid: Mapping[int, Mapping[str, Any]],
|
||||
process_scan_available: bool,
|
||||
now: datetime,
|
||||
) -> dict[str, Any]:
|
||||
pid = _int_or_none(row.get("pid"))
|
||||
member: dict[str, Any] = {
|
||||
"runtime_id": _clean(row.get("runtime_id")),
|
||||
"namespace": _clean(row.get("namespace")),
|
||||
"profile": _clean(row.get("profile")),
|
||||
"role": _clean(row.get("role")),
|
||||
"remote": _clean(row.get("remote")),
|
||||
"org": _clean(row.get("org")),
|
||||
"repo": _clean(row.get("repo")),
|
||||
"repository_root": _clean(row.get("repository_root")),
|
||||
"pid": pid,
|
||||
"cohort_id": _clean(row.get("cohort_id")),
|
||||
"cohort_source": _clean(row.get("cohort_source")) or "unknown",
|
||||
"client_provenance": _clean(row.get("client_provenance")) or "unknown",
|
||||
"boot_id": _clean(row.get("boot_id")),
|
||||
"startup_head": _clean(row.get("startup_head")),
|
||||
"daemon_start_head": _clean(row.get("daemon_start_head")),
|
||||
"transport": _clean(row.get("transport")),
|
||||
"registered_at": _clean(row.get("registered_at")),
|
||||
"last_heartbeat_at": _clean(row.get("last_heartbeat_at")),
|
||||
"recorded_status": _clean(row.get("status")) or "unknown",
|
||||
}
|
||||
|
||||
pid_alive = probe_pid_alive(pid)
|
||||
member["pid_alive"] = pid_alive
|
||||
|
||||
observed = observed_by_pid.get(pid) if pid is not None else None
|
||||
member["process_observed"] = bool(observed) if process_scan_available else None
|
||||
|
||||
if not process_scan_available:
|
||||
# Existence cannot be corroborated; never upgrade to live on the
|
||||
# registry's word alone.
|
||||
member["liveness"] = LIVENESS_UNKNOWN
|
||||
member["liveness_reason"] = (
|
||||
"process observation unavailable; registry rows cannot be corroborated"
|
||||
)
|
||||
elif pid_alive is False:
|
||||
member["liveness"] = LIVENESS_DEAD
|
||||
member["liveness_reason"] = "recorded PID is not running"
|
||||
elif pid_alive is None:
|
||||
member["liveness"] = LIVENESS_UNKNOWN
|
||||
member["liveness_reason"] = "PID liveness could not be determined"
|
||||
elif observed is None:
|
||||
member["liveness"] = LIVENESS_UNOBSERVED
|
||||
member["liveness_reason"] = (
|
||||
"recorded PID is not a running Gitea MCP server process"
|
||||
)
|
||||
else:
|
||||
started_at = _parse_iso(observed.get("started_at"))
|
||||
registered_at = _parse_iso(member["registered_at"])
|
||||
if started_at and registered_at and started_at > registered_at:
|
||||
member["liveness"] = LIVENESS_PID_RECYCLED
|
||||
member["liveness_reason"] = (
|
||||
"the process now holding this PID started after the registry row "
|
||||
"was written; the registered server is gone"
|
||||
)
|
||||
else:
|
||||
member["liveness"] = LIVENESS_LIVE
|
||||
member["liveness_reason"] = None
|
||||
|
||||
heartbeat = _parse_iso(member["last_heartbeat_at"])
|
||||
member["heartbeat_age_seconds"] = (
|
||||
int((now - heartbeat).total_seconds()) if heartbeat else None
|
||||
)
|
||||
return member
|
||||
|
||||
|
||||
def _binding_matches(
|
||||
member: Mapping[str, Any], expected_binding: Mapping[str, Any] | None
|
||||
) -> bool | None:
|
||||
if not expected_binding:
|
||||
return None
|
||||
for field in ("remote", "org", "repo"):
|
||||
expected = _clean(expected_binding.get(field))
|
||||
if expected is None:
|
||||
continue
|
||||
actual = _clean(member.get(field))
|
||||
if actual is None:
|
||||
return None
|
||||
if actual != expected:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def classify_fleet_inventory(
|
||||
*,
|
||||
runtime_rows: Iterable[Mapping[str, Any]],
|
||||
process_scan: Mapping[str, Any] | None = None,
|
||||
expected_fleet: Sequence[Mapping[str, str]] = EXPECTED_PRGS_FLEET,
|
||||
expected_binding: Mapping[str, Any] | None = None,
|
||||
registry_available: bool = True,
|
||||
registry_error: str | None = None,
|
||||
now: datetime | None = None,
|
||||
answering_namespace: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Classify a fleet snapshot. Pure: identical input yields identical output.
|
||||
|
||||
The verdict never depends on which namespace asked, so controller and
|
||||
reconciler agree by construction; ``answering_namespace`` is reported as
|
||||
metadata only.
|
||||
"""
|
||||
moment = now or _utcnow()
|
||||
scan = dict(process_scan or {"available": False, "processes": [], "reason": None})
|
||||
scan_available = bool(scan.get("available"))
|
||||
observed_processes = list(scan.get("processes") or [])
|
||||
observed_by_pid: dict[int, Mapping[str, Any]] = {}
|
||||
for proc in observed_processes:
|
||||
observed_pid = _int_or_none(proc.get("pid"))
|
||||
if observed_pid is not None:
|
||||
observed_by_pid[observed_pid] = proc
|
||||
|
||||
expected_index = _expected_index(expected_fleet)
|
||||
|
||||
members = [
|
||||
_normalize_row(
|
||||
row,
|
||||
observed_by_pid=observed_by_pid,
|
||||
process_scan_available=scan_available,
|
||||
now=moment,
|
||||
)
|
||||
for row in (runtime_rows or [])
|
||||
]
|
||||
|
||||
live = [m for m in members if m["liveness"] == LIVENESS_LIVE]
|
||||
not_live = [m for m in members if m["liveness"] != LIVENESS_LIVE]
|
||||
|
||||
live_by_profile: dict[str, list[dict[str, Any]]] = {}
|
||||
for member in live:
|
||||
live_by_profile.setdefault(member["profile"] or "", []).append(member)
|
||||
|
||||
running_members: list[dict[str, Any]] = []
|
||||
missing_members: list[dict[str, Any]] = []
|
||||
duplicate_members: list[dict[str, Any]] = []
|
||||
unexpected_members: list[dict[str, Any]] = []
|
||||
binding_mismatches: list[dict[str, Any]] = []
|
||||
role_mismatches: list[dict[str, Any]] = []
|
||||
configured_members: list[dict[str, Any]] = []
|
||||
|
||||
for entry in expected_fleet:
|
||||
profile = _clean(entry.get("profile")) or ""
|
||||
instances = sorted(live_by_profile.get(profile, []), key=_sort_key)
|
||||
configured_members.append(
|
||||
{
|
||||
"namespace": _clean(entry.get("namespace")),
|
||||
"profile": profile,
|
||||
"role": _clean(entry.get("role")),
|
||||
"instance_count": len(instances),
|
||||
"health": (
|
||||
HEALTH_MISSING
|
||||
if not instances
|
||||
else HEALTH_RUNNING
|
||||
if len(instances) == 1
|
||||
else HEALTH_DUPLICATE
|
||||
),
|
||||
}
|
||||
)
|
||||
if not instances:
|
||||
missing_members.append(
|
||||
{
|
||||
"namespace": _clean(entry.get("namespace")),
|
||||
"profile": profile,
|
||||
"role": _clean(entry.get("role")),
|
||||
"health": HEALTH_MISSING,
|
||||
"reason": (
|
||||
"no live registry row corroborated by a running server "
|
||||
"process"
|
||||
),
|
||||
}
|
||||
)
|
||||
continue
|
||||
for instance in instances:
|
||||
instance["health"] = (
|
||||
HEALTH_RUNNING if len(instances) == 1 else HEALTH_DUPLICATE
|
||||
)
|
||||
instance["expected"] = True
|
||||
running_members.append(instance)
|
||||
if len(instances) > 1:
|
||||
duplicate_members.append(
|
||||
{
|
||||
"namespace": _clean(entry.get("namespace")),
|
||||
"profile": profile,
|
||||
"role": _clean(entry.get("role")),
|
||||
"health": HEALTH_DUPLICATE,
|
||||
"instance_count": len(instances),
|
||||
"pids": sorted(
|
||||
i["pid"] for i in instances if i["pid"] is not None
|
||||
),
|
||||
"instances": instances,
|
||||
"reason": (
|
||||
"more than one live server is registered for this profile"
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
for member in sorted(live, key=_sort_key):
|
||||
profile = member["profile"] or ""
|
||||
expected_entry = expected_index.get(profile)
|
||||
if expected_entry is not None:
|
||||
expected_role = _clean(expected_entry.get("role"))
|
||||
actual_role = member["role"]
|
||||
if expected_role and actual_role and actual_role != expected_role:
|
||||
role_mismatches.append(
|
||||
{
|
||||
"namespace": member["namespace"],
|
||||
"profile": profile,
|
||||
"pid": member["pid"],
|
||||
"expected_role": expected_role,
|
||||
"declared_role": actual_role,
|
||||
"reason": "declared role does not match the configured profile",
|
||||
}
|
||||
)
|
||||
expected_namespace = _clean(expected_entry.get("namespace"))
|
||||
if (
|
||||
expected_namespace
|
||||
and member["namespace"]
|
||||
and member["namespace"] != expected_namespace
|
||||
):
|
||||
role_mismatches.append(
|
||||
{
|
||||
"namespace": member["namespace"],
|
||||
"profile": profile,
|
||||
"pid": member["pid"],
|
||||
"expected_namespace": expected_namespace,
|
||||
"declared_role": member["role"],
|
||||
"reason": (
|
||||
"profile is served from a namespace it is not "
|
||||
"configured for"
|
||||
),
|
||||
}
|
||||
)
|
||||
else:
|
||||
member["health"] = HEALTH_UNEXPECTED
|
||||
member["expected"] = False
|
||||
unexpected_members.append(member)
|
||||
|
||||
match = _binding_matches(member, expected_binding)
|
||||
member["repository_binding_matches"] = match
|
||||
if match is False:
|
||||
binding_mismatches.append(
|
||||
{
|
||||
"namespace": member["namespace"],
|
||||
"profile": profile,
|
||||
"pid": member["pid"],
|
||||
"remote": member["remote"],
|
||||
"org": member["org"],
|
||||
"repo": member["repo"],
|
||||
"expected": dict(expected_binding or {}),
|
||||
"reason": "member is bound to a different repository",
|
||||
}
|
||||
)
|
||||
elif match is None and expected_binding:
|
||||
binding_mismatches.append(
|
||||
{
|
||||
"namespace": member["namespace"],
|
||||
"profile": profile,
|
||||
"pid": member["pid"],
|
||||
"remote": member["remote"],
|
||||
"org": member["org"],
|
||||
"repo": member["repo"],
|
||||
"expected": dict(expected_binding or {}),
|
||||
"reason": "member did not record a complete repository binding",
|
||||
}
|
||||
)
|
||||
|
||||
stale_members: list[dict[str, Any]] = []
|
||||
for member in sorted(not_live, key=_sort_key):
|
||||
member["health"] = (
|
||||
HEALTH_UNKNOWN if member["liveness"] == LIVENESS_UNKNOWN else HEALTH_STALE
|
||||
)
|
||||
stale_members.append(member)
|
||||
|
||||
registered_pids = {m["pid"] for m in live if m["pid"] is not None}
|
||||
unregistered_processes: list[dict[str, Any]] = []
|
||||
if scan_available:
|
||||
for proc in observed_processes:
|
||||
observed_pid = _int_or_none(proc.get("pid"))
|
||||
if observed_pid is None or observed_pid in registered_pids:
|
||||
continue
|
||||
unregistered_processes.append(
|
||||
{"pid": observed_pid, "started_at": proc.get("started_at")}
|
||||
)
|
||||
unregistered_processes.sort(key=lambda item: item["pid"])
|
||||
|
||||
# Cohort. Derived only from recorded cohort identity — never from revisions.
|
||||
cohort_ids = {m["cohort_id"] for m in live}
|
||||
cohort_unknown = any(cohort_id is None for cohort_id in cohort_ids)
|
||||
known_cohorts = sorted(c for c in cohort_ids if c is not None)
|
||||
if not live or cohort_unknown:
|
||||
single_cohort: bool | None = None
|
||||
mixed_cohort: bool | None = None
|
||||
else:
|
||||
single_cohort = len(known_cohorts) == 1
|
||||
mixed_cohort = len(known_cohorts) > 1
|
||||
|
||||
# Revision spread. Reported independently of cohort; never used to infer it.
|
||||
heads = {m["startup_head"] for m in live}
|
||||
head_unknown = any(head is None for head in heads)
|
||||
known_heads = sorted(h for h in heads if h is not None)
|
||||
mixed_revision = (len(known_heads) > 1) if known_heads else None
|
||||
|
||||
exactly_one_per_profile = not missing_members and not duplicate_members
|
||||
no_unexpected_members = not unexpected_members
|
||||
|
||||
incomplete_reasons: list[str] = []
|
||||
if not registry_available:
|
||||
incomplete_reasons.append(
|
||||
registry_error or "the control-plane runtime registry could not be read"
|
||||
)
|
||||
if not scan_available:
|
||||
incomplete_reasons.append(
|
||||
str(scan.get("reason") or "server-side process observation unavailable")
|
||||
)
|
||||
if unregistered_processes:
|
||||
pids = ", ".join(str(p["pid"]) for p in unregistered_processes)
|
||||
incomplete_reasons.append(
|
||||
f"running Gitea MCP server process(es) with no runtime registry row "
|
||||
f"(PIDs: {pids}); the fleet contains members this inventory cannot "
|
||||
f"describe"
|
||||
)
|
||||
if live and cohort_unknown:
|
||||
incomplete_reasons.append(
|
||||
"one or more live members did not record a client cohort identity; "
|
||||
"matching revisions do not establish a single cohort"
|
||||
)
|
||||
if live and head_unknown:
|
||||
incomplete_reasons.append(
|
||||
"one or more live members did not record a startup revision"
|
||||
)
|
||||
if any(m["liveness"] == LIVENESS_UNKNOWN for m in members):
|
||||
incomplete_reasons.append(
|
||||
"liveness of one or more registry rows could not be determined"
|
||||
)
|
||||
|
||||
inventory_complete = not incomplete_reasons
|
||||
|
||||
blocked_reasons: list[str] = list(incomplete_reasons)
|
||||
if missing_members:
|
||||
names = ", ".join(sorted(m["profile"] for m in missing_members))
|
||||
blocked_reasons.append(f"expected fleet member(s) not running: {names}")
|
||||
if duplicate_members:
|
||||
names = ", ".join(sorted(d["profile"] for d in duplicate_members))
|
||||
blocked_reasons.append(
|
||||
f"duplicate server(s) registered for profile(s): {names}"
|
||||
)
|
||||
if unexpected_members:
|
||||
names = ", ".join(
|
||||
sorted(
|
||||
str(m["profile"] or m["namespace"] or "?") for m in unexpected_members
|
||||
)
|
||||
)
|
||||
blocked_reasons.append(f"unexpected fleet member(s) running: {names}")
|
||||
if mixed_cohort:
|
||||
blocked_reasons.append(
|
||||
"live members span more than one client cohort: "
|
||||
+ ", ".join(known_cohorts)
|
||||
)
|
||||
if mixed_revision:
|
||||
blocked_reasons.append(
|
||||
"live members started at different revisions: " + ", ".join(known_heads)
|
||||
)
|
||||
if binding_mismatches:
|
||||
blocked_reasons.append(
|
||||
"one or more live members are not bound to the expected repository"
|
||||
)
|
||||
if role_mismatches:
|
||||
blocked_reasons.append(
|
||||
"one or more live members declare a role or namespace that does not "
|
||||
"match the configured profile"
|
||||
)
|
||||
|
||||
mutation_gate_satisfied = bool(
|
||||
inventory_complete
|
||||
and exactly_one_per_profile
|
||||
and no_unexpected_members
|
||||
and single_cohort is True
|
||||
and mixed_revision is False
|
||||
and not binding_mismatches
|
||||
and not role_mismatches
|
||||
)
|
||||
if not mutation_gate_satisfied and not blocked_reasons:
|
||||
blocked_reasons.append(
|
||||
"the fleet snapshot did not establish the exact-one-instance-per-"
|
||||
"profile, single-cohort invariant"
|
||||
)
|
||||
|
||||
return {
|
||||
"success": True,
|
||||
"read_only": True,
|
||||
"evidence_authority": EVIDENCE_AUTHORITY,
|
||||
"answering_namespace": _clean(answering_namespace),
|
||||
"generated_at": moment.strftime(_ISO_FORMAT),
|
||||
"inventory_complete": inventory_complete,
|
||||
"incomplete_reasons": incomplete_reasons,
|
||||
"configured_members": configured_members,
|
||||
"running_members": sorted(running_members, key=_sort_key),
|
||||
"missing_members": sorted(missing_members, key=_sort_key),
|
||||
"duplicate_members": sorted(duplicate_members, key=_sort_key),
|
||||
"unexpected_members": sorted(unexpected_members, key=_sort_key),
|
||||
"stale_members": stale_members,
|
||||
"unregistered_processes": unregistered_processes,
|
||||
"repository_binding_mismatches": sorted(binding_mismatches, key=_sort_key),
|
||||
"role_mismatches": sorted(role_mismatches, key=_sort_key),
|
||||
"cohort_ids": known_cohorts,
|
||||
"single_cohort": single_cohort,
|
||||
"mixed_cohort": mixed_cohort,
|
||||
"startup_revisions": known_heads,
|
||||
"mixed_revision": mixed_revision,
|
||||
"exactly_one_per_profile": exactly_one_per_profile,
|
||||
"no_unexpected_members": no_unexpected_members,
|
||||
"expected_member_count": len(expected_fleet),
|
||||
"running_member_count": len(running_members),
|
||||
"mutation_gate_satisfied": mutation_gate_satisfied,
|
||||
"blocked_reason": blocked_reasons[0] if blocked_reasons else None,
|
||||
"blocked_reasons": blocked_reasons,
|
||||
"process_observation": {
|
||||
"available": scan_available,
|
||||
"observed_process_count": len(observed_processes),
|
||||
"reason": scan.get("reason"),
|
||||
},
|
||||
"registry": {
|
||||
"available": registry_available,
|
||||
"row_count": len(members),
|
||||
"error": registry_error,
|
||||
},
|
||||
"mutations_performed": [],
|
||||
}
|
||||
|
||||
|
||||
def summarize(result: Mapping[str, Any]) -> str:
|
||||
"""One-line human summary of a classification result."""
|
||||
if result.get("mutation_gate_satisfied"):
|
||||
return (
|
||||
f"fleet healthy: {result.get('running_member_count')} of "
|
||||
f"{result.get('expected_member_count')} members running in a single "
|
||||
f"cohort at one revision"
|
||||
)
|
||||
return f"fleet not provable: {result.get('blocked_reason')}"
|
||||
@@ -225,6 +225,16 @@ def classify_namespace_probe(
|
||||
# on bad data without treating success as IDE proof).
|
||||
blocks = namespace_health_blocks_task("merge_pr", healthy)
|
||||
|
||||
import gitea_config
|
||||
raw_env = process.get("env") if isinstance(process, dict) else None
|
||||
unconsumed_env = gitea_config.get_unconsumed_gitea_env_overrides(raw_env)
|
||||
is_client_managed = bool(
|
||||
env_summary.get("GITEA_CLIENT_MANAGED") in ("1", "true", "yes", "client_managed")
|
||||
or env_summary.get("GITEA_MCP_CLIENT_MANAGED") in ("1", "true", "yes", "client_managed")
|
||||
or env_summary.get("GITEA_SERVER_PROVENANCE") == "client_managed"
|
||||
)
|
||||
provenance = "client_managed" if is_client_managed else "manual_launch"
|
||||
|
||||
return {
|
||||
"success": healthy,
|
||||
"healthy": healthy,
|
||||
@@ -240,6 +250,9 @@ def classify_namespace_probe(
|
||||
"error_message": error_message or None,
|
||||
"reasons": reasons,
|
||||
"remediation": remediation,
|
||||
"provenance": provenance,
|
||||
"is_client_managed": is_client_managed,
|
||||
"unconsumed_gitea_env": unconsumed_env,
|
||||
"diagnostics": {
|
||||
"namespace": ns,
|
||||
"required_tool": tool,
|
||||
@@ -248,6 +261,9 @@ def classify_namespace_probe(
|
||||
"env": env_summary,
|
||||
"config_path": config_path,
|
||||
"probe_source": source,
|
||||
"provenance": provenance,
|
||||
"is_client_managed": is_client_managed,
|
||||
"unconsumed_gitea_env": unconsumed_env,
|
||||
},
|
||||
"blocks_merge_workflow": blocks,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,505 @@
|
||||
"""Inventory and fail-closed guards for MCP restart/reload/kill paths (#657).
|
||||
|
||||
Single source of truth enumerating every code/script/doc path that can
|
||||
restart, reload, reconnect, kill, or force-recreate an MCP process. Each path
|
||||
is classified and linked to the guard that constrains it. The companion
|
||||
human-readable inventory lives in ``docs/mcp-restart-path-inventory.md`` and is
|
||||
kept in lock-step with this module by ``tests/test_mcp_restart_paths.py``.
|
||||
|
||||
Design intent (aligns with #655 restart-coordinator roadmap):
|
||||
|
||||
* **No unguarded full restart.** The in-process MCP daemon
|
||||
(``gitea_mcp_server.py`` / ``mcp_server.py`` / ``role_session_router.py``)
|
||||
must never replace or kill its own process — replacing the process after the
|
||||
host wired up the stdio pipes desyncs the JSON-RPC transport (observed with
|
||||
Antigravity/Cascade hosts). ``assert_no_daemon_self_replacement`` enforces
|
||||
this against the live source tree.
|
||||
* **No legacy auto-restart helper.** ``_trigger_mcp_auto_restart`` was removed
|
||||
when the stale-runtime resolver became side-effect free (#685);
|
||||
``assert_auto_restart_helper_absent`` keeps it removed.
|
||||
* **Unknown restart attempts fail closed.** LLM tools must route any restart
|
||||
intent through a *registered* path. ``assert_restart_attempt_registered``
|
||||
raises ``UnknownRestartPathError`` for anything not in this inventory.
|
||||
* **pkill stays forbidden (#630).** Manual daemon kills are classified as
|
||||
contamination by :mod:`runtime_recovery_guard`; this module records that path
|
||||
and the test asserts the classification still holds.
|
||||
|
||||
This module performs no restarts, spawns no threads, and touches no config or
|
||||
process state. It is pure inventory + read-only source assertions.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Iterable
|
||||
|
||||
# --- Classifications -------------------------------------------------------
|
||||
|
||||
#: A narrow, one-shot recovery that is safe by construction (e.g. a CLI wrapper
|
||||
#: re-execing into the venv interpreter before importing anything, or an
|
||||
#: in-process profile switch). Never targets the running MCP daemon process.
|
||||
CLASS_SANCTIONED_NARROW = "sanctioned_narrow_recovery"
|
||||
|
||||
#: The path detects a condition that would require a restart, then *fails
|
||||
#: closed* on mutations and emits restart/reconnect guidance. It never restarts
|
||||
#: the process itself (recovery is owned by the host/operator).
|
||||
CLASS_GUARDED_FAIL_CLOSED = "guarded_fail_closed"
|
||||
|
||||
#: The path is forbidden. Attempting it is a workflow-safety violation and,
|
||||
#: where an LLM tool could invoke it, is marked as contamination.
|
||||
CLASS_FORBIDDEN = "forbidden"
|
||||
|
||||
#: A previously-existing unguarded restart primitive that has been deleted. A
|
||||
#: regression guard keeps it absent.
|
||||
CLASS_REMOVED = "removed"
|
||||
|
||||
#: Behavior that lives in the host/IDE and is outside this process's control
|
||||
#: (e.g. a manual ``/mcp reconnect``). Documented, not code-guarded here.
|
||||
CLASS_HOST_RESIDUAL = "host_residual"
|
||||
|
||||
VALID_CLASSIFICATIONS = frozenset(
|
||||
{
|
||||
CLASS_SANCTIONED_NARROW,
|
||||
CLASS_GUARDED_FAIL_CLOSED,
|
||||
CLASS_FORBIDDEN,
|
||||
CLASS_REMOVED,
|
||||
CLASS_HOST_RESIDUAL,
|
||||
}
|
||||
)
|
||||
|
||||
#: The in-process MCP daemon modules. These must never self-replace/self-kill.
|
||||
DAEMON_MODULES = (
|
||||
"gitea_mcp_server.py",
|
||||
"mcp_server.py",
|
||||
"role_session_router.py",
|
||||
)
|
||||
|
||||
#: The legacy auto-restart helper removed in #685. Must stay removed.
|
||||
LEGACY_AUTO_RESTART_HELPER = "_trigger_mcp_auto_restart"
|
||||
|
||||
#: Call patterns that would let the daemon replace or terminate its own
|
||||
#: process. Matched as calls (trailing ``(``) so prose/docstring mentions such
|
||||
#: as "we do NOT os.execv() here" or "never calls ``os._exit``" do not trip the
|
||||
#: scanner (comment lines are stripped first regardless).
|
||||
DAEMON_SELF_REPLACEMENT_PRIMITIVES = (
|
||||
"os.execv(",
|
||||
"os.execve(",
|
||||
"os.execvp(",
|
||||
"os.execvpe(",
|
||||
"os.kill(",
|
||||
"os.killpg(",
|
||||
"os._exit(",
|
||||
"os.abort(",
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartPath:
|
||||
"""One classified restart/reload/kill path in the inventory."""
|
||||
|
||||
path_id: str
|
||||
title: str
|
||||
mechanism: str
|
||||
classification: str
|
||||
guard: str
|
||||
locations: tuple[str, ...]
|
||||
references: tuple[str, ...]
|
||||
residual_host: bool = False
|
||||
notes: str = ""
|
||||
|
||||
|
||||
class UnknownRestartPathError(RuntimeError):
|
||||
"""Raised when a restart attempt is not a registered, classified path."""
|
||||
|
||||
|
||||
# --- The inventory ---------------------------------------------------------
|
||||
|
||||
_RESTART_PATHS: tuple[RestartPath, ...] = (
|
||||
RestartPath(
|
||||
path_id="cli_venv_bootstrap_execv",
|
||||
title="CLI wrapper venv re-exec",
|
||||
mechanism=(
|
||||
"Standalone CLI scripts re-exec into venv/bin/python3 via os.execv "
|
||||
"at import top, guarded by `sys.executable != venv_python`."
|
||||
),
|
||||
classification=CLASS_SANCTIONED_NARROW,
|
||||
guard=(
|
||||
"One-shot, pre-import bootstrap; runs before any MCP transport "
|
||||
"exists and only when not already on the venv interpreter, so it "
|
||||
"cannot desync a live daemon. Idempotent guard condition prevents "
|
||||
"a re-exec loop."
|
||||
),
|
||||
locations=(
|
||||
"create_pr.py",
|
||||
"create_issue.py",
|
||||
"close_issue.py",
|
||||
"merge_pr.py",
|
||||
"review_pr.py",
|
||||
"edit_pr.py",
|
||||
"delete_branch.py",
|
||||
"mark_issue.py",
|
||||
"manage_labels.py",
|
||||
"list_issues.py",
|
||||
"list_prs.py",
|
||||
),
|
||||
references=("#657",),
|
||||
),
|
||||
RestartPath(
|
||||
path_id="daemon_self_replacement",
|
||||
title="MCP daemon self-replacement",
|
||||
mechanism=(
|
||||
"The in-process MCP daemon replacing/terminating its own process "
|
||||
"(os.execv/os.kill/os._exit) to reload code."
|
||||
),
|
||||
classification=CLASS_FORBIDDEN,
|
||||
guard=(
|
||||
"Forbidden by design: replacing the process after the host wired "
|
||||
"up stdio desyncs JSON-RPC (Antigravity/Cascade). Enforced against "
|
||||
"the source tree by assert_no_daemon_self_replacement()."
|
||||
),
|
||||
locations=("gitea_mcp_server.py:~155 (decision comment)",) + DAEMON_MODULES,
|
||||
references=("#657", "#584"),
|
||||
),
|
||||
RestartPath(
|
||||
path_id="legacy_auto_restart_helper",
|
||||
title="Legacy _trigger_mcp_auto_restart helper",
|
||||
mechanism=(
|
||||
"A helper that actively restarted the MCP server from the "
|
||||
"read-only resolver path."
|
||||
),
|
||||
classification=CLASS_REMOVED,
|
||||
guard=(
|
||||
"Removed in #685 when the resolver became side-effect free. Kept "
|
||||
"absent by assert_auto_restart_helper_absent()."
|
||||
),
|
||||
locations=("gitea_mcp_server.py", "mcp_server.py"),
|
||||
references=("#685", "#657"),
|
||||
),
|
||||
RestartPath(
|
||||
path_id="config_touch_reload",
|
||||
title="MCP client config-touch reload",
|
||||
mechanism=(
|
||||
"Touching (utime) the MCP client config file to make the host "
|
||||
"reload/recreate the server process."
|
||||
),
|
||||
classification=CLASS_REMOVED,
|
||||
guard=(
|
||||
"Removed from the resolver in #685: stale-runtime detection is "
|
||||
"report-only and never mutates client config, spawns threads, or "
|
||||
"calls os._exit."
|
||||
),
|
||||
locations=("gitea_mcp_server.py (resolve_task_capability)",),
|
||||
references=("#685", "#657"),
|
||||
),
|
||||
RestartPath(
|
||||
path_id="master_advance_auto_restart",
|
||||
title="Master-advance staleness gate",
|
||||
mechanism=(
|
||||
"On-disk master advancing past the running code. The master-parity "
|
||||
"gate detects it and fails mutations closed with restart guidance."
|
||||
),
|
||||
classification=CLASS_GUARDED_FAIL_CLOSED,
|
||||
guard=(
|
||||
"Detect + fail closed only; the process never self-restarts. "
|
||||
"master_parity_gate captures startup parity and blocks mutations "
|
||||
"while stale, emitting restart/reconnect guidance."
|
||||
),
|
||||
locations=(
|
||||
"master_parity_gate.py",
|
||||
"gitea_mcp_server.py (gitea_assess_master_parity)",
|
||||
),
|
||||
references=("#420", "#591", "#657"),
|
||||
),
|
||||
RestartPath(
|
||||
path_id="stale_runtime_resolver_reconnect",
|
||||
title="Stale-runtime resolver reconnect guidance",
|
||||
mechanism=(
|
||||
"The capability resolver detecting a stale serving process and "
|
||||
"reporting restart_required/stop_required for a client reconnect."
|
||||
),
|
||||
classification=CLASS_GUARDED_FAIL_CLOSED,
|
||||
guard=(
|
||||
"Report-only (#685): returns restart_required/stop_required and an "
|
||||
"exact_safe_next_action pointing at IDE/client reconnect; performs "
|
||||
"no restart, thread spawn, config touch, or os._exit."
|
||||
),
|
||||
locations=(
|
||||
"gitea_mcp_server.py (gitea_resolve_task_capability)",
|
||||
"gitea_mcp_server.py (gitea_request_mcp_reconnect)",
|
||||
"mcp_client_reconnect.py",
|
||||
),
|
||||
references=("#685", "#657", "#678"),
|
||||
),
|
||||
RestartPath(
|
||||
path_id="codex_client_reconnect_request",
|
||||
title="Sanctioned Codex/LLM reconnect request tool",
|
||||
mechanism=(
|
||||
"gitea_request_mcp_reconnect: agents invoke a report-only tool that "
|
||||
"returns namespace/profile/pid/startup SHA/master SHA/boundary "
|
||||
"status plus a typed operator blocker with exact client UI steps."
|
||||
),
|
||||
classification=CLASS_GUARDED_FAIL_CLOSED,
|
||||
guard=(
|
||||
"Report-only (#678): never restarts, kills, reloads, or edits "
|
||||
"config; recovery is always host/operator reconnect. Forbidden "
|
||||
"paths (pkill, touch, .env/config/session-state hacks) are listed "
|
||||
"and never recommended."
|
||||
),
|
||||
locations=(
|
||||
"mcp_client_reconnect.py",
|
||||
"gitea_mcp_server.py (gitea_request_mcp_reconnect)",
|
||||
),
|
||||
references=("#678", "#630", "#685", "#657"),
|
||||
),
|
||||
RestartPath(
|
||||
path_id="manual_daemon_kill",
|
||||
title="Manual daemon kill (pkill/killall/kill)",
|
||||
mechanism=(
|
||||
"Shell kills of the MCP daemon: `pkill -f mcp_server.py`, "
|
||||
"`killall`, broad `pkill -f python` sweeps, or `kill <pid>` of a "
|
||||
"daemon pid."
|
||||
),
|
||||
classification=CLASS_FORBIDDEN,
|
||||
guard=(
|
||||
"Forbidden (#630): runtime_recovery_guard classifies these as "
|
||||
"contamination and gitea_record_daemon_process_kill_attempt writes "
|
||||
"a durable marker that fails subsequent mutations closed. Operator "
|
||||
"maintenance authorization is read only from the environment, not "
|
||||
"from a tool argument."
|
||||
),
|
||||
locations=(
|
||||
"runtime_recovery_guard.py",
|
||||
"gitea_mcp_server.py (gitea_record_daemon_process_kill_attempt)",
|
||||
),
|
||||
references=("#630", "#657"),
|
||||
),
|
||||
RestartPath(
|
||||
path_id="conflict_marker_infra_stop",
|
||||
title="Startup conflict-marker infra stop",
|
||||
mechanism=(
|
||||
"The daemon entrypoint scans for unresolved merge-conflict markers "
|
||||
"at startup and stops (sys.exit(1)) if found."
|
||||
),
|
||||
classification=CLASS_GUARDED_FAIL_CLOSED,
|
||||
guard=(
|
||||
"Fail-closed startup stop, not a restart: the process exits and "
|
||||
"waits for the operator to resolve conflicts and relaunch. Never "
|
||||
"self-restarts or loops."
|
||||
),
|
||||
locations=("mcp_server.py (check_conflict_markers)",),
|
||||
references=("#657",),
|
||||
),
|
||||
RestartPath(
|
||||
path_id="ide_client_reconnect",
|
||||
title="Host/IDE MCP reconnect",
|
||||
mechanism=(
|
||||
"A manual `/mcp reconnect` (or equivalent host action) that the "
|
||||
"IDE performs to recreate the MCP client connection. Agents obtain "
|
||||
"exact UI steps via gitea_request_mcp_reconnect (#678)."
|
||||
),
|
||||
classification=CLASS_HOST_RESIDUAL,
|
||||
guard=(
|
||||
"Outside this process's control. It is the sanctioned recovery the "
|
||||
"gates point operators toward; documented as residual host "
|
||||
"behavior. No in-process code initiates it."
|
||||
),
|
||||
locations=(
|
||||
"host/IDE",
|
||||
"mcp_client_reconnect.py",
|
||||
"gitea_mcp_server.py (gitea_request_mcp_reconnect)",
|
||||
),
|
||||
references=("#584", "#656", "#657", "#678"),
|
||||
residual_host=True,
|
||||
),
|
||||
RestartPath(
|
||||
path_id="profile_switch_runtime",
|
||||
title="Runtime profile switch",
|
||||
mechanism=(
|
||||
"Switching the active execution profile at runtime "
|
||||
"(dynamic-profile mode)."
|
||||
),
|
||||
classification=CLASS_SANCTIONED_NARROW,
|
||||
guard=(
|
||||
"In-process and restart-free: runtime_switching_supported is true, "
|
||||
"so a profile switch rebinds capability without recreating the "
|
||||
"process. No restart primitive is invoked."
|
||||
),
|
||||
locations=("gitea_mcp_server.py (gitea_activate_profile)",),
|
||||
references=("#656", "#657"),
|
||||
),
|
||||
)
|
||||
|
||||
_BY_ID: dict[str, RestartPath] = {p.path_id: p for p in _RESTART_PATHS}
|
||||
|
||||
|
||||
# --- Read-only accessors ---------------------------------------------------
|
||||
|
||||
|
||||
def iter_restart_paths() -> tuple[RestartPath, ...]:
|
||||
"""Return the full inventory as an immutable tuple."""
|
||||
|
||||
return _RESTART_PATHS
|
||||
|
||||
|
||||
def restart_path_ids() -> frozenset[str]:
|
||||
"""Return the set of registered path ids."""
|
||||
|
||||
return frozenset(_BY_ID)
|
||||
|
||||
|
||||
def get_restart_path(path_id: str) -> RestartPath:
|
||||
"""Return the registered path, or raise :class:`UnknownRestartPathError`."""
|
||||
|
||||
try:
|
||||
return _BY_ID[path_id]
|
||||
except KeyError as exc:
|
||||
raise UnknownRestartPathError(
|
||||
f"unknown restart path id {path_id!r}; not in the #657 inventory"
|
||||
) from exc
|
||||
|
||||
|
||||
def paths_by_classification(classification: str) -> tuple[RestartPath, ...]:
|
||||
"""Return all registered paths with the given classification."""
|
||||
|
||||
if classification not in VALID_CLASSIFICATIONS:
|
||||
raise ValueError(f"unknown classification {classification!r}")
|
||||
return tuple(p for p in _RESTART_PATHS if p.classification == classification)
|
||||
|
||||
|
||||
def assert_restart_attempt_registered(path_id: str) -> RestartPath:
|
||||
"""Fail closed unless ``path_id`` is a registered, classified restart path.
|
||||
|
||||
LLM tools that intend to trigger any restart/reload/reconnect must name a
|
||||
registered path so an unknown/novel restart primitive cannot slip through
|
||||
silently. Forbidden and removed paths are registered too — this only
|
||||
asserts the attempt is *known*, not that it is *permitted*; callers must
|
||||
still honor the classification.
|
||||
"""
|
||||
|
||||
return get_restart_path(path_id)
|
||||
|
||||
|
||||
def assert_registry_wellformed() -> None:
|
||||
"""Validate the inventory's own invariants (fail closed on drift)."""
|
||||
|
||||
seen: set[str] = set()
|
||||
for path in _RESTART_PATHS:
|
||||
if path.path_id in seen:
|
||||
raise ValueError(f"duplicate restart path id {path.path_id!r}")
|
||||
seen.add(path.path_id)
|
||||
if path.classification not in VALID_CLASSIFICATIONS:
|
||||
raise ValueError(
|
||||
f"{path.path_id!r} has invalid classification "
|
||||
f"{path.classification!r}"
|
||||
)
|
||||
if not path.guard.strip():
|
||||
raise ValueError(f"{path.path_id!r} is missing a guard description")
|
||||
if not path.references:
|
||||
raise ValueError(f"{path.path_id!r} is missing references")
|
||||
if not path.locations:
|
||||
raise ValueError(f"{path.path_id!r} is missing locations")
|
||||
if path.classification == CLASS_HOST_RESIDUAL and not path.residual_host:
|
||||
raise ValueError(
|
||||
f"{path.path_id!r} is host_residual but residual_host is False"
|
||||
)
|
||||
|
||||
|
||||
# --- Source-tree guards ----------------------------------------------------
|
||||
|
||||
|
||||
def _repo_root(root: str | os.PathLike[str] | None = None) -> Path:
|
||||
if root is not None:
|
||||
return Path(root)
|
||||
return Path(__file__).resolve().parent
|
||||
|
||||
|
||||
def _iter_code_lines(text: str) -> Iterable[tuple[int, str]]:
|
||||
"""Yield (1-based lineno, line) for lines that are not full-line comments."""
|
||||
|
||||
for lineno, line in enumerate(text.splitlines(), start=1):
|
||||
if line.lstrip().startswith("#"):
|
||||
continue
|
||||
yield lineno, line
|
||||
|
||||
|
||||
def scan_daemon_self_replacement(
|
||||
root: str | os.PathLike[str] | None = None,
|
||||
) -> list[dict[str, object]]:
|
||||
"""Return violations where a daemon module could self-replace/self-kill.
|
||||
|
||||
Scans :data:`DAEMON_MODULES` for calls in
|
||||
:data:`DAEMON_SELF_REPLACEMENT_PRIMITIVES`. Full-line comments are ignored,
|
||||
and only call forms (with a trailing ``(``) match, so decision comments and
|
||||
docstrings that merely mention the primitives do not produce false hits.
|
||||
"""
|
||||
|
||||
repo = _repo_root(root)
|
||||
violations: list[dict[str, object]] = []
|
||||
for module in DAEMON_MODULES:
|
||||
path = repo / module
|
||||
if not path.exists():
|
||||
continue
|
||||
text = path.read_text(encoding="utf-8", errors="replace")
|
||||
for lineno, line in _iter_code_lines(text):
|
||||
for primitive in DAEMON_SELF_REPLACEMENT_PRIMITIVES:
|
||||
if primitive in line:
|
||||
violations.append(
|
||||
{
|
||||
"module": module,
|
||||
"line": lineno,
|
||||
"primitive": primitive,
|
||||
"text": line.strip(),
|
||||
}
|
||||
)
|
||||
return violations
|
||||
|
||||
|
||||
def assert_no_daemon_self_replacement(
|
||||
root: str | os.PathLike[str] | None = None,
|
||||
) -> None:
|
||||
"""Fail closed if any daemon module can restart/kill its own process."""
|
||||
|
||||
violations = scan_daemon_self_replacement(root)
|
||||
if violations:
|
||||
rendered = "; ".join(
|
||||
f"{v['module']}:{v['line']} {v['primitive']}" for v in violations
|
||||
)
|
||||
raise AssertionError(
|
||||
"MCP daemon must never self-replace/self-kill (#657); found: "
|
||||
f"{rendered}"
|
||||
)
|
||||
|
||||
|
||||
def scan_auto_restart_helper(
|
||||
root: str | os.PathLike[str] | None = None,
|
||||
) -> list[dict[str, object]]:
|
||||
"""Return occurrences of a *definition* of the legacy auto-restart helper."""
|
||||
|
||||
repo = _repo_root(root)
|
||||
needle = f"def {LEGACY_AUTO_RESTART_HELPER}"
|
||||
hits: list[dict[str, object]] = []
|
||||
for module in DAEMON_MODULES:
|
||||
path = repo / module
|
||||
if not path.exists():
|
||||
continue
|
||||
text = path.read_text(encoding="utf-8", errors="replace")
|
||||
for lineno, line in _iter_code_lines(text):
|
||||
if needle in line:
|
||||
hits.append({"module": module, "line": lineno})
|
||||
return hits
|
||||
|
||||
|
||||
def assert_auto_restart_helper_absent(
|
||||
root: str | os.PathLike[str] | None = None,
|
||||
) -> None:
|
||||
"""Fail closed if the removed ``_trigger_mcp_auto_restart`` reappears."""
|
||||
|
||||
hits = scan_auto_restart_helper(root)
|
||||
if hits:
|
||||
rendered = "; ".join(f"{h['module']}:{h['line']}" for h in hits)
|
||||
raise AssertionError(
|
||||
f"{LEGACY_AUTO_RESTART_HELPER} was removed in #685 and must not "
|
||||
f"return (#657); found definition at: {rendered}"
|
||||
)
|
||||
@@ -40,6 +40,7 @@ NON_TOOL_IDENTIFIERS: frozenset[str] = frozenset(
|
||||
"gitea_auth",
|
||||
"gitea_config",
|
||||
"gitea_mcp_server",
|
||||
"mcp_fleet_inventory",
|
||||
"mcp_server",
|
||||
"offline_mcp_helper",
|
||||
"offline_mcp_runner",
|
||||
|
||||
@@ -566,6 +566,10 @@ def build_pr_cleanup_entry(
|
||||
worktree_state=worktree_state,
|
||||
active_lock=active_lock,
|
||||
)
|
||||
planned = plan_cleanup_execution_order(
|
||||
remote_assessment=remote,
|
||||
local_assessment=local,
|
||||
)
|
||||
return {
|
||||
"pr_number": pr_number,
|
||||
"issue_number": issue_number,
|
||||
@@ -576,9 +580,63 @@ def build_pr_cleanup_entry(
|
||||
"merged": merged,
|
||||
"remote_branch": remote,
|
||||
"local_worktree": local,
|
||||
# #851: dry-run and execute share the same lifecycle order description.
|
||||
"planned_execution_order": planned,
|
||||
}
|
||||
|
||||
|
||||
def plan_cleanup_execution_order(
|
||||
*,
|
||||
remote_assessment: dict[str, Any] | None,
|
||||
local_assessment: dict[str, Any] | None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Describe independent worktree-then-reassess-then-remote cleanup order (#851).
|
||||
|
||||
Remote ownership protection remains fail-closed at execute time. A worktree
|
||||
that is independently safe to remove is never skipped merely because remote
|
||||
deletion may be blocked by that same ``worktree_binding``.
|
||||
"""
|
||||
remote = remote_assessment or {}
|
||||
local = local_assessment or {}
|
||||
steps: list[dict[str, Any]] = []
|
||||
worktree_safe = bool(local.get("safe_to_remove_worktree"))
|
||||
remote_safe = bool(remote.get("safe_to_delete_remote"))
|
||||
|
||||
if worktree_safe:
|
||||
steps.append(
|
||||
{
|
||||
"action": "remove_local_worktree",
|
||||
"reason": "independently_safe_to_remove",
|
||||
"phase": 1,
|
||||
}
|
||||
)
|
||||
if remote_safe:
|
||||
if worktree_safe:
|
||||
steps.append(
|
||||
{
|
||||
"action": "reassess_branch_ownership",
|
||||
"reason": "after_worktree_removal_clear_worktree_binding",
|
||||
"phase": 2,
|
||||
}
|
||||
)
|
||||
steps.append(
|
||||
{
|
||||
"action": "delete_remote_branch",
|
||||
"reason": "only_if_independently_safe_after_reassessment",
|
||||
"phase": 3,
|
||||
}
|
||||
)
|
||||
else:
|
||||
steps.append(
|
||||
{
|
||||
"action": "delete_remote_branch",
|
||||
"reason": "safe_to_delete_and_no_independent_worktree_removal",
|
||||
"phase": 1,
|
||||
}
|
||||
)
|
||||
return steps
|
||||
|
||||
|
||||
def build_reconciliation_report(
|
||||
*,
|
||||
project_root: str,
|
||||
|
||||
@@ -0,0 +1,791 @@
|
||||
"""Post-restart MCP reconciliation and completion proof (#662).
|
||||
|
||||
After an MCP process restart, sessions, leases, capabilities, worktrees, and
|
||||
interrupted mutations are not systematically reconciled; operators rebuild
|
||||
context from chat. This module is the pure classification core of the
|
||||
post-restart reconcile path.
|
||||
|
||||
Design rules (mirrors ``restart_coordinator`` / ``workflow_dashboard``):
|
||||
|
||||
* **Pure classification.** :func:`reconcile_after_restart` takes an already
|
||||
gathered inventory and returns a structured *completion proof*. It never
|
||||
touches the network, the filesystem, or a live process, so multi-session
|
||||
fixtures can drive every branch in unit tests.
|
||||
* **Fail closed.** Incomplete inventory never reports overall ``complete``.
|
||||
Ambiguous interrupted mutations are ``unresolved`` (never silently resumed).
|
||||
* **No blind write resume.** The proof never authorizes replaying a mutation;
|
||||
it only classifies evidence and names follow-up work.
|
||||
* **#660 soft dependency.** When durable session checkpoints are not present
|
||||
in the inventory, the checkpoint dimension is ``skipped`` with an explicit
|
||||
reason rather than inventing a schema (#660 lands separately).
|
||||
* **Log-only then enforce.** Default mode is ``log_only``. ``enforce`` sets
|
||||
``mutation_hold`` when anything remains unresolved so callers can block
|
||||
write ops until reconcile is complete or degraded mode is documented.
|
||||
|
||||
The single sanctioned gather+classify entry point is the MCP tool
|
||||
``gitea_reconcile_after_restart`` (read-only inventory gather + pure classify).
|
||||
Creating durable follow-up Gitea issues from unresolved items is an explicit
|
||||
apply step outside this pure module.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Mapping, Sequence
|
||||
from uuid import uuid4
|
||||
|
||||
import lease_lifecycle
|
||||
|
||||
RECONCILE_VERSION = "1.0.0-issue-662"
|
||||
SCHEMA_VERSION = 1
|
||||
|
||||
# Overall proof statuses.
|
||||
STATUS_COMPLETE = "complete"
|
||||
STATUS_DEGRADED = "degraded"
|
||||
STATUS_FAILED = "failed"
|
||||
|
||||
# Per-dimension item statuses.
|
||||
ITEM_RESOLVED = "resolved"
|
||||
ITEM_UNRESOLVED = "unresolved"
|
||||
ITEM_DEGRADED = "degraded"
|
||||
ITEM_SKIPPED = "skipped"
|
||||
|
||||
# Modes.
|
||||
MODE_LOG_ONLY = "log_only"
|
||||
MODE_ENFORCE = "enforce"
|
||||
|
||||
# Lease / session phases that imply a write critical section was in flight.
|
||||
MUTATING_PHASES = frozenset(
|
||||
{
|
||||
"implementing",
|
||||
"publishing",
|
||||
"merging",
|
||||
"reviewing",
|
||||
"committing",
|
||||
"pushing",
|
||||
"closing",
|
||||
"mutating",
|
||||
"critical_section",
|
||||
}
|
||||
)
|
||||
|
||||
# Dimensions the acceptance criteria require.
|
||||
DIM_SERVICE_HEALTH = "service_health"
|
||||
DIM_CLIENTS = "clients"
|
||||
DIM_SESSIONS = "sessions"
|
||||
DIM_CHECKPOINTS = "checkpoints"
|
||||
DIM_LEASES = "leases"
|
||||
DIM_CAPABILITIES = "capabilities"
|
||||
DIM_WORKTREES = "worktrees"
|
||||
DIM_MUTATIONS = "interrupted_mutations"
|
||||
DIM_DUPLICATES = "duplicates"
|
||||
DIM_QUEUE = "queue"
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _ts(dt: datetime) -> str:
|
||||
return dt.astimezone(timezone.utc).isoformat().replace("+00:00", "Z")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ReconcileItem:
|
||||
"""One dimension of the post-restart reconcile report."""
|
||||
|
||||
dimension: str
|
||||
status: str
|
||||
summary: str
|
||||
details: dict[str, Any] = field(default_factory=dict)
|
||||
follow_up_required: bool = False
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"dimension": self.dimension,
|
||||
"status": self.status,
|
||||
"summary": self.summary,
|
||||
"details": dict(self.details),
|
||||
"follow_up_required": self.follow_up_required,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class FollowUpIssue:
|
||||
"""A durable follow-up issue the apply path may create for unresolved work."""
|
||||
|
||||
title: str
|
||||
body: str
|
||||
dimension: str
|
||||
severity: str = "high"
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"title": self.title,
|
||||
"body": self.body,
|
||||
"dimension": self.dimension,
|
||||
"severity": self.severity,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartCompletionProof:
|
||||
"""Machine-readable post-restart completion proof (#662 AC2)."""
|
||||
|
||||
schema_version: int
|
||||
reconcile_version: str
|
||||
reconcile_id: str
|
||||
started_at: str
|
||||
finished_at: str
|
||||
boot_head_sha: str | None
|
||||
current_head_sha: str | None
|
||||
inventory_complete: bool
|
||||
incomplete_reasons: tuple[str, ...]
|
||||
mode: str
|
||||
mutation_hold: bool
|
||||
overall_status: str
|
||||
items: tuple[ReconcileItem, ...]
|
||||
proposed_follow_ups: tuple[FollowUpIssue, ...]
|
||||
resolved_count: int
|
||||
unresolved_count: int
|
||||
skipped_count: int
|
||||
note: str
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"schema_version": self.schema_version,
|
||||
"reconcile_version": self.reconcile_version,
|
||||
"reconcile_id": self.reconcile_id,
|
||||
"started_at": self.started_at,
|
||||
"finished_at": self.finished_at,
|
||||
"boot_head_sha": self.boot_head_sha,
|
||||
"current_head_sha": self.current_head_sha,
|
||||
"inventory_complete": self.inventory_complete,
|
||||
"incomplete_reasons": list(self.incomplete_reasons),
|
||||
"mode": self.mode,
|
||||
"mutation_hold": self.mutation_hold,
|
||||
"overall_status": self.overall_status,
|
||||
"items": [i.as_dict() for i in self.items],
|
||||
"proposed_follow_ups": [f.as_dict() for f in self.proposed_follow_ups],
|
||||
"resolved_count": self.resolved_count,
|
||||
"unresolved_count": self.unresolved_count,
|
||||
"skipped_count": self.skipped_count,
|
||||
"note": self.note,
|
||||
"links": {
|
||||
"umbrella": 655,
|
||||
"vision": 652,
|
||||
"roadmap": 653,
|
||||
"issue": 662,
|
||||
"checkpoint_schema": 660,
|
||||
"drain_proof": 661,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _item(
|
||||
dimension: str,
|
||||
status: str,
|
||||
summary: str,
|
||||
*,
|
||||
details: dict[str, Any] | None = None,
|
||||
follow_up: bool = False,
|
||||
) -> ReconcileItem:
|
||||
return ReconcileItem(
|
||||
dimension=dimension,
|
||||
status=status,
|
||||
summary=summary,
|
||||
details=dict(details or {}),
|
||||
follow_up_required=follow_up,
|
||||
)
|
||||
|
||||
|
||||
def _lease_freshness(lease: Mapping[str, Any]) -> str:
|
||||
fr = lease.get("freshness")
|
||||
if isinstance(fr, Mapping):
|
||||
return str(fr.get("freshness") or fr.get("status") or "unknown")
|
||||
if isinstance(fr, str):
|
||||
return fr
|
||||
# Fall back to pure classifier when raw lease rows are supplied.
|
||||
try:
|
||||
return str(lease_lifecycle.classify_lease_freshness(dict(lease)).get("freshness") or "unknown")
|
||||
except Exception: # noqa: BLE001 - pure path must not raise on bad rows
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _is_live_freshness(freshness: str) -> bool:
|
||||
return freshness in {"active", "live", "fresh"}
|
||||
|
||||
|
||||
def _is_mutating_phase(phase: str | None) -> bool:
|
||||
p = (phase or "").strip().lower()
|
||||
if not p:
|
||||
return False
|
||||
if p in MUTATING_PHASES:
|
||||
return True
|
||||
# Soft match for compound phases like "author_implementing".
|
||||
return any(token in p for token in MUTATING_PHASES)
|
||||
|
||||
|
||||
def _detect_interrupted_mutations(
|
||||
leases: Sequence[Mapping[str, Any]],
|
||||
pending_mutations: Sequence[Mapping[str, Any]],
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return interrupted-mutation evidence (never auto-resumes writes)."""
|
||||
found: list[dict[str, Any]] = []
|
||||
|
||||
for raw in pending_mutations or ():
|
||||
if not isinstance(raw, Mapping):
|
||||
continue
|
||||
found.append(
|
||||
{
|
||||
"source": "pending_mutation_inventory",
|
||||
"status": "unresolved",
|
||||
"phase": raw.get("phase"),
|
||||
"session_id": raw.get("session_id"),
|
||||
"work_kind": raw.get("work_kind") or raw.get("kind"),
|
||||
"work_number": raw.get("work_number") or raw.get("number"),
|
||||
"reason": raw.get("reason")
|
||||
or "pending mutation recorded across process restart",
|
||||
"resume_allowed": False,
|
||||
}
|
||||
)
|
||||
|
||||
for lease in leases or ():
|
||||
if not isinstance(lease, Mapping):
|
||||
continue
|
||||
phase = lease.get("phase")
|
||||
freshness = _lease_freshness(lease)
|
||||
if not _is_mutating_phase(str(phase) if phase is not None else None):
|
||||
continue
|
||||
# A mutating phase whose owner is not live is interrupted.
|
||||
if _is_live_freshness(freshness):
|
||||
# Still live after restart is itself surprising — flag for review.
|
||||
found.append(
|
||||
{
|
||||
"source": "lease_mutating_phase",
|
||||
"status": "unresolved",
|
||||
"phase": phase,
|
||||
"freshness": freshness,
|
||||
"lease_id": lease.get("lease_id"),
|
||||
"session_id": lease.get("session_id"),
|
||||
"work_kind": lease.get("work_kind"),
|
||||
"work_number": lease.get("work_number"),
|
||||
"worktree_path": lease.get("worktree_path"),
|
||||
"reason": (
|
||||
"mutating lease phase still classified live after restart; "
|
||||
"do not auto-resume writes"
|
||||
),
|
||||
"resume_allowed": False,
|
||||
}
|
||||
)
|
||||
else:
|
||||
found.append(
|
||||
{
|
||||
"source": "lease_mutating_phase",
|
||||
"status": "unresolved",
|
||||
"phase": phase,
|
||||
"freshness": freshness,
|
||||
"lease_id": lease.get("lease_id"),
|
||||
"session_id": lease.get("session_id"),
|
||||
"work_kind": lease.get("work_kind"),
|
||||
"work_number": lease.get("work_number"),
|
||||
"worktree_path": lease.get("worktree_path"),
|
||||
"reason": (
|
||||
f"mutating lease phase '{phase}' with non-live freshness "
|
||||
f"'{freshness}' — interrupted by restart"
|
||||
),
|
||||
"resume_allowed": False,
|
||||
}
|
||||
)
|
||||
return found
|
||||
|
||||
|
||||
def _detect_duplicate_work(
|
||||
leases: Sequence[Mapping[str, Any]],
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Surface duplicate live claims on the same work item."""
|
||||
by_work: dict[tuple[Any, Any], list[Mapping[str, Any]]] = {}
|
||||
for lease in leases or ():
|
||||
if not isinstance(lease, Mapping):
|
||||
continue
|
||||
if not _is_live_freshness(_lease_freshness(lease)):
|
||||
continue
|
||||
key = (lease.get("work_kind"), lease.get("work_number"))
|
||||
if key[0] is None or key[1] is None:
|
||||
continue
|
||||
by_work.setdefault(key, []).append(lease)
|
||||
|
||||
dups: list[dict[str, Any]] = []
|
||||
for (kind, number), rows in sorted(by_work.items(), key=lambda kv: str(kv[0])):
|
||||
if len(rows) < 2:
|
||||
continue
|
||||
dups.append(
|
||||
{
|
||||
"work_kind": kind,
|
||||
"work_number": number,
|
||||
"claim_count": len(rows),
|
||||
"session_ids": [r.get("session_id") for r in rows],
|
||||
"lease_ids": [r.get("lease_id") for r in rows],
|
||||
}
|
||||
)
|
||||
return dups
|
||||
|
||||
|
||||
def _follow_up_for_item(item: ReconcileItem) -> FollowUpIssue | None:
|
||||
if not item.follow_up_required:
|
||||
return None
|
||||
title = f"[post-restart] unresolved {item.dimension} after MCP restart"
|
||||
body = (
|
||||
f"## Post-restart reconcile follow-up (#662)\n\n"
|
||||
f"**Dimension:** `{item.dimension}`\n"
|
||||
f"**Status:** `{item.status}`\n"
|
||||
f"**Summary:** {item.summary}\n\n"
|
||||
f"```json\n{item.details!r}\n```\n\n"
|
||||
f"Parent umbrella: #655 · Vision: #652 · Roadmap: #653 · Reconcile: #662\n"
|
||||
f"Do **not** auto-resume write mutations; reconcile evidence first.\n"
|
||||
)
|
||||
return FollowUpIssue(
|
||||
title=title,
|
||||
body=body,
|
||||
dimension=item.dimension,
|
||||
severity="high" if item.dimension == DIM_MUTATIONS else "medium",
|
||||
)
|
||||
|
||||
|
||||
def reconcile_after_restart(
|
||||
inventory: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
mode: str = MODE_LOG_ONLY,
|
||||
reconcile_id: str | None = None,
|
||||
) -> RestartCompletionProof:
|
||||
"""Classify a post-restart inventory into a completion proof (#662).
|
||||
|
||||
Parameters
|
||||
----------
|
||||
inventory:
|
||||
Gathered facts. Expected keys (all optional except completeness):
|
||||
|
||||
* ``inventory_complete`` (bool) — fail closed when false
|
||||
* ``incomplete_reasons`` (list[str])
|
||||
* ``service_health`` (dict with ``healthy`` bool)
|
||||
* ``clients`` (list) — connected client descriptors
|
||||
* ``sessions`` (list)
|
||||
* ``leases`` (list, optionally with ``freshness``)
|
||||
* ``checkpoints`` (list | None) — durable session checkpoints (#660)
|
||||
* ``checkpoints_available`` (bool) — False when #660 schema absent
|
||||
* ``worktree_bindings`` (list)
|
||||
* ``pending_mutations`` (list) — explicit interrupted-mutation evidence
|
||||
* ``capabilities`` (dict with optional ``stale`` / heads)
|
||||
* ``boot_head_sha`` / ``current_head_sha``
|
||||
* ``queue_state`` (dict)
|
||||
mode:
|
||||
``log_only`` (default) or ``enforce`` (sets mutation_hold on unresolved).
|
||||
"""
|
||||
started = now or _utc_now()
|
||||
mode_norm = (mode or MODE_LOG_ONLY).strip().lower()
|
||||
if mode_norm not in {MODE_LOG_ONLY, MODE_ENFORCE}:
|
||||
mode_norm = MODE_LOG_ONLY
|
||||
|
||||
inventory_complete = bool(inventory.get("inventory_complete", False))
|
||||
incomplete_reasons = tuple(
|
||||
str(r) for r in (inventory.get("incomplete_reasons") or []) if str(r).strip()
|
||||
)
|
||||
|
||||
items: list[ReconcileItem] = []
|
||||
|
||||
# --- service health -------------------------------------------------
|
||||
health = inventory.get("service_health") or {}
|
||||
if not isinstance(health, Mapping):
|
||||
health = {}
|
||||
if not inventory_complete and "service_health" not in inventory:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_SERVICE_HEALTH,
|
||||
ITEM_UNRESOLVED,
|
||||
"service health unknown because inventory is incomplete",
|
||||
details={"inventory_complete": False},
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
elif health.get("healthy") is True:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_SERVICE_HEALTH,
|
||||
ITEM_RESOLVED,
|
||||
"service health verified",
|
||||
details=dict(health),
|
||||
)
|
||||
)
|
||||
elif health.get("healthy") is False:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_SERVICE_HEALTH,
|
||||
ITEM_UNRESOLVED,
|
||||
"service health check failed",
|
||||
details=dict(health),
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_SERVICE_HEALTH,
|
||||
ITEM_DEGRADED,
|
||||
"service health not reported; treating as degraded",
|
||||
details=dict(health),
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
|
||||
# --- clients --------------------------------------------------------
|
||||
clients = list(inventory.get("clients") or [])
|
||||
disconnected = [
|
||||
c
|
||||
for c in clients
|
||||
if isinstance(c, Mapping) and c.get("connected") is False
|
||||
]
|
||||
if "clients" not in inventory:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_CLIENTS,
|
||||
ITEM_SKIPPED,
|
||||
"client inventory not supplied",
|
||||
details={},
|
||||
)
|
||||
)
|
||||
elif disconnected:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_CLIENTS,
|
||||
ITEM_UNRESOLVED,
|
||||
f"{len(disconnected)} disconnected client(s) need reconnect",
|
||||
details={"disconnected": disconnected, "total": len(clients)},
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_CLIENTS,
|
||||
ITEM_RESOLVED,
|
||||
f"{len(clients)} client(s) accounted for",
|
||||
details={"total": len(clients)},
|
||||
)
|
||||
)
|
||||
|
||||
# --- sessions -------------------------------------------------------
|
||||
sessions = [s for s in (inventory.get("sessions") or []) if isinstance(s, Mapping)]
|
||||
orphan_sessions = [
|
||||
s
|
||||
for s in sessions
|
||||
if str(s.get("status") or "").lower() == "active"
|
||||
and s.get("pid") is not None
|
||||
and not lease_lifecycle.is_process_alive(s.get("pid"))
|
||||
]
|
||||
if orphan_sessions:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_SESSIONS,
|
||||
ITEM_UNRESOLVED,
|
||||
f"{len(orphan_sessions)} active session row(s) with dead owner pid",
|
||||
details={
|
||||
"orphan_session_ids": [s.get("session_id") for s in orphan_sessions],
|
||||
"total_sessions": len(sessions),
|
||||
},
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_SESSIONS,
|
||||
ITEM_RESOLVED,
|
||||
f"{len(sessions)} session row(s) reconciled (no dead-pid orphans)",
|
||||
details={"total_sessions": len(sessions)},
|
||||
)
|
||||
)
|
||||
|
||||
# --- checkpoints (#660 soft) ----------------------------------------
|
||||
checkpoints_available = inventory.get("checkpoints_available")
|
||||
checkpoints = inventory.get("checkpoints")
|
||||
if checkpoints_available is False or (
|
||||
checkpoints is None and "checkpoints" not in inventory
|
||||
):
|
||||
items.append(
|
||||
_item(
|
||||
DIM_CHECKPOINTS,
|
||||
ITEM_SKIPPED,
|
||||
"durable session checkpoint schema not available yet (#660)",
|
||||
details={"depends_on": 660},
|
||||
)
|
||||
)
|
||||
else:
|
||||
cp_list = [c for c in (checkpoints or []) if isinstance(c, Mapping)]
|
||||
stale_cp = [c for c in cp_list if c.get("stale") or c.get("invalid")]
|
||||
if stale_cp:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_CHECKPOINTS,
|
||||
ITEM_UNRESOLVED,
|
||||
f"{len(stale_cp)} checkpoint(s) invalid or stale vs live state",
|
||||
details={"stale_count": len(stale_cp), "total": len(cp_list)},
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_CHECKPOINTS,
|
||||
ITEM_RESOLVED,
|
||||
f"{len(cp_list)} checkpoint(s) consistent with live state",
|
||||
details={"total": len(cp_list)},
|
||||
)
|
||||
)
|
||||
|
||||
# --- leases / locks -------------------------------------------------
|
||||
leases = [L for L in (inventory.get("leases") or []) if isinstance(L, Mapping)]
|
||||
live_leases = [L for L in leases if _is_live_freshness(_lease_freshness(L))]
|
||||
items.append(
|
||||
_item(
|
||||
DIM_LEASES,
|
||||
ITEM_RESOLVED if inventory_complete else ITEM_DEGRADED,
|
||||
f"{len(live_leases)} live lease(s) of {len(leases)} inventoried",
|
||||
details={
|
||||
"live_count": len(live_leases),
|
||||
"total": len(leases),
|
||||
"live_lease_ids": [L.get("lease_id") for L in live_leases],
|
||||
},
|
||||
follow_up=not inventory_complete,
|
||||
)
|
||||
)
|
||||
|
||||
# --- capabilities / stale runtime -----------------------------------
|
||||
caps = inventory.get("capabilities") or {}
|
||||
if not isinstance(caps, Mapping):
|
||||
caps = {}
|
||||
if caps.get("stale") is True:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_CAPABILITIES,
|
||||
ITEM_UNRESOLVED,
|
||||
"runtime code is stale vs on-disk master; restart did not reach parity",
|
||||
details=dict(caps),
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_CAPABILITIES,
|
||||
ITEM_RESOLVED,
|
||||
"capability/runtime parity acceptable",
|
||||
details=dict(caps) if caps else {"stale": False},
|
||||
)
|
||||
)
|
||||
|
||||
# --- worktrees ------------------------------------------------------
|
||||
bindings = [
|
||||
b for b in (inventory.get("worktree_bindings") or []) if isinstance(b, Mapping)
|
||||
]
|
||||
missing_wt = [
|
||||
b
|
||||
for b in bindings
|
||||
if b.get("missing") is True or b.get("exists") is False
|
||||
]
|
||||
if "worktree_bindings" not in inventory:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_WORKTREES,
|
||||
ITEM_SKIPPED,
|
||||
"worktree binding inventory not supplied",
|
||||
)
|
||||
)
|
||||
elif missing_wt:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_WORKTREES,
|
||||
ITEM_UNRESOLVED,
|
||||
f"{len(missing_wt)} worktree binding(s) missing on disk",
|
||||
details={"missing": missing_wt, "total": len(bindings)},
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_WORKTREES,
|
||||
ITEM_RESOLVED,
|
||||
f"{len(bindings)} worktree binding(s) present",
|
||||
details={"total": len(bindings)},
|
||||
)
|
||||
)
|
||||
|
||||
# --- interrupted mutations (AC4) ------------------------------------
|
||||
pending = [
|
||||
m
|
||||
for m in (inventory.get("pending_mutations") or [])
|
||||
if isinstance(m, Mapping)
|
||||
]
|
||||
interrupted = _detect_interrupted_mutations(leases, pending)
|
||||
if interrupted:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_MUTATIONS,
|
||||
ITEM_UNRESOLVED,
|
||||
f"{len(interrupted)} interrupted mutation(s); write resume forbidden",
|
||||
details={"interrupted": interrupted},
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_MUTATIONS,
|
||||
ITEM_RESOLVED,
|
||||
"no interrupted mutations detected",
|
||||
details={"interrupted": []},
|
||||
)
|
||||
)
|
||||
|
||||
# --- duplicates -----------------------------------------------------
|
||||
dups = _detect_duplicate_work(leases)
|
||||
if dups:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_DUPLICATES,
|
||||
ITEM_UNRESOLVED,
|
||||
f"{len(dups)} work item(s) have multiple live claims",
|
||||
details={"duplicates": dups},
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_DUPLICATES,
|
||||
ITEM_RESOLVED,
|
||||
"no duplicate live claims detected",
|
||||
details={"duplicates": []},
|
||||
)
|
||||
)
|
||||
|
||||
# --- queue ----------------------------------------------------------
|
||||
queue = inventory.get("queue_state")
|
||||
if queue is None:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_QUEUE,
|
||||
ITEM_SKIPPED,
|
||||
"allocator queue state not supplied",
|
||||
)
|
||||
)
|
||||
elif isinstance(queue, Mapping) and queue.get("safe_to_resume") is False:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_QUEUE,
|
||||
ITEM_UNRESOLVED,
|
||||
"allocator queue not safe to resume",
|
||||
details=dict(queue),
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_QUEUE,
|
||||
ITEM_RESOLVED,
|
||||
"allocator queue state acceptable",
|
||||
details=dict(queue) if isinstance(queue, Mapping) else {},
|
||||
)
|
||||
)
|
||||
|
||||
# Incomplete inventory always degrades the whole proof.
|
||||
if not inventory_complete:
|
||||
# Ensure at least one follow-up names the incomplete inventory.
|
||||
items.append(
|
||||
_item(
|
||||
"inventory",
|
||||
ITEM_UNRESOLVED,
|
||||
"control-plane inventory incomplete; reconcile cannot claim success",
|
||||
details={"reasons": list(incomplete_reasons)},
|
||||
follow_up=True,
|
||||
)
|
||||
)
|
||||
|
||||
resolved = sum(1 for i in items if i.status == ITEM_RESOLVED)
|
||||
unresolved = sum(1 for i in items if i.status in {ITEM_UNRESOLVED, ITEM_DEGRADED})
|
||||
skipped = sum(1 for i in items if i.status == ITEM_SKIPPED)
|
||||
|
||||
if not inventory_complete or any(i.status == ITEM_UNRESOLVED for i in items):
|
||||
if any(i.status == ITEM_UNRESOLVED for i in items) and inventory_complete:
|
||||
overall = STATUS_DEGRADED
|
||||
elif not inventory_complete:
|
||||
overall = STATUS_FAILED
|
||||
else:
|
||||
overall = STATUS_DEGRADED
|
||||
elif any(i.status == ITEM_DEGRADED for i in items):
|
||||
overall = STATUS_DEGRADED
|
||||
else:
|
||||
overall = STATUS_COMPLETE
|
||||
|
||||
# Enforce mode holds mutations whenever anything is unresolved/failed.
|
||||
mutation_hold = False
|
||||
if mode_norm == MODE_ENFORCE and overall in {STATUS_DEGRADED, STATUS_FAILED}:
|
||||
mutation_hold = True
|
||||
if mode_norm == MODE_ENFORCE and any(
|
||||
i.dimension == DIM_MUTATIONS and i.status == ITEM_UNRESOLVED for i in items
|
||||
):
|
||||
mutation_hold = True
|
||||
|
||||
follow_ups = tuple(
|
||||
fu for i in items if (fu := _follow_up_for_item(i)) is not None
|
||||
)
|
||||
|
||||
finished = _utc_now() if now is None else now
|
||||
note = (
|
||||
"Read-only completion proof. Never auto-resumes write mutations. "
|
||||
"Unresolved items require durable follow-up before claiming clean restart. "
|
||||
f"Mode={mode_norm}."
|
||||
)
|
||||
|
||||
return RestartCompletionProof(
|
||||
schema_version=SCHEMA_VERSION,
|
||||
reconcile_version=RECONCILE_VERSION,
|
||||
reconcile_id=(reconcile_id or f"reconcile-{uuid4().hex[:12]}"),
|
||||
started_at=_ts(started),
|
||||
finished_at=_ts(finished),
|
||||
boot_head_sha=(
|
||||
str(inventory.get("boot_head_sha")).strip()
|
||||
if inventory.get("boot_head_sha")
|
||||
else None
|
||||
),
|
||||
current_head_sha=(
|
||||
str(inventory.get("current_head_sha")).strip()
|
||||
if inventory.get("current_head_sha")
|
||||
else None
|
||||
),
|
||||
inventory_complete=inventory_complete,
|
||||
incomplete_reasons=incomplete_reasons,
|
||||
mode=mode_norm,
|
||||
mutation_hold=mutation_hold,
|
||||
overall_status=overall,
|
||||
items=tuple(items),
|
||||
proposed_follow_ups=follow_ups,
|
||||
resolved_count=resolved,
|
||||
unresolved_count=unresolved,
|
||||
skipped_count=skipped,
|
||||
note=note,
|
||||
)
|
||||
|
||||
|
||||
def mutations_allowed(proof: RestartCompletionProof | Mapping[str, Any] | None) -> bool:
|
||||
"""Return whether write mutations may proceed under the given proof."""
|
||||
if proof is None:
|
||||
return True # no proof yet → caller decides; enforce path sets hold
|
||||
if isinstance(proof, RestartCompletionProof):
|
||||
return not proof.mutation_hold
|
||||
if isinstance(proof, Mapping):
|
||||
return not bool(proof.get("mutation_hold"))
|
||||
return True
|
||||
@@ -0,0 +1,3 @@
|
||||
[pytest]
|
||||
testpaths = tests
|
||||
norecursedirs = branches .git venv __pycache__ graphify-out
|
||||
@@ -0,0 +1,583 @@
|
||||
"""Scoped MCP recovery playbook (#669).
|
||||
|
||||
Operational recovery must prefer the *narrowest* action that can fix the
|
||||
symptom. Full MCP / host restarts are last-resort rungs on a documented
|
||||
ladder; the coordinator refuses those rungs unless a prior attempt log
|
||||
shows narrower recoveries already failed (or break-glass is authorized).
|
||||
|
||||
This module is pure classification and recommendation:
|
||||
|
||||
* No network, filesystem, or process I/O.
|
||||
* Never restarts anything.
|
||||
* Narrow recovery *execution* is delegated to existing tools/docs (linked
|
||||
per rung) — the playbook records which rung to try next and whether
|
||||
escalation to a broad restart is allowed.
|
||||
|
||||
Design lineage: umbrella #655, class matrix #663, coordinator #658,
|
||||
auto-reconnect #584, stale-runtime #610, contamination #630, audit #665.
|
||||
Vision #652 / roadmap #653.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from enum import Enum
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
PLAYBOOK_VERSION = "1.0.0-issue-669"
|
||||
|
||||
# Attempt outcomes that count as "tried and insufficient" for escalation.
|
||||
INSUFFICIENT_OUTCOMES = frozenset(
|
||||
{
|
||||
"failed",
|
||||
"insufficient",
|
||||
"denied",
|
||||
"unresolved",
|
||||
"timeout",
|
||||
"error",
|
||||
}
|
||||
)
|
||||
|
||||
# Break-glass / operator override still records that the ladder was skipped.
|
||||
OUTCOME_BREAK_GLASS = "break_glass"
|
||||
OUTCOME_SUCCESS = "success"
|
||||
OUTCOME_SKIPPED = "skipped"
|
||||
|
||||
|
||||
class RecoveryAction(str, Enum):
|
||||
"""Ordered recovery ladder (narrow → broad)."""
|
||||
|
||||
CLIENT_RECONNECT = "client_reconnect"
|
||||
CAPABILITY_REFRESH = "capability_refresh"
|
||||
SESSION_RECONNECT = "session_reconnect"
|
||||
CONFIGURATION_RELOAD = "configuration_reload"
|
||||
LEASE_RECOVERY = "lease_recovery"
|
||||
WORKER_RESTART = "worker_restart"
|
||||
ROLE_RUNTIME_RESTART = "role_runtime_restart"
|
||||
CONNECTOR_RESTART = "connector_restart"
|
||||
ROLLING_MCP_RESTART = "rolling_mcp_restart"
|
||||
FULL_MCP_RESTART = "full_mcp_restart"
|
||||
HOST_RESTART = "host_restart"
|
||||
|
||||
|
||||
# Classes that require a prior narrow-attempt log (unless break-glass).
|
||||
BROAD_RESTART_ACTIONS: frozenset[RecoveryAction] = frozenset(
|
||||
{
|
||||
RecoveryAction.ROLLING_MCP_RESTART,
|
||||
RecoveryAction.FULL_MCP_RESTART,
|
||||
RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
)
|
||||
|
||||
# Map #663 restart_class strings onto playbook actions.
|
||||
RESTART_CLASS_TO_ACTION: dict[str, RecoveryAction] = {
|
||||
"client_reconnect": RecoveryAction.CLIENT_RECONNECT,
|
||||
"session_reconnect": RecoveryAction.SESSION_RECONNECT,
|
||||
"configuration_reload": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"worker_restart": RecoveryAction.WORKER_RESTART,
|
||||
"role_runtime_restart": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
"connector_restart": RecoveryAction.CONNECTOR_RESTART,
|
||||
"rolling_mcp_restart": RecoveryAction.ROLLING_MCP_RESTART,
|
||||
"full_mcp_restart": RecoveryAction.FULL_MCP_RESTART,
|
||||
"host_restart": RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecoveryRung:
|
||||
"""One rung on the recovery ladder."""
|
||||
|
||||
action: RecoveryAction
|
||||
rank: int
|
||||
summary: str
|
||||
# Existing implementation or explicit delegation target.
|
||||
implementation: str
|
||||
issue_links: tuple[str, ...]
|
||||
self_service: bool
|
||||
# Restart-class permission when this rung is requested via coordinator.
|
||||
restart_class: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"action": self.action.value,
|
||||
"rank": self.rank,
|
||||
"summary": self.summary,
|
||||
"implementation": self.implementation,
|
||||
"issue_links": list(self.issue_links),
|
||||
"self_service": self.self_service,
|
||||
"restart_class": self.restart_class,
|
||||
}
|
||||
|
||||
|
||||
# Canonical ladder. Rank 0 is narrowest.
|
||||
RECOVERY_LADDER: tuple[RecoveryRung, ...] = (
|
||||
RecoveryRung(
|
||||
RecoveryAction.CLIENT_RECONNECT,
|
||||
0,
|
||||
"Reconnect the IDE/client MCP transport (EOF / transport flap).",
|
||||
"Host auto-reconnect or explicit client reconnect; "
|
||||
"docs/mcp-namespace-eof-recovery.md",
|
||||
("#584", "#655"),
|
||||
True,
|
||||
"client_reconnect",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CAPABILITY_REFRESH,
|
||||
1,
|
||||
"Re-resolve task capability and clear stale permission context.",
|
||||
"Delegated: gitea_resolve_task_capability + gitea_whoami "
|
||||
"(no process change).",
|
||||
("#610", "#685", "#655"),
|
||||
True,
|
||||
None,
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.SESSION_RECONNECT,
|
||||
2,
|
||||
"Rebind identity, workspace, and namespace for one session.",
|
||||
"Delegated: gitea_get_runtime_context + explicit worktree_path "
|
||||
"rebind (#618); docs/mcp-namespace-health.md",
|
||||
("#543", "#618", "#655"),
|
||||
True,
|
||||
"session_reconnect",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CONFIGURATION_RELOAD,
|
||||
3,
|
||||
"Gracefully reload configuration without replacing the daemon.",
|
||||
"restart_coordinator class configuration_reload; console "
|
||||
"system.reload_namespace (#642).",
|
||||
("#642", "#663", "#655"),
|
||||
False,
|
||||
"configuration_reload",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.LEASE_RECOVERY,
|
||||
4,
|
||||
"Recover or rebind stale leases/locks without a process restart.",
|
||||
"Delegated: issue lock recovery / lease lifecycle paths "
|
||||
"(#702, #753, #790).",
|
||||
("#702", "#753", "#790", "#655"),
|
||||
False,
|
||||
None,
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.WORKER_RESTART,
|
||||
5,
|
||||
"Restart one worker after its own lease and mutation scope drains.",
|
||||
"restart_coordinator class worker_restart (#663).",
|
||||
("#663", "#655"),
|
||||
False,
|
||||
"worker_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
6,
|
||||
"Restart one role runtime and re-probe that namespace only.",
|
||||
"restart_coordinator class role_runtime_restart; console "
|
||||
"system.restart_namespace (#642).",
|
||||
("#642", "#663", "#655"),
|
||||
False,
|
||||
"role_runtime_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CONNECTOR_RESTART,
|
||||
7,
|
||||
"Restart one connector while unrelated runtimes stay available.",
|
||||
"restart_coordinator class connector_restart (#663).",
|
||||
("#663", "#655"),
|
||||
False,
|
||||
"connector_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.ROLLING_MCP_RESTART,
|
||||
8,
|
||||
"Drain/restart/verify one instance at a time (HA path).",
|
||||
"restart_coordinator class rolling_mcp_restart; design #668.",
|
||||
("#668", "#663", "#655"),
|
||||
False,
|
||||
"rolling_mcp_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.FULL_MCP_RESTART,
|
||||
9,
|
||||
"Full stable-control MCP process restart after verified full drain.",
|
||||
"restart_coordinator class full_mcp_restart; requires attempt log "
|
||||
"unless break-glass (#669).",
|
||||
("#658", "#661", "#663", "#669", "#655"),
|
||||
False,
|
||||
"full_mcp_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.HOST_RESTART,
|
||||
10,
|
||||
"Host/infrastructure restart — broadest last-resort action.",
|
||||
"restart_coordinator class host_restart; operator-owned.",
|
||||
("#663", "#669", "#655"),
|
||||
False,
|
||||
"host_restart",
|
||||
),
|
||||
)
|
||||
|
||||
_LADDER_BY_ACTION: dict[RecoveryAction, RecoveryRung] = {
|
||||
rung.action: rung for rung in RECOVERY_LADDER
|
||||
}
|
||||
|
||||
# Symptom tokens → preferred first rung (decision tree, #663 lineage).
|
||||
SYMPTOM_TO_FIRST_ACTION: dict[str, RecoveryAction] = {
|
||||
"transport_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||
"client_closing_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||
"transport_flap": RecoveryAction.CLIENT_RECONNECT,
|
||||
"namespace_disconnected": RecoveryAction.CLIENT_RECONNECT,
|
||||
"stale_capability": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"permission_stale": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"runtime_reconnect_required": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"stale_runtime": RecoveryAction.SESSION_RECONNECT,
|
||||
"worktree_unbound": RecoveryAction.SESSION_RECONNECT,
|
||||
"namespace_unhealthy": RecoveryAction.SESSION_RECONNECT,
|
||||
"config_drift": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"profile_misbound": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"stale_lease": RecoveryAction.LEASE_RECOVERY,
|
||||
"dead_pid_lock": RecoveryAction.LEASE_RECOVERY,
|
||||
"orphan_worktree": RecoveryAction.LEASE_RECOVERY,
|
||||
"single_worker_stuck": RecoveryAction.WORKER_RESTART,
|
||||
"role_runtime_dead": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
"connector_dead": RecoveryAction.CONNECTOR_RESTART,
|
||||
"ha_instance_unhealthy": RecoveryAction.ROLLING_MCP_RESTART,
|
||||
"daemon_corrupt": RecoveryAction.FULL_MCP_RESTART,
|
||||
"full_process_deadlock": RecoveryAction.FULL_MCP_RESTART,
|
||||
"host_unresponsive": RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def resolve_action(value: RecoveryAction | str) -> RecoveryAction:
|
||||
"""Resolve a recovery action or fail closed for unknown values."""
|
||||
if isinstance(value, RecoveryAction):
|
||||
return value
|
||||
text = str(value or "").strip()
|
||||
# Accept #663 restart_class aliases.
|
||||
if text in RESTART_CLASS_TO_ACTION:
|
||||
return RESTART_CLASS_TO_ACTION[text]
|
||||
try:
|
||||
return RecoveryAction(text)
|
||||
except ValueError as exc:
|
||||
raise ValueError(
|
||||
f"unknown recovery action {value!r}; deny (fail closed, #669)"
|
||||
) from exc
|
||||
|
||||
|
||||
def ladder_rank(action: RecoveryAction | str) -> int:
|
||||
resolved = resolve_action(action)
|
||||
return _LADDER_BY_ACTION[resolved].rank
|
||||
|
||||
|
||||
def rung_for(action: RecoveryAction | str) -> RecoveryRung:
|
||||
return _LADDER_BY_ACTION[resolve_action(action)]
|
||||
|
||||
|
||||
def normalize_attempt(raw: Mapping[str, Any]) -> dict[str, Any] | None:
|
||||
"""Normalize one prior-recovery attempt record; return None if unusable."""
|
||||
if not isinstance(raw, Mapping):
|
||||
return None
|
||||
action_raw = raw.get("action") or raw.get("recovery_action") or raw.get(
|
||||
"restart_class"
|
||||
)
|
||||
if not action_raw:
|
||||
return None
|
||||
try:
|
||||
action = resolve_action(str(action_raw))
|
||||
except ValueError:
|
||||
return None
|
||||
outcome = str(
|
||||
raw.get("outcome") or raw.get("status") or raw.get("result") or ""
|
||||
).strip().lower()
|
||||
if not outcome:
|
||||
return None
|
||||
recorded_at = raw.get("recorded_at") or raw.get("at") or raw.get("timestamp")
|
||||
reason = str(raw.get("reason") or raw.get("detail") or "").strip()
|
||||
actor = str(raw.get("actor") or raw.get("session_id") or "").strip()
|
||||
return {
|
||||
"action": action.value,
|
||||
"outcome": outcome,
|
||||
"reason": reason,
|
||||
"actor": actor,
|
||||
"recorded_at": recorded_at,
|
||||
"rank": ladder_rank(action),
|
||||
"raw": dict(raw),
|
||||
}
|
||||
|
||||
|
||||
def normalize_attempt_log(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return usable attempt records in ladder order."""
|
||||
out: list[dict[str, Any]] = []
|
||||
for raw in attempts or ():
|
||||
norm = normalize_attempt(raw)
|
||||
if norm is not None:
|
||||
out.append(norm)
|
||||
out.sort(key=lambda a: (a["rank"], str(a.get("recorded_at") or "")))
|
||||
return out
|
||||
|
||||
|
||||
def narrower_insufficient_attempts(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
*,
|
||||
requested: RecoveryAction | str,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return prior attempts narrower than *requested* that were insufficient."""
|
||||
target_rank = ladder_rank(requested)
|
||||
usable = []
|
||||
for attempt in normalize_attempt_log(attempts):
|
||||
if attempt["rank"] >= target_rank:
|
||||
continue
|
||||
if attempt["outcome"] in INSUFFICIENT_OUTCOMES:
|
||||
usable.append(attempt)
|
||||
return usable
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EscalationAssessment:
|
||||
"""Whether a requested broad recovery may proceed given the attempt log."""
|
||||
|
||||
requested_action: str
|
||||
allowed: bool
|
||||
require_attempt_log: bool
|
||||
break_glass: bool
|
||||
reasons: list[str] = field(default_factory=list)
|
||||
qualifying_attempts: list[dict[str, Any]] = field(default_factory=list)
|
||||
recommended_next: list[dict[str, Any]] = field(default_factory=list)
|
||||
playbook_version: str = PLAYBOOK_VERSION
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"playbook_version": self.playbook_version,
|
||||
"requested_action": self.requested_action,
|
||||
"allowed": self.allowed,
|
||||
"require_attempt_log": self.require_attempt_log,
|
||||
"break_glass": self.break_glass,
|
||||
"reasons": list(self.reasons),
|
||||
"qualifying_attempts": list(self.qualifying_attempts),
|
||||
"recommended_next": list(self.recommended_next),
|
||||
}
|
||||
|
||||
|
||||
def assess_escalation(
|
||||
requested: RecoveryAction | str,
|
||||
*,
|
||||
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||
break_glass: bool = False,
|
||||
) -> EscalationAssessment:
|
||||
"""Gate broad restarts on a prior narrow-attempt log (#669 AC3).
|
||||
|
||||
Narrow / mid-ladder actions do not require a prior attempt log.
|
||||
``full_mcp_restart``, ``host_restart``, and ``rolling_mcp_restart``
|
||||
require at least one *insufficient* narrower attempt unless
|
||||
``break_glass`` is true.
|
||||
"""
|
||||
action = resolve_action(requested)
|
||||
require_log = action in BROAD_RESTART_ACTIONS
|
||||
reasons: list[str] = []
|
||||
qualifying = narrower_insufficient_attempts(
|
||||
prior_recovery_attempts, requested=action
|
||||
)
|
||||
|
||||
if not require_log:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=False,
|
||||
break_glass=bool(break_glass),
|
||||
reasons=["narrow recovery; attempt log not required"],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
if break_glass:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=True,
|
||||
break_glass=True,
|
||||
reasons=[
|
||||
"break-glass authorized; broad restart permitted without "
|
||||
"narrow-attempt log (#669)"
|
||||
],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
if qualifying:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=True,
|
||||
break_glass=False,
|
||||
reasons=[
|
||||
f"{len(qualifying)} narrower recovery attempt(s) recorded as "
|
||||
"insufficient; escalation permitted"
|
||||
],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
# Deny: recommend the next untried narrow rung(s).
|
||||
recommended = recommend_actions(
|
||||
symptoms=(),
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
max_actions=3,
|
||||
)
|
||||
reasons.append(
|
||||
f"{action.value} requires a prior attempt log of insufficient "
|
||||
"narrower recoveries (or break-glass); none found — deny (fail "
|
||||
"closed, #669)"
|
||||
)
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=False,
|
||||
require_attempt_log=True,
|
||||
break_glass=False,
|
||||
reasons=reasons,
|
||||
qualifying_attempts=[],
|
||||
recommended_next=recommended.get("recommended_actions") or [],
|
||||
)
|
||||
|
||||
|
||||
def recommend_actions(
|
||||
*,
|
||||
symptoms: Sequence[str] = (),
|
||||
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||
max_actions: int = 5,
|
||||
) -> dict[str, Any]:
|
||||
"""Return ordered recommended recovery actions for the given symptoms.
|
||||
|
||||
Soft mode (rollout): recommendations only — callers decide whether to
|
||||
hard-gate. Hard mode for broad restarts is :func:`assess_escalation`.
|
||||
"""
|
||||
attempted_success = {
|
||||
a["action"]
|
||||
for a in normalize_attempt_log(prior_recovery_attempts)
|
||||
if a["outcome"] == OUTCOME_SUCCESS
|
||||
}
|
||||
attempted_any = {
|
||||
a["action"] for a in normalize_attempt_log(prior_recovery_attempts)
|
||||
}
|
||||
|
||||
first_actions: list[RecoveryAction] = []
|
||||
for symptom in symptoms:
|
||||
key = str(symptom or "").strip().lower().replace(" ", "_").replace("-", "_")
|
||||
mapped = SYMPTOM_TO_FIRST_ACTION.get(key)
|
||||
if mapped is not None and mapped not in first_actions:
|
||||
first_actions.append(mapped)
|
||||
|
||||
# Default entry: client reconnect then walk the ladder.
|
||||
if not first_actions:
|
||||
first_actions = [RecoveryAction.CLIENT_RECONNECT]
|
||||
|
||||
recommended: list[dict[str, Any]] = []
|
||||
seen: set[str] = set()
|
||||
min_rank = min(ladder_rank(a) for a in first_actions)
|
||||
|
||||
for rung in RECOVERY_LADDER:
|
||||
if rung.rank < min_rank:
|
||||
continue
|
||||
if rung.action.value in attempted_success:
|
||||
continue
|
||||
if rung.action.value in seen:
|
||||
continue
|
||||
# Prefer rungs not yet attempted; still list previously-failed ones
|
||||
# only if nothing else remains.
|
||||
entry = rung.as_dict()
|
||||
entry["already_attempted"] = rung.action.value in attempted_any
|
||||
recommended.append(entry)
|
||||
seen.add(rung.action.value)
|
||||
if len(recommended) >= max(1, int(max_actions)):
|
||||
break
|
||||
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"symptoms": [str(s) for s in symptoms],
|
||||
"recommended_actions": recommended,
|
||||
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||
"read_only": True,
|
||||
"hard_gate_note": (
|
||||
"Broad restarts (rolling/full/host) still require "
|
||||
"assess_escalation / coordinator attempt-log enforcement."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def build_attempt_record(
|
||||
action: RecoveryAction | str,
|
||||
*,
|
||||
outcome: str,
|
||||
reason: str = "",
|
||||
actor: str = "",
|
||||
recorded_at: str | None = None,
|
||||
extra: Mapping[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build a durable-shaped attempt log entry for inventory/audit (#665)."""
|
||||
resolved = resolve_action(action)
|
||||
record = {
|
||||
"action": resolved.value,
|
||||
"outcome": str(outcome or "").strip().lower(),
|
||||
"reason": str(reason or "").strip(),
|
||||
"actor": str(actor or "").strip(),
|
||||
"recorded_at": recorded_at or _utc_now().isoformat(),
|
||||
"rank": ladder_rank(resolved),
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
}
|
||||
if extra:
|
||||
record["extra"] = dict(extra)
|
||||
return record
|
||||
|
||||
|
||||
def recovery_metrics(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
) -> dict[str, Any]:
|
||||
"""Compute the fraction of recoveries that avoided full/host restart.
|
||||
|
||||
A recovery *episode* is approximated as one attempt with
|
||||
``outcome=success``. Successes on non-broad rungs count as avoided full
|
||||
restart; successes on full/host count as full-restart recoveries.
|
||||
"""
|
||||
norms = normalize_attempt_log(attempts)
|
||||
successes = [a for a in norms if a["outcome"] == OUTCOME_SUCCESS]
|
||||
broad_success = [
|
||||
a
|
||||
for a in successes
|
||||
if resolve_action(a["action"])
|
||||
in {RecoveryAction.FULL_MCP_RESTART, RecoveryAction.HOST_RESTART}
|
||||
]
|
||||
avoided = [a for a in successes if a not in broad_success]
|
||||
total = len(successes)
|
||||
fraction_avoided = (len(avoided) / total) if total else None
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"attempts_total": len(norms),
|
||||
"successes_total": total,
|
||||
"successes_avoided_full_restart": len(avoided),
|
||||
"successes_full_or_host_restart": len(broad_success),
|
||||
"fraction_avoided_full_restart": fraction_avoided,
|
||||
"insufficient_attempts": sum(
|
||||
1 for a in norms if a["outcome"] in INSUFFICIENT_OUTCOMES
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def ladder_document() -> dict[str, Any]:
|
||||
"""Machine-readable ladder for docs/tools inventory."""
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"parent_issues": ["#655", "#652", "#653"],
|
||||
"enforcement_issue": "#669",
|
||||
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||
"broad_restart_actions": [a.value for a in sorted(BROAD_RESTART_ACTIONS, key=lambda x: x.value)],
|
||||
"insufficient_outcomes": sorted(INSUFFICIENT_OUTCOMES),
|
||||
"symptom_map": {k: v.value for k, v in sorted(SYMPTOM_TO_FIRST_ACTION.items())},
|
||||
}
|
||||
@@ -9,6 +9,8 @@ cryptography==49.0.0
|
||||
h11==0.16.0
|
||||
httpcore==1.0.9
|
||||
httpx==0.28.1
|
||||
# Starlette 1.3.x TestClient prefers httpx2; plain httpx remains for MCP/runtime (#682).
|
||||
httpx2==2.9.1
|
||||
httpx-sse==0.4.3
|
||||
idna==3.18
|
||||
iniconfig==2.3.0
|
||||
|
||||
@@ -0,0 +1,851 @@
|
||||
"""MCP restart coordinator and impact analysis (#658 / #669).
|
||||
|
||||
Before any sanctioned MCP restart, a central coordinator must evaluate the
|
||||
live control-plane state — active sessions, leases/locks, in-flight issue/PR
|
||||
work, mutations, worktrees, and recovery history — and produce an *impact
|
||||
preview* so operators (and the web console, #642/#652) can see the blast
|
||||
radius **before** concurrent LLM work is disrupted.
|
||||
|
||||
Design rules (mirrors the read-only posture of ``workflow_dashboard`` /
|
||||
``lease_lifecycle``):
|
||||
|
||||
* **Pure classification.** :func:`evaluate_restart_impact` takes an already
|
||||
gathered inventory and returns a structured report. It never touches the
|
||||
network, the filesystem, or a live process, so multi-session fixtures can
|
||||
drive every branch in unit tests. The coordinator *never restarts anything*;
|
||||
a mutative apply path is a later child gated by a drain proof (non-goal here).
|
||||
* **Fail closed.** If the inventory is not explicitly complete, the verdict is
|
||||
``unsafe`` / deny — an incomplete evaluation must never green-light a restart.
|
||||
* **Narrow-first (#669).** Broad classes (rolling / full / host) require a
|
||||
prior attempt log of insufficient narrower recoveries unless break-glass is
|
||||
authorized. See :mod:`recovery_playbook`.
|
||||
* **No secrets.** Session ids, pids, and profiles are operational metadata, not
|
||||
credentials; nothing secret flows through this module.
|
||||
|
||||
The single sanctioned entry point post-#657 is the MCP tool
|
||||
``gitea_request_mcp_restart`` (dry-run by default), which gathers the inventory
|
||||
from the #613 control-plane DB and calls :func:`evaluate_restart_impact`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from enum import Enum
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
import lease_lifecycle
|
||||
import recovery_playbook
|
||||
|
||||
COORDINATOR_VERSION = "1.2.0-issue-669"
|
||||
|
||||
# Restart verdicts. Exactly the three the acceptance criteria name.
|
||||
VERDICT_SAFE = "safe"
|
||||
VERDICT_UNSAFE = "unsafe"
|
||||
VERDICT_OVERRIDE = "override"
|
||||
|
||||
# Blast-radius severity bands.
|
||||
BLAST_NONE = "none"
|
||||
BLAST_LOW = "low"
|
||||
BLAST_MEDIUM = "medium"
|
||||
BLAST_HIGH = "high"
|
||||
|
||||
# A live lease with a live owner process is treated as active in-flight work.
|
||||
LEASE_FRESHNESS_LIVE = "active"
|
||||
|
||||
# Default staleness window for a session heartbeat (seconds). A session whose
|
||||
# last heartbeat is older than this is not counted as live even if its row is
|
||||
# still marked ``active`` — it is assumed dead/detached.
|
||||
DEFAULT_SESSION_HEARTBEAT_STALE_SECONDS = 900
|
||||
|
||||
|
||||
class RestartClass(str, Enum):
|
||||
"""The only restart/recovery classes accepted by the coordinator."""
|
||||
|
||||
CLIENT_RECONNECT = "client_reconnect"
|
||||
SESSION_RECONNECT = "session_reconnect"
|
||||
WORKER_RESTART = "worker_restart"
|
||||
ROLE_RUNTIME_RESTART = "role_runtime_restart"
|
||||
CONNECTOR_RESTART = "connector_restart"
|
||||
CONFIGURATION_RELOAD = "configuration_reload"
|
||||
ROLLING_MCP_RESTART = "rolling_mcp_restart"
|
||||
FULL_MCP_RESTART = "full_mcp_restart"
|
||||
HOST_RESTART = "host_restart"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartClassPolicy:
|
||||
"""Least-privilege policy for one :class:`RestartClass`."""
|
||||
|
||||
restart_class: RestartClass
|
||||
required_permission: str
|
||||
expected_blast_radius: str
|
||||
drain_requirement: str
|
||||
full_drain_required: bool
|
||||
approval_requirement: str
|
||||
audit_requirement: str
|
||||
recovery_behavior: str
|
||||
request_roles: tuple[str, ...]
|
||||
execution_roles: tuple[str, ...]
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"restart_class": self.restart_class.value,
|
||||
"required_permission": self.required_permission,
|
||||
"expected_blast_radius": self.expected_blast_radius,
|
||||
"drain_requirement": self.drain_requirement,
|
||||
"full_drain_required": self.full_drain_required,
|
||||
"approval_requirement": self.approval_requirement,
|
||||
"audit_requirement": self.audit_requirement,
|
||||
"recovery_behavior": self.recovery_behavior,
|
||||
"request_roles": list(self.request_roles),
|
||||
"execution_roles": list(self.execution_roles),
|
||||
}
|
||||
|
||||
|
||||
WORKER_ROLES = ("author", "reviewer", "merger", "reconciler")
|
||||
CONTROL_ROLES = ("controller", "operator", "admin")
|
||||
ALL_REQUEST_ROLES = WORKER_ROLES + CONTROL_ROLES
|
||||
|
||||
RESTART_CLASS_POLICIES: dict[RestartClass, RestartClassPolicy] = {
|
||||
RestartClass.CLIENT_RECONNECT: RestartClassPolicy(
|
||||
RestartClass.CLIENT_RECONNECT,
|
||||
"mcp.reconnect.client",
|
||||
BLAST_NONE,
|
||||
"none",
|
||||
False,
|
||||
"self_service",
|
||||
"record class, actor, client namespace, reason, and outcome",
|
||||
"Reconnect only the caller's client transport; no daemon or peer session changes.",
|
||||
ALL_REQUEST_ROLES,
|
||||
ALL_REQUEST_ROLES,
|
||||
),
|
||||
RestartClass.SESSION_RECONNECT: RestartClassPolicy(
|
||||
RestartClass.SESSION_RECONNECT,
|
||||
"mcp.reconnect.session",
|
||||
BLAST_LOW,
|
||||
"requesting_session_safe_point",
|
||||
False,
|
||||
"self_service",
|
||||
"record class, actor, session id, reason, and outcome",
|
||||
"Rebind identity, capability, and workspace state for one session.",
|
||||
ALL_REQUEST_ROLES,
|
||||
ALL_REQUEST_ROLES,
|
||||
),
|
||||
RestartClass.WORKER_RESTART: RestartClassPolicy(
|
||||
RestartClass.WORKER_RESTART,
|
||||
"mcp.restart.worker.request",
|
||||
BLAST_LOW,
|
||||
"target_worker",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, target worker, approval, drain proof, and outcome",
|
||||
"Restart one worker after its own lease and mutation scope is drained.",
|
||||
ALL_REQUEST_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.ROLE_RUNTIME_RESTART: RestartClassPolicy(
|
||||
RestartClass.ROLE_RUNTIME_RESTART,
|
||||
"mcp.restart.role_runtime.request",
|
||||
BLAST_MEDIUM,
|
||||
"target_role_runtime",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, role namespace, approval, drain proof, and outcome",
|
||||
"Restart only the selected role runtime and then re-probe that namespace.",
|
||||
ALL_REQUEST_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.CONNECTOR_RESTART: RestartClassPolicy(
|
||||
RestartClass.CONNECTOR_RESTART,
|
||||
"mcp.restart.connector.request",
|
||||
BLAST_MEDIUM,
|
||||
"target_connector",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, connector id, approval, drain proof, and outcome",
|
||||
"Restart one connector while unrelated role runtimes remain available.",
|
||||
ALL_REQUEST_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.CONFIGURATION_RELOAD: RestartClassPolicy(
|
||||
RestartClass.CONFIGURATION_RELOAD,
|
||||
"mcp.reload.configuration.request",
|
||||
BLAST_LOW,
|
||||
"mutation_quiesce",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, configuration revision, approval, and outcome",
|
||||
"Gracefully reload configuration without replacing the daemon process.",
|
||||
ALL_REQUEST_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.ROLLING_MCP_RESTART: RestartClassPolicy(
|
||||
RestartClass.ROLLING_MCP_RESTART,
|
||||
"mcp.restart.rolling.request",
|
||||
BLAST_MEDIUM,
|
||||
"one_instance_at_a_time",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, instance order, approval, per-instance drains, and outcome",
|
||||
"Drain, restart, verify, and restore one instance before advancing to the next.",
|
||||
CONTROL_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.FULL_MCP_RESTART: RestartClassPolicy(
|
||||
RestartClass.FULL_MCP_RESTART,
|
||||
"mcp.restart.full.request",
|
||||
BLAST_HIGH,
|
||||
"all_sessions_and_mutations",
|
||||
True,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, full impact report, approval, drain proof, and outcome",
|
||||
"Stop and restore the complete MCP runtime only after a verified full drain.",
|
||||
CONTROL_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.HOST_RESTART: RestartClassPolicy(
|
||||
RestartClass.HOST_RESTART,
|
||||
"mcp.restart.host.request",
|
||||
BLAST_HIGH,
|
||||
"all_host_work",
|
||||
True,
|
||||
"controller_approval_plus_infrastructure_operator",
|
||||
"record class, actor, host, incident or change id, approval, drain proof, and outcome",
|
||||
"Hand off to infrastructure ownership; reconcile every runtime after the host returns.",
|
||||
("controller", "operator", "admin"),
|
||||
("operator", "admin"),
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def resolve_restart_class(value: RestartClass | str) -> RestartClass:
|
||||
"""Resolve a restart class or fail closed for an unknown value."""
|
||||
|
||||
if isinstance(value, RestartClass):
|
||||
return value
|
||||
try:
|
||||
return RestartClass(str(value).strip())
|
||||
except ValueError as exc:
|
||||
raise ValueError(f"unknown restart class {value!r}; deny (fail closed)") from exc
|
||||
|
||||
|
||||
def restart_class_policy(value: RestartClass | str) -> RestartClassPolicy:
|
||||
"""Return the canonical policy for *value*."""
|
||||
|
||||
return RESTART_CLASS_POLICIES[resolve_restart_class(value)]
|
||||
|
||||
|
||||
def permissions_for_role(role: str | None) -> tuple[str, ...]:
|
||||
"""Return request permissions granted to a workflow role by this policy."""
|
||||
|
||||
normalized = str(role or "").strip().lower()
|
||||
return tuple(
|
||||
policy.required_permission
|
||||
for policy in RESTART_CLASS_POLICIES.values()
|
||||
if normalized in policy.request_roles
|
||||
)
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _parse_ts(value: str | None) -> datetime | None:
|
||||
return lease_lifecycle._parse_ts(value)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SessionImpact:
|
||||
"""One MCP session a restart would terminate."""
|
||||
|
||||
session_id: str
|
||||
role: str | None
|
||||
profile: str | None
|
||||
pid: int | None
|
||||
status: str | None
|
||||
alive: bool | None
|
||||
heartbeat_stale: bool
|
||||
is_requester: bool
|
||||
live: bool
|
||||
connector: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"session_id": self.session_id,
|
||||
"role": self.role,
|
||||
"profile": self.profile,
|
||||
"pid": self.pid,
|
||||
"status": self.status,
|
||||
"alive": self.alive,
|
||||
"heartbeat_stale": self.heartbeat_stale,
|
||||
"is_requester": self.is_requester,
|
||||
"live": self.live,
|
||||
"connector": self.connector,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LeaseImpact:
|
||||
"""One control-plane lease a restart would disrupt."""
|
||||
|
||||
lease_id: str | None
|
||||
session_id: str | None
|
||||
role: str | None
|
||||
phase: str | None
|
||||
freshness: str | None
|
||||
work_kind: str | None
|
||||
work_number: int | None
|
||||
worktree_path: str | None
|
||||
disruptive: bool
|
||||
is_mutation: bool
|
||||
is_critical_section: bool
|
||||
connector: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"lease_id": self.lease_id,
|
||||
"session_id": self.session_id,
|
||||
"role": self.role,
|
||||
"phase": self.phase,
|
||||
"freshness": self.freshness,
|
||||
"work_kind": self.work_kind,
|
||||
"work_number": self.work_number,
|
||||
"worktree_path": self.worktree_path,
|
||||
"disruptive": self.disruptive,
|
||||
"is_mutation": self.is_mutation,
|
||||
"is_critical_section": self.is_critical_section,
|
||||
"connector": self.connector,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartImpactReport:
|
||||
"""Impact preview DTO returned to the console / operator (#642/#652)."""
|
||||
|
||||
coordinator_version: str
|
||||
restart_class: str
|
||||
restart_policy: dict[str, Any]
|
||||
policy_enforced: bool
|
||||
permission_authorized: bool
|
||||
role_authorized: bool
|
||||
approval_satisfied: bool
|
||||
authorization_reasons: list[str]
|
||||
evaluated_at: str
|
||||
dry_run: bool
|
||||
restart_performed: bool
|
||||
inventory_complete: bool
|
||||
verdict: str
|
||||
allow_restart: bool
|
||||
override_would_allow: bool
|
||||
operator_override: bool
|
||||
blast_radius: str
|
||||
reasons: list[str]
|
||||
affected_sessions: list[SessionImpact]
|
||||
affected_leases: list[LeaseImpact]
|
||||
critical_sections: list[LeaseImpact]
|
||||
affected_issues: list[int]
|
||||
affected_prs: list[int]
|
||||
mutations: list[LeaseImpact]
|
||||
terminal_lock: dict[str, Any] | None
|
||||
ack_state: dict[str, str]
|
||||
prior_recovery_attempts: list[dict[str, Any]]
|
||||
counts: dict[str, int]
|
||||
audit_record: dict[str, Any]
|
||||
incomplete_reasons: list[str] = field(default_factory=list)
|
||||
# #669 playbook escalation gate (attempt-log enforcement).
|
||||
playbook_escalation: dict[str, Any] = field(default_factory=dict)
|
||||
attempt_log_satisfied: bool = True
|
||||
break_glass: bool = False
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"coordinator_version": self.coordinator_version,
|
||||
"restart_class": self.restart_class,
|
||||
"restart_policy": dict(self.restart_policy),
|
||||
"policy_enforced": self.policy_enforced,
|
||||
"permission_authorized": self.permission_authorized,
|
||||
"role_authorized": self.role_authorized,
|
||||
"approval_satisfied": self.approval_satisfied,
|
||||
"authorization_reasons": list(self.authorization_reasons),
|
||||
"evaluated_at": self.evaluated_at,
|
||||
"dry_run": self.dry_run,
|
||||
"restart_performed": self.restart_performed,
|
||||
"inventory_complete": self.inventory_complete,
|
||||
"incomplete_reasons": list(self.incomplete_reasons),
|
||||
"verdict": self.verdict,
|
||||
"allow_restart": self.allow_restart,
|
||||
"override_would_allow": self.override_would_allow,
|
||||
"operator_override": self.operator_override,
|
||||
"blast_radius": self.blast_radius,
|
||||
"reasons": list(self.reasons),
|
||||
"affected_sessions": [s.as_dict() for s in self.affected_sessions],
|
||||
"affected_leases": [l.as_dict() for l in self.affected_leases],
|
||||
"critical_sections": [l.as_dict() for l in self.critical_sections],
|
||||
"affected_issues": list(self.affected_issues),
|
||||
"affected_prs": list(self.affected_prs),
|
||||
"mutations": [l.as_dict() for l in self.mutations],
|
||||
"terminal_lock": self.terminal_lock,
|
||||
"ack_state": dict(self.ack_state),
|
||||
"prior_recovery_attempts": list(self.prior_recovery_attempts),
|
||||
"counts": dict(self.counts),
|
||||
"audit_record": dict(self.audit_record),
|
||||
"playbook_escalation": dict(self.playbook_escalation),
|
||||
"attempt_log_satisfied": self.attempt_log_satisfied,
|
||||
"break_glass": self.break_glass,
|
||||
}
|
||||
|
||||
|
||||
def _classify_session(
|
||||
row: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime,
|
||||
requesting_session_id: str | None,
|
||||
heartbeat_stale_seconds: int,
|
||||
) -> SessionImpact:
|
||||
session_id = str(row.get("session_id") or "")
|
||||
pid = row.get("pid")
|
||||
status = (row.get("status") or "").strip().lower() or None
|
||||
alive = lease_lifecycle.is_process_alive(pid) if pid is not None else None
|
||||
hb = _parse_ts(row.get("last_heartbeat_at"))
|
||||
heartbeat_stale = bool(
|
||||
hb is not None and (now - hb).total_seconds() > heartbeat_stale_seconds
|
||||
)
|
||||
live = bool(status == "active" and alive is not False and not heartbeat_stale)
|
||||
return SessionImpact(
|
||||
session_id=session_id,
|
||||
role=row.get("role"),
|
||||
profile=row.get("profile"),
|
||||
pid=pid,
|
||||
status=status,
|
||||
alive=alive,
|
||||
heartbeat_stale=heartbeat_stale,
|
||||
is_requester=bool(
|
||||
requesting_session_id and session_id == requesting_session_id
|
||||
),
|
||||
live=live,
|
||||
connector=(str(row.get("connector") or "").strip() or None),
|
||||
)
|
||||
|
||||
|
||||
# Lease phases that represent an active mutation in flight (as opposed to a
|
||||
# mere allocation/claim with no work committed yet). An active lease in any of
|
||||
# these phases is a critical section a restart must not sever.
|
||||
_MUTATING_PHASES = frozenset(
|
||||
{
|
||||
"implementing",
|
||||
"publishing",
|
||||
"pushing",
|
||||
"committing",
|
||||
"reviewing",
|
||||
"merging",
|
||||
"reconciling",
|
||||
"conflict_fix",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _classify_lease(row: Mapping[str, Any]) -> LeaseImpact:
|
||||
freshness_obj = row.get("freshness")
|
||||
if isinstance(freshness_obj, Mapping):
|
||||
freshness = str(freshness_obj.get("freshness") or "").strip().lower() or None
|
||||
else:
|
||||
freshness = str(freshness_obj or "").strip().lower() or None
|
||||
phase = (row.get("phase") or "").strip().lower() or None
|
||||
worktree = row.get("worktree_path")
|
||||
disruptive = freshness == LEASE_FRESHNESS_LIVE
|
||||
# A live lease is a mutation-in-flight if it carries an author worktree or
|
||||
# its phase names a mutating step. All disruptive leases are critical
|
||||
# sections a restart would sever regardless.
|
||||
is_mutation = bool(
|
||||
disruptive and (bool(worktree) or (phase in _MUTATING_PHASES))
|
||||
)
|
||||
number = row.get("work_number")
|
||||
try:
|
||||
number = int(number) if number is not None else None
|
||||
except (TypeError, ValueError):
|
||||
number = None
|
||||
return LeaseImpact(
|
||||
lease_id=row.get("lease_id"),
|
||||
session_id=row.get("session_id"),
|
||||
role=row.get("role"),
|
||||
phase=phase,
|
||||
freshness=freshness,
|
||||
work_kind=(str(row.get("work_kind") or "").strip().lower() or None),
|
||||
work_number=number,
|
||||
worktree_path=worktree,
|
||||
disruptive=disruptive,
|
||||
is_mutation=is_mutation,
|
||||
is_critical_section=disruptive,
|
||||
connector=(str(row.get("connector") or "").strip() or None),
|
||||
)
|
||||
|
||||
|
||||
def _blast_radius(*, session_count: int, work_count: int, mutation_count: int) -> str:
|
||||
if mutation_count > 0 or work_count >= 3 or session_count >= 3:
|
||||
return BLAST_HIGH
|
||||
if work_count > 0 or session_count == 2:
|
||||
return BLAST_MEDIUM
|
||||
if session_count == 1:
|
||||
return BLAST_LOW
|
||||
return BLAST_NONE
|
||||
|
||||
|
||||
def evaluate_restart_impact(
|
||||
inventory: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
operator_override: bool = False,
|
||||
requesting_session_id: str | None = None,
|
||||
dry_run: bool = True,
|
||||
session_heartbeat_stale_seconds: int = DEFAULT_SESSION_HEARTBEAT_STALE_SECONDS,
|
||||
restart_class: RestartClass | str | None = None,
|
||||
requester_role: str | None = None,
|
||||
requester_permissions: Sequence[str] | None = None,
|
||||
controller_approved: bool = False,
|
||||
operator_authorized: bool = False,
|
||||
target_session_id: str | None = None,
|
||||
target_role: str | None = None,
|
||||
target_connector: str | None = None,
|
||||
break_glass: bool = False,
|
||||
) -> RestartImpactReport:
|
||||
"""Evaluate a proposed MCP restart and return an impact preview.
|
||||
|
||||
``inventory`` is a mapping with:
|
||||
|
||||
* ``sessions`` — session rows (session_id, role, profile, pid, status,
|
||||
last_heartbeat_at).
|
||||
* ``leases`` — control-plane lease rows, each ideally carrying an enriched
|
||||
``freshness`` dict (as :func:`lease_lifecycle.list_active_leases` returns);
|
||||
a bare string freshness is also accepted.
|
||||
* ``terminal_lock`` — the active terminal (merge) lock, if any.
|
||||
* ``prior_recovery_attempts`` — narrower recovery attempts already tried
|
||||
(e.g. sanctioned reconnects) so the operator sees escalation history.
|
||||
* ``inventory_complete`` — bool. **Must** be explicitly True; a missing or
|
||||
falsy value forces a deny (fail closed).
|
||||
* ``incomplete_reasons`` — optional reasons the inventory is incomplete.
|
||||
|
||||
The coordinator never restarts anything: ``restart_performed`` is always
|
||||
False and the mutative apply path is a later drain-gated child.
|
||||
"""
|
||||
moment = now or _utc_now()
|
||||
reasons: list[str] = []
|
||||
authorization_reasons: list[str] = []
|
||||
|
||||
# ``None`` preserves the pre-#663 impact-only API for callers that have not
|
||||
# yet been migrated. All MCP requests pass an explicit class and therefore
|
||||
# take the fail-closed policy path.
|
||||
policy_enforced = restart_class is not None
|
||||
try:
|
||||
resolved_class = resolve_restart_class(
|
||||
restart_class or RestartClass.FULL_MCP_RESTART
|
||||
)
|
||||
policy = RESTART_CLASS_POLICIES[resolved_class]
|
||||
unknown_class = False
|
||||
except ValueError as exc:
|
||||
resolved_class = None
|
||||
policy = None
|
||||
unknown_class = True
|
||||
authorization_reasons.append(str(exc))
|
||||
|
||||
normalized_role = str(requester_role or "").strip().lower()
|
||||
granted = {str(p).strip() for p in (requester_permissions or ())}
|
||||
if policy_enforced and policy is not None:
|
||||
permission_authorized = policy.required_permission in granted
|
||||
role_authorized = normalized_role in policy.request_roles
|
||||
if not permission_authorized:
|
||||
authorization_reasons.append(
|
||||
f"missing required permission {policy.required_permission!r}"
|
||||
)
|
||||
if not role_authorized:
|
||||
authorization_reasons.append(
|
||||
f"role {normalized_role or 'unknown'!r} may not request "
|
||||
f"{policy.restart_class.value}"
|
||||
)
|
||||
elif unknown_class:
|
||||
permission_authorized = False
|
||||
role_authorized = False
|
||||
else:
|
||||
permission_authorized = True
|
||||
role_authorized = True
|
||||
|
||||
if policy_enforced and policy is not None:
|
||||
approval = policy.approval_requirement
|
||||
if approval == "self_service":
|
||||
approval_satisfied = True
|
||||
elif approval == "controller_approval_plus_infrastructure_operator":
|
||||
approval_satisfied = bool(controller_approved and operator_authorized)
|
||||
else:
|
||||
approval_satisfied = bool(controller_approved)
|
||||
if not approval_satisfied:
|
||||
authorization_reasons.append(
|
||||
f"approval requirement not satisfied: {approval}"
|
||||
)
|
||||
elif unknown_class:
|
||||
approval_satisfied = False
|
||||
else:
|
||||
approval_satisfied = True
|
||||
|
||||
inventory_complete = bool(inventory.get("inventory_complete", False))
|
||||
incomplete_reasons = [str(r) for r in (inventory.get("incomplete_reasons") or [])]
|
||||
|
||||
sessions_raw: Sequence[Mapping[str, Any]] = inventory.get("sessions") or []
|
||||
leases_raw: Sequence[Mapping[str, Any]] = inventory.get("leases") or []
|
||||
terminal_lock = inventory.get("terminal_lock") or None
|
||||
prior_recovery_attempts = [
|
||||
dict(a) for a in (inventory.get("prior_recovery_attempts") or [])
|
||||
]
|
||||
|
||||
# #669: broad restarts require a prior narrow-attempt log unless break-glass.
|
||||
playbook_escalation: dict[str, Any] = {}
|
||||
attempt_log_satisfied = True
|
||||
if policy_enforced and resolved_class is not None:
|
||||
try:
|
||||
escalation = recovery_playbook.assess_escalation(
|
||||
resolved_class.value,
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
break_glass=bool(break_glass),
|
||||
)
|
||||
playbook_escalation = escalation.as_dict()
|
||||
attempt_log_satisfied = bool(escalation.allowed)
|
||||
if not attempt_log_satisfied:
|
||||
authorization_reasons.extend(list(escalation.reasons))
|
||||
except ValueError as exc:
|
||||
# Unknown mapping should never happen for enum values; fail closed.
|
||||
attempt_log_satisfied = False
|
||||
playbook_escalation = {
|
||||
"allowed": False,
|
||||
"reasons": [str(exc)],
|
||||
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||
}
|
||||
authorization_reasons.append(str(exc))
|
||||
|
||||
session_impacts = [
|
||||
_classify_session(
|
||||
s,
|
||||
now=moment,
|
||||
requesting_session_id=requesting_session_id,
|
||||
heartbeat_stale_seconds=session_heartbeat_stale_seconds,
|
||||
)
|
||||
for s in sessions_raw
|
||||
]
|
||||
lease_impacts = [_classify_lease(l) for l in leases_raw]
|
||||
|
||||
# Route impact through the selected class. Narrow classes never inherit a
|
||||
# full-runtime drain merely because unrelated work exists.
|
||||
target_complete = True
|
||||
if resolved_class in {
|
||||
RestartClass.CLIENT_RECONNECT,
|
||||
RestartClass.SESSION_RECONNECT,
|
||||
RestartClass.CONFIGURATION_RELOAD,
|
||||
}:
|
||||
scoped_sessions: list[SessionImpact] = []
|
||||
scoped_leases: list[LeaseImpact] = []
|
||||
elif resolved_class == RestartClass.WORKER_RESTART:
|
||||
selected_session = (target_session_id or "").strip()
|
||||
target_complete = bool(selected_session)
|
||||
scoped_sessions = [
|
||||
s for s in session_impacts if s.session_id == selected_session
|
||||
]
|
||||
scoped_leases = [
|
||||
l for l in lease_impacts if l.session_id == selected_session
|
||||
]
|
||||
elif resolved_class == RestartClass.ROLE_RUNTIME_RESTART:
|
||||
selected_role = (target_role or "").strip().lower()
|
||||
target_complete = bool(selected_role)
|
||||
scoped_sessions = [
|
||||
s for s in session_impacts if str(s.role or "").lower() == selected_role
|
||||
]
|
||||
scoped_leases = [
|
||||
l for l in lease_impacts if str(l.role or "").lower() == selected_role
|
||||
]
|
||||
elif resolved_class == RestartClass.CONNECTOR_RESTART:
|
||||
selected_connector = (target_connector or "").strip()
|
||||
target_complete = bool(selected_connector)
|
||||
scoped_sessions = [
|
||||
s for s in session_impacts if s.connector == selected_connector
|
||||
]
|
||||
scoped_leases = [
|
||||
l for l in lease_impacts if l.connector == selected_connector
|
||||
]
|
||||
else:
|
||||
scoped_sessions = list(session_impacts)
|
||||
scoped_leases = list(lease_impacts)
|
||||
|
||||
if policy_enforced and not target_complete:
|
||||
authorization_reasons.append(
|
||||
f"target required for {resolved_class.value if resolved_class else 'unknown class'}"
|
||||
)
|
||||
|
||||
other_live_sessions = [
|
||||
s for s in scoped_sessions if s.live and not s.is_requester
|
||||
]
|
||||
disruptive_leases = [l for l in scoped_leases if l.disruptive]
|
||||
critical_sections = [l for l in scoped_leases if l.is_critical_section]
|
||||
mutations = [l for l in scoped_leases if l.is_mutation]
|
||||
terminal_lock_in_scope = (
|
||||
terminal_lock
|
||||
if resolved_class
|
||||
not in {
|
||||
RestartClass.CLIENT_RECONNECT,
|
||||
RestartClass.SESSION_RECONNECT,
|
||||
}
|
||||
else None
|
||||
)
|
||||
|
||||
affected_issues = sorted(
|
||||
{
|
||||
l.work_number
|
||||
for l in disruptive_leases
|
||||
if l.work_kind == "issue" and l.work_number is not None
|
||||
}
|
||||
)
|
||||
affected_prs = sorted(
|
||||
{
|
||||
l.work_number
|
||||
for l in disruptive_leases
|
||||
if l.work_kind == "pr" and l.work_number is not None
|
||||
}
|
||||
)
|
||||
|
||||
disruptive = bool(
|
||||
disruptive_leases or other_live_sessions or terminal_lock_in_scope
|
||||
)
|
||||
|
||||
authorization_ok = bool(
|
||||
not unknown_class
|
||||
and permission_authorized
|
||||
and role_authorized
|
||||
and approval_satisfied
|
||||
and target_complete
|
||||
and attempt_log_satisfied
|
||||
)
|
||||
|
||||
if policy_enforced and not authorization_ok:
|
||||
verdict = VERDICT_UNSAFE
|
||||
allow_restart = False
|
||||
reasons.append("restart class authorization denied (fail closed)")
|
||||
reasons.extend(authorization_reasons)
|
||||
elif not inventory_complete:
|
||||
verdict = VERDICT_UNSAFE
|
||||
allow_restart = False
|
||||
reasons.append(
|
||||
"inventory incomplete: restart evaluation cannot confirm blast "
|
||||
"radius — deny (fail closed, #658)"
|
||||
)
|
||||
reasons.extend(incomplete_reasons)
|
||||
elif not disruptive:
|
||||
verdict = VERDICT_SAFE
|
||||
allow_restart = True
|
||||
reasons.append("no other live sessions, live leases, or terminal lock")
|
||||
elif operator_override:
|
||||
verdict = VERDICT_OVERRIDE
|
||||
allow_restart = True
|
||||
reasons.append(
|
||||
"live work present; operator override accepts the blast radius"
|
||||
)
|
||||
else:
|
||||
verdict = VERDICT_UNSAFE
|
||||
allow_restart = False
|
||||
reasons.append(
|
||||
"live work would be disrupted; restart denied without operator "
|
||||
"override"
|
||||
)
|
||||
|
||||
if critical_sections and inventory_complete:
|
||||
reasons.append(
|
||||
f"{len(critical_sections)} critical section(s) in flight "
|
||||
"(active lease with a live owner)"
|
||||
)
|
||||
if terminal_lock_in_scope:
|
||||
reasons.append("active terminal (merge) lock present")
|
||||
|
||||
override_would_allow = bool(inventory_complete and disruptive)
|
||||
|
||||
blast_radius = _blast_radius(
|
||||
session_count=len(other_live_sessions),
|
||||
work_count=len(affected_issues) + len(affected_prs),
|
||||
mutation_count=len(mutations),
|
||||
)
|
||||
|
||||
# Acknowledgement is a later child (drain protocol); expose per-session
|
||||
# placeholders so the console can render the ack column now.
|
||||
ack_state = {s.session_id: "pending" for s in other_live_sessions}
|
||||
|
||||
counts = {
|
||||
"sessions_total": len(session_impacts),
|
||||
"sessions_live_other": len(other_live_sessions),
|
||||
"leases_total": len(lease_impacts),
|
||||
"leases_disruptive": len(disruptive_leases),
|
||||
"critical_sections": len(critical_sections),
|
||||
"mutations": len(mutations),
|
||||
"affected_issues": len(affected_issues),
|
||||
"affected_prs": len(affected_prs),
|
||||
"prior_recovery_attempts": len(prior_recovery_attempts),
|
||||
"attempt_log_satisfied": attempt_log_satisfied,
|
||||
}
|
||||
|
||||
audit_record = {
|
||||
"event": "restart_impact_evaluated",
|
||||
"coordinator_version": COORDINATOR_VERSION,
|
||||
"restart_class": (
|
||||
resolved_class.value if resolved_class else str(restart_class or "")
|
||||
),
|
||||
"required_permission": (
|
||||
policy.required_permission if policy is not None else None
|
||||
),
|
||||
"evaluated_at": moment.isoformat(),
|
||||
"dry_run": dry_run,
|
||||
"operator_override": bool(operator_override),
|
||||
"requesting_session_id": requesting_session_id,
|
||||
"inventory_complete": inventory_complete,
|
||||
"verdict": verdict,
|
||||
"allow_restart": allow_restart,
|
||||
"blast_radius": blast_radius,
|
||||
"counts": counts,
|
||||
"attempt_log_satisfied": attempt_log_satisfied,
|
||||
"break_glass": bool(break_glass),
|
||||
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||
}
|
||||
|
||||
return RestartImpactReport(
|
||||
coordinator_version=COORDINATOR_VERSION,
|
||||
restart_class=(
|
||||
resolved_class.value if resolved_class else str(restart_class or "")
|
||||
),
|
||||
restart_policy=policy.as_dict() if policy is not None else {},
|
||||
policy_enforced=policy_enforced,
|
||||
permission_authorized=permission_authorized,
|
||||
role_authorized=role_authorized,
|
||||
approval_satisfied=approval_satisfied,
|
||||
authorization_reasons=authorization_reasons,
|
||||
evaluated_at=moment.isoformat(),
|
||||
dry_run=dry_run,
|
||||
restart_performed=False,
|
||||
inventory_complete=inventory_complete,
|
||||
verdict=verdict,
|
||||
allow_restart=allow_restart,
|
||||
override_would_allow=override_would_allow,
|
||||
operator_override=bool(operator_override),
|
||||
blast_radius=blast_radius,
|
||||
reasons=reasons,
|
||||
affected_sessions=session_impacts,
|
||||
affected_leases=lease_impacts,
|
||||
critical_sections=critical_sections,
|
||||
affected_issues=affected_issues,
|
||||
affected_prs=affected_prs,
|
||||
mutations=mutations,
|
||||
terminal_lock=(
|
||||
dict(terminal_lock_in_scope)
|
||||
if isinstance(terminal_lock_in_scope, Mapping)
|
||||
else terminal_lock_in_scope
|
||||
),
|
||||
ack_state=ack_state,
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
counts=counts,
|
||||
audit_record=audit_record,
|
||||
incomplete_reasons=incomplete_reasons,
|
||||
playbook_escalation=playbook_escalation,
|
||||
attempt_log_satisfied=attempt_log_satisfied,
|
||||
break_glass=bool(break_glass),
|
||||
)
|
||||
@@ -73,6 +73,8 @@ AUTHOR_TASKS = frozenset({
|
||||
"claim_issue",
|
||||
"create_branch",
|
||||
"push_branch",
|
||||
"bootstrap_author_issue_worktree",
|
||||
"gitea_bootstrap_author_issue_worktree",
|
||||
"create_pr",
|
||||
"comment_pr",
|
||||
"address_pr_change_requests",
|
||||
|
||||
+10
-8
@@ -43,19 +43,21 @@ repo_root="$(cd "$script_dir/.." && pwd)"
|
||||
|
||||
# Enforce issue-linked, traceable branch names (issue → branch → worktree → PR).
|
||||
if [[ "$allow_unlinked" -eq 0 ]]; then
|
||||
locked_branch=$(python3 -c "
|
||||
if [[ "$dry_run" -eq 0 ]] && [[ ! "$branch" =~ ^review/pr-[0-9]+-.+ ]]; then
|
||||
locked_branch=$(python3 -c "
|
||||
import sys
|
||||
sys.path.insert(0, '$repo_root')
|
||||
import issue_lock_store
|
||||
print(issue_lock_store.resolve_locked_branch_for_session('$branch'))
|
||||
")
|
||||
if [[ -z "$locked_branch" ]]; then
|
||||
echo "Error: No session issue lock is bound. Call gitea_lock_issue before branch creation (fail closed)." >&2
|
||||
exit 2
|
||||
fi
|
||||
if [[ "$branch" != "$locked_branch" ]]; then
|
||||
echo "Error: Requested branch '$branch' does not match locked branch '$locked_branch' (fail closed)." >&2
|
||||
exit 2
|
||||
if [[ -z "$locked_branch" ]]; then
|
||||
echo "Error: No session issue lock is bound. Call gitea_lock_issue before branch creation (fail closed)." >&2
|
||||
exit 2
|
||||
fi
|
||||
if [[ "$branch" != "$locked_branch" ]]; then
|
||||
echo "Error: Requested branch '$branch' does not match locked branch '$locked_branch' (fail closed)." >&2
|
||||
exit 2
|
||||
fi
|
||||
fi
|
||||
|
||||
if [[ "$branch" =~ ^(fix|feat|docs|chore)/issue-[0-9]+-.+ ]] \
|
||||
|
||||
@@ -252,6 +252,31 @@ Helpers: `scripts/worktree-start`, `scripts/worktree-review`,
|
||||
- Never place raw tokens in LLM/MCP config.
|
||||
- Use `gitea_whoami` and `gitea_resolve_task_capability` before mutating.
|
||||
|
||||
## Fleet inventory
|
||||
|
||||
`gitea_whoami`, `gitea_get_runtime_context` and `gitea_assess_master_parity` each
|
||||
describe only the server answering the call. Five namespaces independently
|
||||
reporting the same revision never proved that five processes exist, that no sixth
|
||||
exists, or that all five belong to one client cohort.
|
||||
|
||||
`gitea_assess_fleet_inventory` is the read-only capability that does prove it. It
|
||||
takes no evidence parameters: it combines the control-plane runtime registry,
|
||||
which each server writes about itself at native transport bind, with a process
|
||||
observation the answering server performs. Classification is a pure function of
|
||||
that snapshot, so `gitea-controller` and `gitea-reconciler` return the same
|
||||
verdict for the same fleet.
|
||||
|
||||
Consume `mutation_gate_satisfied`. When it is false, report `blocked_reason`
|
||||
verbatim and stop — `missing_members`, `duplicate_members`, `unexpected_members`,
|
||||
`unregistered_processes`, `mixed_cohort` and `mixed_revision` are reported
|
||||
separately because each needs a different operator action. Treat
|
||||
`inventory_complete: false` and `single_cohort: null` as *unknown*, never as
|
||||
healthy. The capability never terminates a duplicate process, restarts,
|
||||
reconnects, or touches a lease.
|
||||
|
||||
Details, field meanings, and the gate-consumption sequence:
|
||||
[`docs/mcp-fleet-inventory.md`](../../docs/mcp-fleet-inventory.md) (#949).
|
||||
|
||||
## Tool inventory
|
||||
|
||||
[`docs/mcp-tool-inventory.md`](../../docs/mcp-tool-inventory.md) is the canonical
|
||||
|
||||
@@ -63,6 +63,11 @@ CONTAMINATION_GATED_TASKS = frozenset({
|
||||
"merge_pr",
|
||||
"delete_branch",
|
||||
"complete_issue",
|
||||
# Web console recovery playbooks that write (#644). These mutate runtime
|
||||
# binding and process state, so a live contamination marker must block them
|
||||
# exactly as it blocks the Gitea-side mutations above. The reconciler
|
||||
# cleanup playbook is the designated remedy and is exempted by its caller.
|
||||
"console_recovery_apply",
|
||||
})
|
||||
|
||||
CONTAMINATION_KIND = "stable_branch_push"
|
||||
|
||||
@@ -32,6 +32,35 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
},
|
||||
# #790 Slice A: prove an owned author lease is still active. Strictly
|
||||
# narrower than lock_issue — it can only slide a lease this exact session
|
||||
# already owns, never acquire, take over, or revive one — so it gates on the
|
||||
# same authority rather than introducing an operation name that every
|
||||
# already-configured author profile would be missing.
|
||||
"heartbeat_issue_lock": {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
},
|
||||
# #860: dirty orphaned same-claimant worktree recovery (explicit operation).
|
||||
"recover_dirty_orphaned_issue_worktree": {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
},
|
||||
"gitea_recover_dirty_orphaned_issue_worktree": {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
},
|
||||
# #864: dirty-preserving same-claimant author-session rebind (dead owner PID).
|
||||
# Author MCP tool path. Reconciler execute is gated inside the tool via
|
||||
# authorize_reconciler_execute + role_kind checks (not this map entry).
|
||||
"rebind_dirty_same_claimant_author_session": {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
},
|
||||
"gitea_rebind_dirty_same_claimant_author_session": {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
},
|
||||
"set_issue_labels": {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
@@ -58,6 +87,14 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.branch.create",
|
||||
"role": "author",
|
||||
},
|
||||
"bootstrap_author_issue_worktree": {
|
||||
"permission": "gitea.branch.create",
|
||||
"role": "author",
|
||||
},
|
||||
"gitea_bootstrap_author_issue_worktree": {
|
||||
"permission": "gitea.branch.create",
|
||||
"role": "author",
|
||||
},
|
||||
"push_branch": {
|
||||
"permission": "gitea.branch.push",
|
||||
"role": "author",
|
||||
@@ -95,6 +132,46 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.branch.push",
|
||||
"role": "author",
|
||||
},
|
||||
# #662: post-restart reconcile is read-only inventory + pure classification.
|
||||
# Durable follow-up issue creation is a separate apply path (not this task).
|
||||
"reconcile_after_restart": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
"gitea_reconcile_after_restart": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
# #644: Phase 2 Web Console recovery tasks.
|
||||
"clear_stale_binding": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
"rebind_session_worktree": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
# The console playbook orchestrates gitea_reconcile_merged_cleanups, whose
|
||||
# own gate is gitea.read (matching the existing reconcile_merged_cleanups
|
||||
# entry). Declaring a stricter permission here stated a second, conflicting
|
||||
# authority for one operation.
|
||||
"reconcile_cleanups": {
|
||||
"permission": "gitea.read",
|
||||
"role": "reconciler",
|
||||
},
|
||||
# #949: fleet inventory is strictly read-only evidence. It is deliberately
|
||||
# not role-exclusive — the controller and reconciler gates are its named
|
||||
# consumers, but any namespace holding gitea.read must be able to prove the
|
||||
# exact-five-process/single-cohort invariant before it acts, and the verdict
|
||||
# is a pure function of the snapshot so every namespace agrees.
|
||||
"assess_fleet_inventory": {
|
||||
"permission": "gitea.read",
|
||||
"role": "reconciler",
|
||||
},
|
||||
"gitea_assess_fleet_inventory": {
|
||||
"permission": "gitea.read",
|
||||
"role": "reconciler",
|
||||
},
|
||||
# PR synchronization lifecycle: assess is read-only (any role with gitea.read);
|
||||
# update-by-merge is author-only and mutates the PR head via Gitea API.
|
||||
"assess_pr_sync_status": {
|
||||
@@ -335,6 +412,21 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"role": "controller",
|
||||
},
|
||||
|
||||
# #642: sanctioned host-daemon lifecycle controls. Deliberately *not* a
|
||||
# ``gitea.*`` operation — restarting an MCP namespace is a host action, not
|
||||
# a Gitea API call, and no configured Gitea profile should be able to
|
||||
# satisfy it by accident. Authority comes from the console RBAC model plus
|
||||
# out-of-band operator authorization (#630); these entries exist so the
|
||||
# console cannot invent an authority the capability layer never declared.
|
||||
"restart_namespace": {
|
||||
"permission": "runtime.restart_namespace",
|
||||
"role": "controller",
|
||||
},
|
||||
"reload_namespace": {
|
||||
"permission": "runtime.reload_namespace",
|
||||
"role": "controller",
|
||||
},
|
||||
|
||||
# #601 first-class lease lifecycle — inspect/list need read; mutations gate on
|
||||
# ownership in the control-plane DB (not a separate Gitea write permission).
|
||||
"list_workflow_leases": {
|
||||
@@ -467,6 +559,15 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
},
|
||||
|
||||
# #651 console analytics ingest — control-plane DB write, not a Gitea API
|
||||
# call. Authority comes from console RBAC (operator+) plus phase gating;
|
||||
# permission string is a non-Gitea runtime capability so no Gitea profile
|
||||
# can satisfy it by accident.
|
||||
"record_analytics_usage": {
|
||||
"permission": "runtime.record_analytics_usage",
|
||||
"role": "author",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@@ -477,6 +578,15 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
# merger lease (#763).
|
||||
_PREFLIGHT_TASK_TRANSITIONS = frozenset({
|
||||
("review_pr", "acquire_reviewer_pr_lease"),
|
||||
# #850: native author issue worktree bootstrap
|
||||
("work_issue", "bootstrap_author_issue_worktree"),
|
||||
("bootstrap_author_issue_worktree", "lock_issue"),
|
||||
# #860: dirty-orphan recovery and related work_issue transitions (master)
|
||||
("work_issue", "lock_issue"),
|
||||
("work_issue", "recover_dirty_orphaned_issue_worktree"),
|
||||
("work_issue", "gitea_recover_dirty_orphaned_issue_worktree"),
|
||||
("work_issue", "commit_files"),
|
||||
("work_issue", "gitea_commit_files"),
|
||||
})
|
||||
|
||||
|
||||
@@ -523,6 +633,8 @@ ROLE_EXCLUSIVE_TASKS: frozenset[str] = frozenset(
|
||||
"gitea_release_merger_pr_lease",
|
||||
"create_branch",
|
||||
"push_branch",
|
||||
"bootstrap_author_issue_worktree",
|
||||
"gitea_bootstrap_author_issue_worktree",
|
||||
"publish_unpublished_branch",
|
||||
"create_pr",
|
||||
"commit_files",
|
||||
@@ -548,6 +660,7 @@ ISSUE_MUTATION_TOOL_TASKS: dict[str, str] = {
|
||||
"gitea_set_issue_labels": "set_issue_labels",
|
||||
"gitea_cleanup_terminal_pr_labels": "cleanup_terminal_pr_labels",
|
||||
"gitea_create_label": "create_label",
|
||||
"gitea_bootstrap_author_issue_worktree": "bootstrap_author_issue_worktree",
|
||||
"gitea_commit_files": "commit_files",
|
||||
}
|
||||
|
||||
|
||||
@@ -44,6 +44,8 @@ def _reset_mutation_authority(monkeypatch):
|
||||
]:
|
||||
monkeypatch.delenv(env_key, raising=False)
|
||||
|
||||
monkeypatch.setenv("GITEA_CLIENT_MANAGED", "1")
|
||||
|
||||
# Isolate durable session-state files so tests never share host cache (#559).
|
||||
import tempfile
|
||||
|
||||
|
||||
@@ -0,0 +1,243 @@
|
||||
"""Allocator epic / child-only container pre-rank exclusion (#844).
|
||||
|
||||
Covers:
|
||||
* Issue #631-shaped child-only epic is excluded before ranking.
|
||||
* Implementable child issues remain eligible and can be selected.
|
||||
* Ordinary issues that merely mention "epic" in title/body are not excluded.
|
||||
* Excluded containers never receive assignments or workflow leases.
|
||||
* Structured skip reason ``epic_or_child_only_container`` is reported.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
|
||||
from allocator_service import (
|
||||
OUTCOME_ASSIGNED,
|
||||
OUTCOME_PREVIEW,
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
WorkCandidate,
|
||||
allocate_next_work,
|
||||
classify_epic_or_child_only_container,
|
||||
)
|
||||
from control_plane_db import ControlPlaneDB
|
||||
|
||||
REMOTE = "prgs"
|
||||
ORG = "Scaled-Tech-Consulting"
|
||||
REPO = "Gitea-Tools"
|
||||
|
||||
# Minimal body mirroring issue #631 authoritative scope language.
|
||||
_EPIC_631_BODY = """
|
||||
## Scope (umbrella)
|
||||
|
||||
This epic owns the **product roadmap and linkage** for the Web Console.
|
||||
Implementation is delivered via child issues only.
|
||||
|
||||
## Explicit non-goals
|
||||
|
||||
* Do not implement product features in this epic issue itself.
|
||||
* No product feature implementation is claimed complete solely on this epic.
|
||||
"""
|
||||
|
||||
_CHILD_BODY = """
|
||||
## Problem
|
||||
|
||||
Operators need a workflow-event timeline model for Phase 1.
|
||||
|
||||
## Acceptance criteria
|
||||
|
||||
- [ ] Timeline model API exists
|
||||
"""
|
||||
|
||||
|
||||
def _issue(
|
||||
number: int,
|
||||
*,
|
||||
title: str = "",
|
||||
body: str = "",
|
||||
labels: tuple[str, ...] = ("status:ready", "type:feature"),
|
||||
priority: int = 20,
|
||||
) -> WorkCandidate:
|
||||
return WorkCandidate(
|
||||
kind="issue",
|
||||
number=number,
|
||||
state="open",
|
||||
labels=labels,
|
||||
title=title or f"issue {number}",
|
||||
body=body,
|
||||
priority=priority,
|
||||
)
|
||||
|
||||
|
||||
class ClassifyEpicContainerTest(unittest.TestCase):
|
||||
def test_631_shaped_body_and_title_is_container(self) -> None:
|
||||
c = _issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIsNotNone(detail)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_body_markers_without_epic_title(self) -> None:
|
||||
c = _issue(
|
||||
900,
|
||||
title="Control plane roadmap tracker",
|
||||
body="Implementation is delivered via child issues only.",
|
||||
)
|
||||
is_c, _ = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
|
||||
def test_epic_label_alone_is_container(self) -> None:
|
||||
c = _issue(
|
||||
901,
|
||||
title="Roadmap linkage",
|
||||
body="Track children.",
|
||||
labels=("status:ready", "type:epic"),
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("type:epic", detail or "")
|
||||
|
||||
def test_title_epic_prefix_alone_not_container(self) -> None:
|
||||
"""Title-only 'Epic:' without body scope evidence stays eligible (#844)."""
|
||||
c = _issue(
|
||||
902,
|
||||
title="Epic: something mentioned only in title",
|
||||
body="Implement a concrete fix for the allocator skip list.",
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
self.assertIsNone(detail)
|
||||
|
||||
def test_incidental_epic_word_not_container(self) -> None:
|
||||
c = _issue(
|
||||
903,
|
||||
title="Document epic handoff conventions",
|
||||
body=(
|
||||
"Update the docs so implementable issues that mention an epic "
|
||||
"remain independently executable."
|
||||
),
|
||||
)
|
||||
is_c, _ = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
|
||||
def test_prs_never_classified(self) -> None:
|
||||
pr = WorkCandidate(
|
||||
kind="pr",
|
||||
number=10,
|
||||
state="open",
|
||||
title="Epic: fake",
|
||||
body="Implementation is delivered via child issues only.",
|
||||
head_sha="a" * 40,
|
||||
priority=5,
|
||||
)
|
||||
is_c, _ = classify_epic_or_child_only_container(pr)
|
||||
self.assertFalse(is_c)
|
||||
|
||||
|
||||
class AllocateEpicContainerExclusionTest(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(os.path.join(self._tmp.name, "cp.sqlite3"))
|
||||
|
||||
def _alloc(self, candidates, **kwargs):
|
||||
defaults = dict(
|
||||
session_id="sess-844",
|
||||
role="author",
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
profile_name="prgs-author",
|
||||
username="jcwalker3",
|
||||
claims={},
|
||||
apply=False,
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
return allocate_next_work(self.db, candidates=candidates, **defaults)
|
||||
|
||||
def test_631_shaped_epic_excluded_child_selected(self) -> None:
|
||||
epic = _issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
)
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
res = self._alloc([epic, child], apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
skipped = {s["number"]: s for s in res["skipped"]}
|
||||
self.assertIn(631, skipped)
|
||||
self.assertEqual(
|
||||
skipped[631]["reason_code"], SKIP_EPIC_OR_CHILD_ONLY_CONTAINER
|
||||
)
|
||||
self.assertIn(SKIP_EPIC_OR_CHILD_ONLY_CONTAINER, skipped[631]["reason"])
|
||||
|
||||
def test_container_cannot_receive_assignment_or_lease(self) -> None:
|
||||
epic = _issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
)
|
||||
res = self._alloc([epic], apply=True)
|
||||
self.assertTrue(res["success"], res)
|
||||
# Only container present → no safe work; never assigned_work.
|
||||
self.assertNotEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertIsNone(res.get("assignment"))
|
||||
self.assertIsNone(res.get("selected"))
|
||||
skipped = {s["number"]: s for s in res["skipped"]}
|
||||
self.assertEqual(
|
||||
skipped[631]["reason_code"], SKIP_EPIC_OR_CHILD_ONLY_CONTAINER
|
||||
)
|
||||
# No lease row for the epic.
|
||||
leases = self.db.list_active_leases(
|
||||
remote=REMOTE, org=ORG, repo=REPO
|
||||
) if hasattr(self.db, "list_active_leases") else []
|
||||
# Prefer generic inventory if available.
|
||||
if not leases and hasattr(self.db, "list_leases"):
|
||||
leases = self.db.list_leases(remote=REMOTE, org=ORG, repo=REPO)
|
||||
for lease in leases or []:
|
||||
work_number = lease.get("work_number") if isinstance(lease, dict) else None
|
||||
self.assertNotEqual(work_number, 631)
|
||||
|
||||
def test_incidental_epic_title_remains_eligible(self) -> None:
|
||||
ordinary = _issue(
|
||||
700,
|
||||
title="Document epic handoff conventions",
|
||||
body="Write runbook text about epic vs child issues.",
|
||||
)
|
||||
res = self._alloc([ordinary], apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["selected"]["number"], 700)
|
||||
self.assertEqual(res["skipped"], [])
|
||||
|
||||
def test_apply_selects_child_not_epic(self) -> None:
|
||||
epic = _issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
)
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
res = self._alloc([epic, child], apply=True)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
self.assertEqual(res["assignment"]["work_number"], 637)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -7,6 +7,7 @@ import tempfile
|
||||
import threading
|
||||
import unittest
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from allocator_service import (
|
||||
OUTCOME_ASSIGNED,
|
||||
@@ -15,6 +16,7 @@ from allocator_service import (
|
||||
OUTCOME_PREVIEW,
|
||||
OUTCOME_WAIT,
|
||||
WorkCandidate,
|
||||
_drop_expired_claims,
|
||||
allocate_next_work,
|
||||
candidate_from_dict,
|
||||
classify_skip,
|
||||
@@ -362,5 +364,161 @@ class AllocatorServiceTest(unittest.TestCase):
|
||||
self.assertIn("unavailable", res["reasons"][0].lower())
|
||||
|
||||
|
||||
class SideEffectFreeAllocationTest(unittest.TestCase):
|
||||
"""``side_effect_free`` dry runs write nothing to the control plane (#643).
|
||||
|
||||
A plain ``apply=False`` still called ``upsert_session`` and
|
||||
``expire_stale_leases`` before the apply branch was consulted, so a caller
|
||||
advertising a read-only preview mutated on every call — one unreferenced
|
||||
session row per preview, plus a global lease sweep.
|
||||
"""
|
||||
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.db = ControlPlaneDB(os.path.join(self._tmp.name, "cp.sqlite3"))
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _alloc(self, **kwargs):
|
||||
defaults = dict(
|
||||
db=self.db,
|
||||
session_id="s-preview",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
candidates=[
|
||||
WorkCandidate(kind="issue", number=643, labels=("status:ready",))
|
||||
],
|
||||
apply=False,
|
||||
profile_name="prgs-author",
|
||||
username="jcwalker3",
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
return allocate_next_work(**defaults)
|
||||
|
||||
def _session_ids(self) -> set[str]:
|
||||
return {str(r.get("session_id")) for r in self.db.list_sessions()}
|
||||
|
||||
def test_side_effect_free_preview_writes_no_session_row(self):
|
||||
before = self._session_ids()
|
||||
result = self._alloc(side_effect_free=True)
|
||||
self.assertEqual(result["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(self._session_ids(), before)
|
||||
self.assertNotIn("s-preview", self._session_ids())
|
||||
|
||||
def test_plain_dry_run_still_registers_a_session(self):
|
||||
# The default is unchanged for every existing caller.
|
||||
self._alloc()
|
||||
self.assertIn("s-preview", self._session_ids())
|
||||
|
||||
def test_repeated_previews_do_not_accumulate_rows(self):
|
||||
for index in range(5):
|
||||
self._alloc(side_effect_free=True, session_id=f"s-{index}")
|
||||
self.assertEqual(self._session_ids(), set())
|
||||
|
||||
def test_side_effect_free_does_not_sweep_stale_leases(self):
|
||||
self.db.upsert_session(session_id="owner", role="author", pid=1)
|
||||
assigned = self.db.assign_and_lease(
|
||||
session_id="owner",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=999,
|
||||
lease_ttl_seconds=-60, # already expired
|
||||
)
|
||||
self.assertEqual(assigned.outcome, "assigned")
|
||||
|
||||
self._alloc(side_effect_free=True)
|
||||
|
||||
# The expired row is still 'active' in the DB: nothing swept it.
|
||||
statuses = {
|
||||
r["lease_id"]: r["status"]
|
||||
for r in self.db.list_leases(
|
||||
remote="prgs", org="org", repo="repo",
|
||||
statuses=("active", "expired"),
|
||||
)
|
||||
}
|
||||
self.assertEqual(statuses.get(assigned.lease_id), "active")
|
||||
|
||||
def test_expired_claims_are_filtered_in_memory_so_work_stays_selectable(self):
|
||||
"""The read-only mirror of the sweep: expired claims must not block."""
|
||||
self.db.upsert_session(session_id="owner", role="author", pid=1)
|
||||
self.db.assign_and_lease(
|
||||
session_id="owner",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=643,
|
||||
lease_ttl_seconds=-60, # expired: must not withhold #643
|
||||
)
|
||||
result = self._alloc(side_effect_free=True)
|
||||
self.assertEqual(result["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(result["selected"]["number"], 643)
|
||||
|
||||
def test_a_live_claim_still_withholds_the_work(self):
|
||||
self.db.upsert_session(session_id="owner", role="author", pid=1)
|
||||
self.db.assign_and_lease(
|
||||
session_id="owner",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=643,
|
||||
lease_ttl_seconds=3600,
|
||||
)
|
||||
result = self._alloc(side_effect_free=True)
|
||||
self.assertNotEqual(result["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertNotEqual((result.get("selected") or {}).get("number"), 643)
|
||||
|
||||
def test_side_effect_free_with_apply_fails_closed(self):
|
||||
result = self._alloc(side_effect_free=True, apply=True)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], OUTCOME_NO_SAFE)
|
||||
self.assertIsNone(result["assignment"])
|
||||
self.assertIn("incompatible with apply", result["reasons"][0])
|
||||
# And it reserved nothing.
|
||||
self.assertEqual(
|
||||
self.db.list_leases(remote="prgs", org="org", repo="repo"), []
|
||||
)
|
||||
|
||||
|
||||
class DropExpiredClaimsTest(unittest.TestCase):
|
||||
"""The in-memory expiry filter behind side-effect-free previews (#643)."""
|
||||
|
||||
def test_unparseable_expiry_is_kept_rather_than_assumed_free(self):
|
||||
claims = {
|
||||
("issue", 1): {"lease_id": "l1", "expires_at": "not-a-date"},
|
||||
("issue", 2): {"lease_id": "l2"},
|
||||
("issue", 3): {"lease_id": "l3", "expires_at": None},
|
||||
}
|
||||
self.assertEqual(_drop_expired_claims(claims), claims)
|
||||
|
||||
def test_expired_dropped_and_future_kept(self):
|
||||
now = datetime(2026, 7, 25, 12, 0, tzinfo=timezone.utc)
|
||||
claims = {
|
||||
("issue", 1): {"expires_at": "2026-07-25T11:59:59+00:00"},
|
||||
("issue", 2): {"expires_at": "2026-07-25T12:00:01+00:00"},
|
||||
("issue", 3): {"expires_at": "2026-07-25T12:00:00+00:00"}, # boundary
|
||||
}
|
||||
kept = _drop_expired_claims(claims, now=now)
|
||||
self.assertEqual(set(kept), {("issue", 2)})
|
||||
|
||||
def test_naive_and_zulu_timestamps_are_treated_as_utc(self):
|
||||
now = datetime(2026, 7, 25, 12, 0, tzinfo=timezone.utc)
|
||||
claims = {
|
||||
("issue", 1): {"expires_at": "2026-07-25T11:00:00"}, # naive, past
|
||||
("issue", 2): {"expires_at": "2026-07-25T13:00:00Z"}, # zulu, future
|
||||
}
|
||||
kept = _drop_expired_claims(claims, now=now)
|
||||
self.assertEqual(set(kept), {("issue", 2)})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -0,0 +1,601 @@
|
||||
"""Regression test suite for native author issue worktree bootstrap (#850)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest import mock
|
||||
|
||||
import author_issue_bootstrap
|
||||
import task_capability_map
|
||||
|
||||
|
||||
def _concurrent_bootstrap_worker(args: tuple[str, int, str, str, str, str]) -> dict:
|
||||
repo_dir, issue_num, key, lock_dir, journal_dir, master_sha = args
|
||||
os.environ["GITEA_BOOTSTRAP_JOURNAL_DIR"] = journal_dir
|
||||
return author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=issue_num,
|
||||
canonical_repo_root=repo_dir,
|
||||
expected_base_sha=master_sha,
|
||||
idempotency_key=key,
|
||||
lock_dir=lock_dir,
|
||||
owner_session="session-concurrent-test",
|
||||
active_identity="jcwalker3",
|
||||
active_profile="prgs-author",
|
||||
)
|
||||
|
||||
|
||||
class TestAuthorIssueBootstrap(unittest.TestCase):
|
||||
"""Test suite covering AC1-AC10 and comment #14959 specification."""
|
||||
|
||||
def setUp(self):
|
||||
self.tmp_dir = tempfile.mkdtemp(prefix="test_bootstrap_")
|
||||
self.repo_dir = os.path.join(self.tmp_dir, "repo")
|
||||
os.makedirs(self.repo_dir)
|
||||
|
||||
# Initialize synthetic git repo
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=self.repo_dir, check=True, capture_output=True)
|
||||
subprocess.run(["git", "config", "user.name", "Test User"], cwd=self.repo_dir, check=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=self.repo_dir, check=True)
|
||||
|
||||
readme = os.path.join(self.repo_dir, "README.md")
|
||||
with open(readme, "w", encoding="utf-8") as f:
|
||||
f.write("# Test Repo\n")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=self.repo_dir, check=True, capture_output=True)
|
||||
subprocess.run(["git", "commit", "-m", "initial commit"], cwd=self.repo_dir, check=True, capture_output=True)
|
||||
|
||||
rev_res = subprocess.run(["git", "rev-parse", "HEAD"], cwd=self.repo_dir, capture_output=True, text=True, check=True)
|
||||
self.master_sha = rev_res.stdout.strip()
|
||||
|
||||
self.branches_dir = os.path.join(self.repo_dir, "branches")
|
||||
os.makedirs(self.branches_dir, exist_ok=True)
|
||||
self.lock_dir = os.path.join(self.tmp_dir, "locks")
|
||||
os.makedirs(self.lock_dir, exist_ok=True)
|
||||
self.journal_dir = os.path.join(self.tmp_dir, "journals")
|
||||
os.makedirs(self.journal_dir, exist_ok=True)
|
||||
os.environ["GITEA_BOOTSTRAP_JOURNAL_DIR"] = self.journal_dir
|
||||
|
||||
def tearDown(self):
|
||||
os.environ.pop("GITEA_BOOTSTRAP_JOURNAL_DIR", None)
|
||||
shutil.rmtree(self.tmp_dir, ignore_errors=True)
|
||||
|
||||
def test_bootstrap_success_path(self):
|
||||
"""AC1/AC3/AC8: Successful bootstrap creates branch, worktree, registration, and lock proof."""
|
||||
key = "test_key_success_1"
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
# assignment_id/lease_id omitted: optional unless verified live.
|
||||
expected_base_sha=self.master_sha,
|
||||
idempotency_key=key,
|
||||
remote="prgs",
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertTrue(res.get("success"), f"Bootstrap failed: {res}")
|
||||
self.assertFalse(res.get("replayed"))
|
||||
self.assertEqual(res.get("issue_number"), 850)
|
||||
self.assertEqual(res.get("base_sha"), self.master_sha)
|
||||
self.assertIn("branches/fix-issue-850-native-mcp-bootstrap", res.get("worktree_path"))
|
||||
|
||||
# Verify worktree directory exists and is registered
|
||||
worktree_path = res["worktree_path"]
|
||||
self.assertTrue(os.path.isdir(worktree_path))
|
||||
|
||||
wt_list = subprocess.run(["git", "-C", self.repo_dir, "worktree", "list"], capture_output=True, text=True, check=True)
|
||||
self.assertIn(worktree_path, wt_list.stdout)
|
||||
|
||||
# Verify phase journal written
|
||||
journal = author_issue_bootstrap.load_phase_journal(key, journal_dir=self.lock_dir)
|
||||
self.assertIsNotNone(journal)
|
||||
self.assertTrue(journal.get("completed"))
|
||||
self.assertEqual(journal.get("current_phase"), author_issue_bootstrap.PHASE_7_TRANSITION_COMPLETED)
|
||||
|
||||
def test_idempotent_replay(self):
|
||||
"""Item 2: Replaying with identical key returns cached transition without duplicate creation."""
|
||||
key = "test_key_idempotent_1"
|
||||
res1 = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
idempotency_key=key,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertTrue(res1["success"], f"res1 failed: {res1}")
|
||||
self.assertFalse(res1.get("replayed"))
|
||||
|
||||
# Second call
|
||||
res2 = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
idempotency_key=key,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertTrue(res2["success"], f"res2 failed: {res2}")
|
||||
self.assertTrue(res2.get("replayed"))
|
||||
self.assertEqual(res1["worktree_path"], res2["worktree_path"])
|
||||
|
||||
def test_stale_concurrency_pin_refusal(self):
|
||||
"""Item 3: Mismatched expected base SHA fails closed without silent rebasing."""
|
||||
stale_sha = "0000000000000000000000000000000000000000"
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
expected_base_sha=stale_sha,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res.get("reason_code"), "stale_concurrency_pin")
|
||||
self.assertIn("exact_next_action", res)
|
||||
|
||||
def test_path_outside_branches_root_refusal(self):
|
||||
"""Item 6: Worktree path outside branches/ root is refused."""
|
||||
outside_path = os.path.join(self.tmp_dir, "outside_worktree")
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
worktree_path=outside_path,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res.get("reason_code"), "path_outside_canonical_branches_root")
|
||||
|
||||
def test_preexisting_dirty_worktree_preservation(self):
|
||||
"""Item 6: Preexisting dirty worktree fails closed and is NOT modified or cleaned."""
|
||||
branch = "fix/issue-850-dirty-test"
|
||||
wt_path = os.path.join(self.branches_dir, "fix-issue-850-dirty-test")
|
||||
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", "-b", branch, wt_path], check=True, capture_output=True)
|
||||
|
||||
# Create dirty untracked file
|
||||
dirty_file = os.path.join(wt_path, "dirty.txt")
|
||||
with open(dirty_file, "w") as f:
|
||||
f.write("dirty edits\n")
|
||||
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
branch_name=branch,
|
||||
worktree_path=wt_path,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res.get("reason_code"), "preexisting_dirty_worktree")
|
||||
|
||||
# Prove dirty file is preserved byte-for-byte
|
||||
self.assertTrue(os.path.exists(dirty_file))
|
||||
with open(dirty_file, "r") as f:
|
||||
self.assertEqual(f.read(), "dirty edits\n")
|
||||
|
||||
def test_compensating_recovery_on_failed_phase(self):
|
||||
"""AC4/Item 4: Failure during transition rolls back ONLY newly created artifacts."""
|
||||
key = "test_key_recovery_1"
|
||||
# Simulate partial progress in journal
|
||||
journal = {
|
||||
"idempotency_key": key,
|
||||
"issue_number": 850,
|
||||
"branch_name": "fix/issue-850-recovery-test",
|
||||
"worktree_path": os.path.join(self.branches_dir, "fix-issue-850-recovery-test"),
|
||||
"artifacts_created": {
|
||||
"branch_created": True,
|
||||
"worktree_dir_created": True,
|
||||
"worktree_registered": True,
|
||||
"lock_created": False,
|
||||
},
|
||||
"failure_reason": "simulated lock failure",
|
||||
"current_phase": author_issue_bootstrap.PHASE_5_REGISTRATION_VERIFIED,
|
||||
"completed": False,
|
||||
}
|
||||
# Create the branch and worktree manually to simulate partial state
|
||||
subprocess.run(["git", "-C", self.repo_dir, "branch", journal["branch_name"]], check=True, capture_output=True)
|
||||
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", journal["worktree_path"], journal["branch_name"]], check=True, capture_output=True)
|
||||
|
||||
# Run compensating recovery
|
||||
rec = author_issue_bootstrap.run_compensating_recovery(journal, self.repo_dir)
|
||||
self.assertTrue(rec["executed"])
|
||||
self.assertIn(f"worktree_path:{journal['worktree_path']}", rec["rolled_back"])
|
||||
self.assertIn(f"branch:{journal['branch_name']}", rec["rolled_back"])
|
||||
|
||||
# Prove worktree directory and branch were rolled back
|
||||
self.assertFalse(os.path.exists(journal["worktree_path"]))
|
||||
branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", journal["branch_name"]], capture_output=True, text=True, check=False)
|
||||
self.assertNotEqual(branch_check.returncode, 0)
|
||||
|
||||
def test_cross_process_concurrency(self):
|
||||
"""Review #525 Finding 1: Genuine cross-process concurrency locking prevents corruption."""
|
||||
import concurrent.futures
|
||||
|
||||
key = "test_concurrent_key_850"
|
||||
args = (self.repo_dir, 850, key, self.lock_dir, self.journal_dir, self.master_sha)
|
||||
|
||||
with concurrent.futures.ProcessPoolExecutor(max_workers=2) as executor:
|
||||
fut1 = executor.submit(_concurrent_bootstrap_worker, args)
|
||||
fut2 = executor.submit(_concurrent_bootstrap_worker, args)
|
||||
res1 = fut1.result(timeout=10)
|
||||
res2 = fut2.result(timeout=10)
|
||||
|
||||
self.assertTrue(res1["success"], f"res1 failed: {res1}")
|
||||
self.assertTrue(res2["success"], f"res2 failed: {res2}")
|
||||
# One process performs creation, the other process receives idempotent replay
|
||||
replayed_count = sum(1 for r in (res1, res2) if r.get("replayed"))
|
||||
created_count = sum(1 for r in (res1, res2) if not r.get("replayed"))
|
||||
self.assertEqual(replayed_count, 1)
|
||||
self.assertEqual(created_count, 1)
|
||||
self.assertEqual(res1["worktree_path"], res2["worktree_path"])
|
||||
|
||||
def test_interrupted_replay_preserves_artifacts_created_provenance(self):
|
||||
"""Review #525 Finding 2: Replaying incomplete journal preserves creation provenance monotonically."""
|
||||
key = "test_key_interrupted_replay_1"
|
||||
branch = "fix/issue-850-interrupted-replay"
|
||||
wt_path = os.path.join(self.branches_dir, "fix-issue-850-interrupted-replay")
|
||||
|
||||
# Simulate Phase 2/3 completion where branch and worktree directory were created by this transition
|
||||
journal = {
|
||||
"idempotency_key": key,
|
||||
"issue_number": 850,
|
||||
"branch_name": branch,
|
||||
"worktree_path": wt_path,
|
||||
"active_identity": "jcwalker3",
|
||||
"active_profile": "prgs-author",
|
||||
"remote": "prgs",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"phases": {
|
||||
author_issue_bootstrap.PHASE_1_REQUEST_ACCEPTED: {"status": "completed"},
|
||||
author_issue_bootstrap.PHASE_2_BRANCH_CONFIRMED: {"status": "completed", "created": True},
|
||||
},
|
||||
"artifacts_created": {
|
||||
"branch_created": True,
|
||||
"worktree_dir_created": True,
|
||||
"worktree_registered": True,
|
||||
"lock_created": False,
|
||||
},
|
||||
"current_phase": author_issue_bootstrap.PHASE_3_PATH_RESERVED,
|
||||
"completed": False,
|
||||
}
|
||||
# Pre-create the branch and worktree on disk to simulate partial state after crash
|
||||
subprocess.run(["git", "-C", self.repo_dir, "branch", branch, self.master_sha], check=True, capture_output=True)
|
||||
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", wt_path, branch], check=True, capture_output=True)
|
||||
author_issue_bootstrap.save_phase_journal(journal, journal_dir=self.lock_dir)
|
||||
|
||||
# Now resume/replay the transition but simulate lock binding failure during Phase 6
|
||||
with mock.patch("issue_lock_store.bind_session_lock", side_effect=RuntimeError("Lock failure test")):
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
branch_name=branch,
|
||||
worktree_path=wt_path,
|
||||
idempotency_key=key,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res.get("reason_code"), "issue_lock_acquisition_failed")
|
||||
|
||||
# Verify that compensating recovery correctly deleted transition-created branch & worktree
|
||||
# because creation provenance was preserved across replay (NOT downgraded to False!)
|
||||
self.assertFalse(os.path.exists(wt_path))
|
||||
branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", branch], capture_output=True, text=True, check=False)
|
||||
self.assertNotEqual(branch_check.returncode, 0)
|
||||
|
||||
def test_transition_created_only_compensation(self):
|
||||
"""Review #525 Finding 4: Preexisting branch is NOT deleted by compensation when only worktree was transition-created."""
|
||||
key = "test_key_preexisting_branch_compensation"
|
||||
preexisting_branch = "fix/issue-850-preexisting"
|
||||
wt_path = os.path.join(self.branches_dir, "fix-issue-850-preexisting")
|
||||
|
||||
# Create branch BEFORE bootstrap (preexisting branch)
|
||||
subprocess.run(["git", "-C", self.repo_dir, "branch", preexisting_branch, self.master_sha], check=True, capture_output=True)
|
||||
|
||||
# Call bootstrap with simulated failure during Phase 6 (lock binding)
|
||||
with mock.patch("issue_lock_store.bind_session_lock", side_effect=RuntimeError("Simulated lock failure")):
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
branch_name=preexisting_branch,
|
||||
worktree_path=wt_path,
|
||||
idempotency_key=key,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
|
||||
self.assertFalse(res["success"])
|
||||
# Worktree dir was created by transition -> removed by compensation
|
||||
self.assertFalse(os.path.exists(wt_path))
|
||||
|
||||
# Preexisting branch was NOT created by transition -> MUST BE PRESERVED!
|
||||
branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", preexisting_branch], capture_output=True, text=True, check=False)
|
||||
self.assertEqual(branch_check.returncode, 0, "Preexisting branch was deleted by mistake!")
|
||||
|
||||
def test_incompatible_idempotency_replay_refusal(self):
|
||||
"""Review #525 Finding 4: Replaying key with incompatible parameters returns refusal."""
|
||||
key = "test_key_incompatible_replay"
|
||||
res1 = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
branch_name="fix/issue-850-param-a",
|
||||
idempotency_key=key,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertTrue(res1["success"])
|
||||
|
||||
# Second call with different branch_name
|
||||
res2 = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
branch_name="fix/issue-850-param-b",
|
||||
idempotency_key=key,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertFalse(res2["success"])
|
||||
self.assertEqual(res2.get("reason_code"), "incompatible_idempotency_replay")
|
||||
|
||||
def test_exact_next_action_satisfiable_via_mcp(self):
|
||||
"""Review #525 Finding 4: exact_next_action provides satisfiable MCP actions, not shell commands."""
|
||||
key = "test_key_next_action_mcp"
|
||||
stale_sha = "0000000000000000000000000000000000000000"
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
expected_base_sha=stale_sha,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
next_action = res.get("exact_next_action", "")
|
||||
self.assertNotIn("scripts/worktree-start", next_action)
|
||||
self.assertNotIn("git worktree add", next_action)
|
||||
self.assertNotIn("bash", next_action.lower())
|
||||
|
||||
def test_missing_owner_session_refusal(self):
|
||||
"""Finding D: Missing owner_session context fails closed with typed refusal and zero mutation."""
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
owner_session=None,
|
||||
lock_dir=self.lock_dir,
|
||||
)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res.get("reason_code"), "missing_owner_session")
|
||||
self.assertIn("exact_next_action", res)
|
||||
|
||||
def test_symlink_lock_file_refusal(self):
|
||||
"""Finding C: BootstrapTransitionLock refuses to follow symlinks."""
|
||||
key = "test_symlink_lock_key"
|
||||
safe_key = "".join(c if c.isalnum() or c in ("-", "_", ".") else "_" for c in key)
|
||||
lock_path = os.path.join(self.lock_dir, f"{safe_key}.lock")
|
||||
target_file = os.path.join(self.tmp_dir, "fake_target")
|
||||
with open(target_file, "w") as f:
|
||||
f.write("target")
|
||||
os.symlink(target_file, lock_path)
|
||||
|
||||
with self.assertRaises(RuntimeError) as ctx:
|
||||
with author_issue_bootstrap.BootstrapTransitionLock(key, journal_dir=self.lock_dir):
|
||||
pass
|
||||
self.assertIn("symlink", str(ctx.exception).lower())
|
||||
|
||||
def test_lock_directory_escape_refusal(self):
|
||||
"""Finding C: BootstrapTransitionLock refuses keys that escape lock directory."""
|
||||
with mock.patch("os.path.abspath", return_value="/tmp/outside/evil_key.lock"):
|
||||
with self.assertRaises(RuntimeError) as ctx:
|
||||
author_issue_bootstrap.BootstrapTransitionLock("key", journal_dir=self.lock_dir)
|
||||
self.assertIn("escapes", str(ctx.exception).lower())
|
||||
|
||||
def test_missing_active_identity_refusal(self):
|
||||
"""F-5: Missing active_identity parameter fails closed."""
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
owner_session="session-test-1234",
|
||||
active_identity=None,
|
||||
active_profile="prgs-author",
|
||||
lock_dir=self.lock_dir,
|
||||
)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res.get("reason_code"), "missing_active_identity")
|
||||
|
||||
def test_missing_active_profile_refusal(self):
|
||||
"""F-5: Missing active_profile parameter fails closed."""
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
owner_session="session-test-1234",
|
||||
active_identity="jcwalker3",
|
||||
active_profile=None,
|
||||
lock_dir=self.lock_dir,
|
||||
)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res.get("reason_code"), "missing_active_profile")
|
||||
|
||||
def test_dirty_worktree_preserved_during_recovery(self):
|
||||
"""F-4: Compensating recovery does not delete dirty worktree."""
|
||||
branch = "fix/issue-850-rec-dirty"
|
||||
wt_path = os.path.join(self.branches_dir, "fix-issue-850-rec-dirty")
|
||||
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", "-b", branch, wt_path], check=True, capture_output=True)
|
||||
dirty_file = os.path.join(wt_path, "dirty.txt")
|
||||
with open(dirty_file, "w") as f:
|
||||
f.write("uncommitted work")
|
||||
|
||||
journal = {
|
||||
"idempotency_key": "test_dirty_rec",
|
||||
"issue_number": 850,
|
||||
"branch_name": branch,
|
||||
"worktree_path": wt_path,
|
||||
"artifacts_created": {
|
||||
"worktree_dir_created": True,
|
||||
"worktree_registered": True,
|
||||
},
|
||||
"failure_reason": "test dirty recovery",
|
||||
}
|
||||
rec = author_issue_bootstrap.run_compensating_recovery(journal, self.repo_dir, journal_dir=self.lock_dir)
|
||||
self.assertTrue(os.path.exists(wt_path))
|
||||
self.assertIn(f"worktree_path_preserved_dirty:{wt_path}", rec["rolled_back"])
|
||||
|
||||
def test_branch_with_commits_preserved_during_recovery(self):
|
||||
"""F-4: Compensating recovery does not delete branch with author commits."""
|
||||
branch = "fix/issue-850-rec-commits"
|
||||
subprocess.run(["git", "-C", self.repo_dir, "branch", branch, self.master_sha], check=True, capture_output=True)
|
||||
# Add a commit on the branch
|
||||
wt_path = os.path.join(self.branches_dir, "fix-issue-850-rec-commits")
|
||||
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", wt_path, branch], check=True, capture_output=True)
|
||||
cfile = os.path.join(wt_path, "commit.txt")
|
||||
with open(cfile, "w") as f:
|
||||
f.write("author commit")
|
||||
subprocess.run(["git", "-C", wt_path, "add", "commit.txt"], check=True, capture_output=True)
|
||||
subprocess.run(["git", "-C", wt_path, "commit", "-m", "author commit"], check=True, capture_output=True)
|
||||
subprocess.run(["git", "-C", self.repo_dir, "worktree", "remove", "--force", wt_path], check=True, capture_output=True)
|
||||
|
||||
journal = {
|
||||
"idempotency_key": "test_commits_rec",
|
||||
"issue_number": 850,
|
||||
"branch_name": branch,
|
||||
"resolved_base_sha": self.master_sha,
|
||||
"artifacts_created": {
|
||||
"branch_created": True,
|
||||
},
|
||||
"failure_reason": "test commit branch recovery",
|
||||
}
|
||||
rec = author_issue_bootstrap.run_compensating_recovery(journal, self.repo_dir, journal_dir=self.lock_dir)
|
||||
branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", branch], capture_output=True, text=True, check=False)
|
||||
self.assertEqual(branch_check.returncode, 0, "Branch with commits was deleted!")
|
||||
self.assertIn(f"branch_preserved_commits:{branch}", rec["rolled_back"])
|
||||
|
||||
def test_task_capability_map_integration(self):
|
||||
"""Verify task_capability_map has bootstrap_author_issue_worktree configured correctly."""
|
||||
self.assertEqual(task_capability_map.required_role("bootstrap_author_issue_worktree"), "author")
|
||||
self.assertEqual(task_capability_map.required_permission("bootstrap_author_issue_worktree"), "gitea.branch.create")
|
||||
self.assertTrue(task_capability_map.preflight_task_matches("work_issue", "bootstrap_author_issue_worktree"))
|
||||
self.assertTrue(task_capability_map.preflight_task_matches("bootstrap_author_issue_worktree", "lock_issue"))
|
||||
|
||||
def test_unverified_assignment_lease_ids_fail_closed(self):
|
||||
"""Review #531 Finding 4: fabricated assignment/lease IDs are refused."""
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
assignment_id="asn-fabricated",
|
||||
lease_id="lease-fabricated",
|
||||
expected_base_sha=self.master_sha,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertIn(
|
||||
res.get("reason_code"),
|
||||
{
|
||||
"unknown_lease_id",
|
||||
"assignment_lease_lookup_failed",
|
||||
"incomplete_assignment_lease_ids",
|
||||
},
|
||||
)
|
||||
|
||||
def test_partial_assignment_lease_ids_fail_closed(self):
|
||||
"""Review #531 Finding 4: one of assignment_id/lease_id alone is incomplete."""
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
assignment_id="asn-only",
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res.get("reason_code"), "incomplete_assignment_lease_ids")
|
||||
|
||||
def test_stale_diverged_branch_is_not_accepted_via_merge_base(self):
|
||||
"""Review #531 Finding 3: any common ancestor is not enough; require master ⊆ branch."""
|
||||
branch = "fix/issue-850-stale-divergent"
|
||||
# Create branch from current master, then advance master so branch lacks tip.
|
||||
subprocess.run(
|
||||
["git", "-C", self.repo_dir, "branch", branch, self.master_sha],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
)
|
||||
# Make a new commit on master (orphan path so branch does not contain it).
|
||||
marker = os.path.join(self.repo_dir, "master-advance.txt")
|
||||
with open(marker, "w") as f:
|
||||
f.write("advance master\n")
|
||||
subprocess.run(["git", "-C", self.repo_dir, "add", "master-advance.txt"], check=True, capture_output=True)
|
||||
subprocess.run(
|
||||
["git", "-C", self.repo_dir, "commit", "-m", "advance master past branch"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
)
|
||||
new_master = subprocess.run(
|
||||
["git", "-C", self.repo_dir, "rev-parse", "HEAD"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
).stdout.strip()
|
||||
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
|
||||
issue_number=850,
|
||||
canonical_repo_root=self.repo_dir,
|
||||
branch_name=branch,
|
||||
expected_base_sha=new_master,
|
||||
lock_dir=self.lock_dir,
|
||||
owner_session="session-test-1234",
|
||||
)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res.get("reason_code"), "incompatible_existing_branch")
|
||||
|
||||
def test_compensating_recovery_attempts_lease_release(self):
|
||||
"""Review #531 Finding 5: recovery invokes lease release when lease_id is present."""
|
||||
from unittest import mock
|
||||
|
||||
journal = {
|
||||
"idempotency_key": "test_lease_rec",
|
||||
"issue_number": 850,
|
||||
"owner_session": "session-test-1234",
|
||||
"lease_id": "lease-abc",
|
||||
"branch_name": "fix/issue-850-lease-rec",
|
||||
"artifacts_created": {"lock_created": True},
|
||||
"failure_reason": "simulated",
|
||||
"completed": False,
|
||||
}
|
||||
with mock.patch.object(
|
||||
author_issue_bootstrap.lease_lifecycle,
|
||||
"release_lease",
|
||||
return_value={"success": True},
|
||||
) as rel, mock.patch.object(
|
||||
author_issue_bootstrap.control_plane_db,
|
||||
"ControlPlaneDB",
|
||||
return_value=mock.Mock(),
|
||||
):
|
||||
rec = author_issue_bootstrap.run_compensating_recovery(
|
||||
journal, self.repo_dir, journal_dir=self.lock_dir
|
||||
)
|
||||
self.assertTrue(rec["executed"])
|
||||
rel.assert_called_once()
|
||||
self.assertIn("lease:lease-abc", rec["rolled_back"])
|
||||
|
||||
|
||||
class TestCanonicalRootNoStringSplit(unittest.TestCase):
|
||||
def test_fallback_uses_commonpath_not_substring_split(self):
|
||||
"""Review #531 Finding 2: no norm.split('/branches/') fallback."""
|
||||
import inspect
|
||||
import author_mutation_worktree as amw
|
||||
|
||||
src = inspect.getsource(amw.resolve_canonical_repo_root)
|
||||
self.assertNotIn('split("/branches/")', src)
|
||||
self.assertNotIn("split('/branches/')", src)
|
||||
|
||||
# Fallback recovers repo root from a nested branches worktree path.
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
repo = os.path.join(tmp, "repo")
|
||||
wt = os.path.join(repo, "branches", "fix-issue-850-x")
|
||||
os.makedirs(wt)
|
||||
# git unavailable path: pass missing workspace so fallback is used.
|
||||
root = amw.resolve_canonical_repo_root("/missing/path", wt)
|
||||
self.assertEqual(root, os.path.realpath(repo))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -28,6 +28,21 @@ class TestPathUnderBranches(unittest.TestCase):
|
||||
amw.is_path_under_branches("/repo/other-checkout", self.ROOT)
|
||||
)
|
||||
|
||||
def test_unrelated_branches_dir_fails(self):
|
||||
self.assertFalse(
|
||||
amw.is_path_under_branches("/tmp/branches/evil", self.ROOT)
|
||||
)
|
||||
|
||||
def test_prefix_confusion_fails(self):
|
||||
self.assertFalse(
|
||||
amw.is_path_under_branches(f"{self.ROOT}/branches-other/foo", self.ROOT)
|
||||
)
|
||||
|
||||
def test_traversal_fails(self):
|
||||
self.assertFalse(
|
||||
amw.is_path_under_branches(f"{self.ROOT}/branches/../evil", self.ROOT)
|
||||
)
|
||||
|
||||
|
||||
class TestAssessAuthorMutationWorktree(unittest.TestCase):
|
||||
ROOT = "/repo/Gitea-Tools"
|
||||
@@ -71,6 +86,13 @@ class TestAssessAuthorMutationWorktree(unittest.TestCase):
|
||||
self.assertTrue(result["proven"])
|
||||
self.assertFalse(result["block"])
|
||||
|
||||
def test_path_shaped_branches_ancestry_without_isdir(self):
|
||||
"""Review #551: commonpath recovery must not require on-disk isdir."""
|
||||
fake_wt = "/repo/Gitea-Tools/branches/issue-274"
|
||||
root = amw.resolve_canonical_repo_root(fake_wt, fake_wt)
|
||||
self.assertEqual(root, "/repo/Gitea-Tools")
|
||||
self.assertNotIn('split("/branches/")', open(amw.__file__).read())
|
||||
|
||||
|
||||
class TestPreflightIntegration(unittest.TestCase):
|
||||
def test_verify_preflight_blocks_control_checkout_with_test_porcelain(self):
|
||||
|
||||
@@ -1266,6 +1266,730 @@ class TestSecondRemediationIntegration(unittest.TestCase):
|
||||
self.assertIn("delete_acknowledged", delete_actions[0])
|
||||
self.assertTrue(delete_actions[0].get("verified_absent"))
|
||||
|
||||
def test_issue_851_worktree_removed_when_remote_blocked_only_by_worktree_binding(self):
|
||||
"""#851: remote blocked by worktree_binding must not skip safe worktree removal.
|
||||
|
||||
Lifecycle: remove clean owned worktree → reassess ownership → delete
|
||||
remote only if independently safe. Unrelated entries stay untouched.
|
||||
"""
|
||||
from mcp_server import gitea_reconcile_merged_cleanups
|
||||
|
||||
target_branch = "fix/issue-844-exclude-epic-containers"
|
||||
foreign_branch = "fix/issue-999-unrelated-active"
|
||||
worktree_path = "/tmp/branches/fix-issue-844-exclude-epic-containers"
|
||||
ownership_calls = []
|
||||
remove_calls = []
|
||||
delete_api_calls = []
|
||||
|
||||
def fake_collect(**kwargs):
|
||||
ownership_calls.append(dict(kwargs))
|
||||
# Ownership is reassessed *after* independent worktree removal (#851).
|
||||
# Target worktree is already gone → no worktree_binding remains.
|
||||
# Foreign branch keeps an active author lease → remote delete blocked.
|
||||
if kwargs.get("branch") == foreign_branch:
|
||||
# Match session-bound org/repo + host used by the tool resolve path.
|
||||
return {
|
||||
"records": [
|
||||
{
|
||||
"category": guard.OWNERSHIP_CATEGORY_AUTHOR_LEASE,
|
||||
"status": "active",
|
||||
"remote": kwargs.get("remote") or "prgs",
|
||||
"host": kwargs.get("host") or "gitea.example.com",
|
||||
"org": kwargs.get("org") or "Scaled-Tech-Consulting",
|
||||
"repo": kwargs.get("repo") or "Gitea-Tools",
|
||||
"branch": foreign_branch,
|
||||
"reclaim_allowed": False,
|
||||
}
|
||||
],
|
||||
"inventory_error": False,
|
||||
}
|
||||
return {"records": [], "inventory_error": False}
|
||||
|
||||
def fake_remove(project_root, branch, worktree_path=None):
|
||||
remove_calls.append(
|
||||
{"branch": branch, "worktree_path": worktree_path}
|
||||
)
|
||||
return {
|
||||
"success": True,
|
||||
"performed": True,
|
||||
"message": f"removed worktree {worktree_path}",
|
||||
"worktree_path": worktree_path,
|
||||
}
|
||||
|
||||
def fake_probe(h, o, r, auth, br):
|
||||
return guard.classify_branch_readback_http_status(
|
||||
404, not_found_scope=guard.NOT_FOUND_SCOPE_BRANCH
|
||||
)
|
||||
|
||||
def fake_api(method, url, auth, **kwargs):
|
||||
if method == "DELETE":
|
||||
delete_api_calls.append(url)
|
||||
return {}
|
||||
|
||||
report = {
|
||||
"entries": [
|
||||
{
|
||||
"pr_number": 848,
|
||||
"head_branch": target_branch,
|
||||
"remote_branch": {"safe_to_delete_remote": True},
|
||||
"local_worktree": {
|
||||
"safe_to_remove_worktree": True,
|
||||
"worktree_path": worktree_path,
|
||||
},
|
||||
},
|
||||
{
|
||||
"pr_number": 999,
|
||||
"head_branch": foreign_branch,
|
||||
"remote_branch": {"safe_to_delete_remote": True},
|
||||
"local_worktree": {
|
||||
"safe_to_remove_worktree": False,
|
||||
"worktree_path": None,
|
||||
},
|
||||
},
|
||||
],
|
||||
"reviewer_scratch_entries": [],
|
||||
}
|
||||
patch(
|
||||
"mcp_server.get_profile",
|
||||
return_value={
|
||||
"profile_name": "prgs-reconciler",
|
||||
"role": "reconciler",
|
||||
"allowed_operations": [
|
||||
"gitea.read",
|
||||
"gitea.branch.delete",
|
||||
"gitea.pr.close",
|
||||
],
|
||||
"forbidden_operations": [],
|
||||
},
|
||||
).start()
|
||||
patch("mcp_server.api_get_all", return_value=[]).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.build_reconciliation_report",
|
||||
return_value=report,
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.discover_reviewer_scratch_worktrees",
|
||||
return_value=[],
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.audit_reconciliation_mode.check_cleanup_execution_allowed",
|
||||
return_value=(True, []),
|
||||
).start()
|
||||
patch("mcp_server.verify_preflight_purity", return_value=None).start()
|
||||
patch(
|
||||
"mcp_server._collect_branch_ownership_records",
|
||||
side_effect=fake_collect,
|
||||
).start()
|
||||
patch("mcp_server._probe_remote_branch", side_effect=fake_probe).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.remove_local_worktree",
|
||||
side_effect=fake_remove,
|
||||
).start()
|
||||
self.mock_api.side_effect = fake_api
|
||||
|
||||
res = gitea_reconcile_merged_cleanups(
|
||||
dry_run=False,
|
||||
execute_confirmed=True,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertTrue(res.get("performed") or res.get("executed"))
|
||||
actions = res.get("actions") or []
|
||||
|
||||
remove_actions = [
|
||||
a for a in actions if a.get("action") == "remove_local_worktree"
|
||||
]
|
||||
self.assertEqual(len(remove_actions), 1, actions)
|
||||
self.assertTrue(remove_actions[0].get("success"))
|
||||
self.assertEqual(remove_calls[0]["branch"], target_branch)
|
||||
self.assertEqual(remove_calls[0]["worktree_path"], worktree_path)
|
||||
|
||||
# Target remote delete succeeds after worktree removal + reassessment.
|
||||
target_deletes = [
|
||||
a
|
||||
for a in actions
|
||||
if a.get("action") == "delete_remote_branch"
|
||||
and a.get("branch") == target_branch
|
||||
]
|
||||
self.assertEqual(len(target_deletes), 1, actions)
|
||||
self.assertTrue(target_deletes[0].get("success"))
|
||||
self.assertTrue(target_deletes[0].get("after_worktree_removal"))
|
||||
self.assertTrue(target_deletes[0].get("ownership_reassessed"))
|
||||
self.assertTrue(target_deletes[0].get("verified_absent"))
|
||||
|
||||
# Foreign branch remains protected (author lease) and is not deleted.
|
||||
foreign_deletes = [
|
||||
a
|
||||
for a in actions
|
||||
if a.get("action") == "delete_remote_branch"
|
||||
and a.get("branch") == foreign_branch
|
||||
]
|
||||
self.assertEqual(len(foreign_deletes), 1, actions)
|
||||
self.assertFalse(foreign_deletes[0].get("success"))
|
||||
self.assertEqual(
|
||||
foreign_deletes[0].get("blocker_kind"), "active_branch_ownership"
|
||||
)
|
||||
self.assertIn(
|
||||
guard.OWNERSHIP_CATEGORY_AUTHOR_LEASE,
|
||||
foreign_deletes[0].get("blocking_categories") or [],
|
||||
)
|
||||
# Only the target branch should hit the DELETE API.
|
||||
self.assertEqual(len(delete_api_calls), 1)
|
||||
|
||||
# Ownership collected for target (post-removal) and foreign; worktree
|
||||
# removal happened before target remote delete in the action log.
|
||||
target_idx = next(
|
||||
i
|
||||
for i, a in enumerate(actions)
|
||||
if a.get("action") == "remove_local_worktree"
|
||||
)
|
||||
delete_idx = next(
|
||||
i
|
||||
for i, a in enumerate(actions)
|
||||
if a.get("action") == "delete_remote_branch"
|
||||
and a.get("branch") == target_branch
|
||||
and a.get("success")
|
||||
)
|
||||
self.assertLess(target_idx, delete_idx)
|
||||
|
||||
def test_issue_851_dirty_worktree_not_removed_and_remote_stays_protected(self):
|
||||
"""#851: dirty/foreign worktrees remain protected; no unsafe cleanup."""
|
||||
from mcp_server import gitea_reconcile_merged_cleanups
|
||||
|
||||
branch = "fix/issue-851-dirty"
|
||||
remove_calls = []
|
||||
|
||||
def fake_collect(**kwargs):
|
||||
return {
|
||||
"records": [
|
||||
{
|
||||
"category": guard.OWNERSHIP_CATEGORY_WORKTREE_BINDING,
|
||||
"status": "active",
|
||||
"remote": kwargs.get("remote") or "prgs",
|
||||
"host": kwargs.get("host") or "gitea.example.com",
|
||||
"org": kwargs.get("org") or "Scaled-Tech-Consulting",
|
||||
"repo": kwargs.get("repo") or "Gitea-Tools",
|
||||
"branch": branch,
|
||||
"reclaim_allowed": False,
|
||||
}
|
||||
],
|
||||
"inventory_error": False,
|
||||
}
|
||||
|
||||
report = {
|
||||
"entries": [
|
||||
{
|
||||
"pr_number": 851,
|
||||
"head_branch": branch,
|
||||
"remote_branch": {"safe_to_delete_remote": True},
|
||||
"local_worktree": {
|
||||
"safe_to_remove_worktree": False,
|
||||
"worktree_path": "/tmp/dirty-wt",
|
||||
},
|
||||
}
|
||||
],
|
||||
"reviewer_scratch_entries": [],
|
||||
}
|
||||
patch(
|
||||
"mcp_server.get_profile",
|
||||
return_value={
|
||||
"profile_name": "prgs-reconciler",
|
||||
"role": "reconciler",
|
||||
"allowed_operations": [
|
||||
"gitea.read",
|
||||
"gitea.branch.delete",
|
||||
],
|
||||
"forbidden_operations": [],
|
||||
},
|
||||
).start()
|
||||
patch("mcp_server.api_get_all", return_value=[]).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.build_reconciliation_report",
|
||||
return_value=report,
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.discover_reviewer_scratch_worktrees",
|
||||
return_value=[],
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.audit_reconciliation_mode.check_cleanup_execution_allowed",
|
||||
return_value=(True, []),
|
||||
).start()
|
||||
patch("mcp_server.verify_preflight_purity", return_value=None).start()
|
||||
patch(
|
||||
"mcp_server._collect_branch_ownership_records",
|
||||
side_effect=fake_collect,
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.remove_local_worktree",
|
||||
side_effect=lambda *a, **k: remove_calls.append(k) or {
|
||||
"success": True,
|
||||
"performed": True,
|
||||
},
|
||||
).start()
|
||||
self.mock_api.side_effect = lambda *a, **k: {}
|
||||
|
||||
res = gitea_reconcile_merged_cleanups(
|
||||
dry_run=False,
|
||||
execute_confirmed=True,
|
||||
remote="prgs",
|
||||
)
|
||||
actions = res.get("actions") or []
|
||||
self.assertEqual(remove_calls, [])
|
||||
self.assertFalse(
|
||||
any(a.get("action") == "remove_local_worktree" for a in actions)
|
||||
)
|
||||
deletes = [
|
||||
a for a in actions if a.get("action") == "delete_remote_branch"
|
||||
]
|
||||
self.assertEqual(len(deletes), 1)
|
||||
self.assertFalse(deletes[0].get("success"))
|
||||
self.assertEqual(deletes[0].get("blocker_kind"), "active_branch_ownership")
|
||||
self.assertIn(
|
||||
guard.OWNERSHIP_CATEGORY_WORKTREE_BINDING,
|
||||
deletes[0].get("blocking_categories") or [],
|
||||
)
|
||||
|
||||
def test_issue_851_idempotent_resume_when_worktree_already_absent(self):
|
||||
"""#851: partial failures remain resumable and idempotent."""
|
||||
from mcp_server import gitea_reconcile_merged_cleanups
|
||||
|
||||
branch = "fix/issue-851-resume"
|
||||
ownership_calls = []
|
||||
|
||||
def fake_collect(**kwargs):
|
||||
ownership_calls.append(kwargs)
|
||||
return {"records": [], "inventory_error": False}
|
||||
|
||||
def fake_remove(project_root, branch, worktree_path=None):
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"message": f"worktree not found: {worktree_path}",
|
||||
}
|
||||
|
||||
def fake_probe(h, o, r, auth, br):
|
||||
return guard.classify_branch_readback_http_status(
|
||||
404, not_found_scope=guard.NOT_FOUND_SCOPE_BRANCH
|
||||
)
|
||||
|
||||
report = {
|
||||
"entries": [
|
||||
{
|
||||
"pr_number": 851,
|
||||
"head_branch": branch,
|
||||
"remote_branch": {"safe_to_delete_remote": True},
|
||||
"local_worktree": {
|
||||
"safe_to_remove_worktree": True,
|
||||
"worktree_path": "/tmp/already-gone",
|
||||
},
|
||||
}
|
||||
],
|
||||
"reviewer_scratch_entries": [],
|
||||
}
|
||||
patch(
|
||||
"mcp_server.get_profile",
|
||||
return_value={
|
||||
"profile_name": "prgs-reconciler",
|
||||
"role": "reconciler",
|
||||
"allowed_operations": [
|
||||
"gitea.read",
|
||||
"gitea.branch.delete",
|
||||
],
|
||||
"forbidden_operations": [],
|
||||
},
|
||||
).start()
|
||||
patch("mcp_server.api_get_all", return_value=[]).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.build_reconciliation_report",
|
||||
return_value=report,
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.discover_reviewer_scratch_worktrees",
|
||||
return_value=[],
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.audit_reconciliation_mode.check_cleanup_execution_allowed",
|
||||
return_value=(True, []),
|
||||
).start()
|
||||
patch("mcp_server.verify_preflight_purity", return_value=None).start()
|
||||
patch(
|
||||
"mcp_server._collect_branch_ownership_records",
|
||||
side_effect=fake_collect,
|
||||
).start()
|
||||
patch("mcp_server._probe_remote_branch", side_effect=fake_probe).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.remove_local_worktree",
|
||||
side_effect=fake_remove,
|
||||
).start()
|
||||
self.mock_api.side_effect = lambda *a, **k: {}
|
||||
|
||||
res = gitea_reconcile_merged_cleanups(
|
||||
dry_run=False,
|
||||
execute_confirmed=True,
|
||||
remote="prgs",
|
||||
)
|
||||
actions = res.get("actions") or []
|
||||
removes = [a for a in actions if a.get("action") == "remove_local_worktree"]
|
||||
deletes = [a for a in actions if a.get("action") == "delete_remote_branch"]
|
||||
self.assertEqual(len(removes), 1)
|
||||
self.assertFalse(removes[0].get("success"))
|
||||
self.assertEqual(len(deletes), 1)
|
||||
self.assertTrue(deletes[0].get("success"))
|
||||
self.assertTrue(deletes[0].get("after_worktree_removal"))
|
||||
self.assertTrue(ownership_calls)
|
||||
|
||||
|
||||
class TestIssue855ExactPrSelector(unittest.TestCase):
|
||||
"""#855: exact pr_number pin for reconcile_merged_cleanups (#851 lifecycle)."""
|
||||
|
||||
def setUp(self):
|
||||
self._remotes = patch.dict(
|
||||
mcp_server.REMOTES,
|
||||
{
|
||||
"prgs": {
|
||||
"host": "gitea.example.com",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
}
|
||||
},
|
||||
)
|
||||
self._remotes.start()
|
||||
patch("gitea_audit.audit_enabled", return_value=False).start()
|
||||
self.mock_api = patch("mcp_server.api_request").start()
|
||||
self.mock_all = patch("mcp_server.api_get_all", return_value=[]).start()
|
||||
patch("mcp_server.get_auth_header", return_value=FAKE_AUTH).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.is_head_ancestor_of_ref",
|
||||
return_value=True,
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.get_profile",
|
||||
return_value=dict(RECONCILER_WITH_DELETE),
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server._profile_operation_gate",
|
||||
return_value=[],
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server._collect_branch_ownership_records",
|
||||
return_value={"records": [], "inventory_error": False},
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.discover_reviewer_scratch_worktrees",
|
||||
return_value=[],
|
||||
).start()
|
||||
patch("mcp_server.verify_preflight_purity", return_value=None).start()
|
||||
patch(
|
||||
"mcp_server.audit_reconciliation_mode.check_cleanup_execution_allowed",
|
||||
return_value=(True, []),
|
||||
).start()
|
||||
|
||||
def tearDown(self):
|
||||
patch.stopall()
|
||||
|
||||
def _merged_pr(self, number, branch, sha="c" * 40):
|
||||
return {
|
||||
"number": number,
|
||||
"title": f"PR {number}",
|
||||
"body": f"Closes #{number - 4}",
|
||||
"merged": True,
|
||||
"merged_at": "2026-07-23T12:00:00Z",
|
||||
"merge_commit_sha": "f" * 40,
|
||||
"state": "closed",
|
||||
"head": {"ref": branch, "sha": sha},
|
||||
"base": {"ref": "master"},
|
||||
}
|
||||
|
||||
def test_exact_pr_848_ignores_newer_852_in_batch_queue(self):
|
||||
"""pr_number=848 selects only #848 even when #852 is newer/first."""
|
||||
from mcp_server import gitea_reconcile_merged_cleanups
|
||||
|
||||
pr_848 = self._merged_pr(
|
||||
848, "fix/issue-844-exclude-epic-containers", sha="c3f282ba" + "0" * 32
|
||||
)
|
||||
# Closed list would rank #852 first in batch mode; exact pin must ignore it.
|
||||
closed_batch = [
|
||||
self._merged_pr(852, "fix/issue-851-cleanup-worktree-before-remote-delete"),
|
||||
pr_848,
|
||||
self._merged_pr(849, "fix/issue-849-other"),
|
||||
self._merged_pr(846, "fix/issue-846-other"),
|
||||
self._merged_pr(845, "fix/issue-845-other"),
|
||||
]
|
||||
batch_fetch_calls = []
|
||||
|
||||
def fake_api(method, url, *args, **kwargs):
|
||||
if method == "GET" and url.rstrip("/").endswith("/pulls/848"):
|
||||
return dict(pr_848)
|
||||
if method == "GET" and "/pulls/" in url:
|
||||
raise AssertionError(f"unexpected PR fetch: {url}")
|
||||
if method == "GET" and "/branches/" in url:
|
||||
return {"name": "present"}
|
||||
return {}
|
||||
|
||||
def fake_all(url, auth, limit=None):
|
||||
batch_fetch_calls.append((url, limit))
|
||||
if "state=open" in url:
|
||||
return []
|
||||
if "state=closed" in url:
|
||||
# Exact mode must not use the closed batch list.
|
||||
raise AssertionError(
|
||||
"exact pr_number mode must not page closed PRs: " + url
|
||||
)
|
||||
return []
|
||||
|
||||
self.mock_api.side_effect = fake_api
|
||||
self.mock_all.side_effect = fake_all
|
||||
patch(
|
||||
"mcp_server._remote_branch_exists",
|
||||
return_value=True,
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.build_reconciliation_report",
|
||||
side_effect=lambda **kwargs: {
|
||||
"entries": [
|
||||
{
|
||||
"pr_number": int(pr["number"]),
|
||||
"head_branch": (pr.get("head") or {}).get("ref"),
|
||||
"issue_number": 844,
|
||||
"remote_branch": {
|
||||
"safe_to_delete_remote": True,
|
||||
"head_branch": (pr.get("head") or {}).get("ref"),
|
||||
},
|
||||
"local_worktree": {
|
||||
"safe_to_remove_worktree": True,
|
||||
"worktree_path": (
|
||||
"/tmp/branches/fix-issue-844-exclude-epic-containers"
|
||||
),
|
||||
},
|
||||
"planned_execution_order": (
|
||||
mcp_server.merged_cleanup_reconcile.plan_cleanup_execution_order(
|
||||
remote_assessment={"safe_to_delete_remote": True},
|
||||
local_assessment={"safe_to_remove_worktree": True},
|
||||
)
|
||||
),
|
||||
}
|
||||
for pr in kwargs.get("closed_prs") or []
|
||||
if pr.get("merged_at") or pr.get("merged")
|
||||
],
|
||||
"reviewer_scratch_entries": [],
|
||||
"merged_pr_count": len(kwargs.get("closed_prs") or []),
|
||||
},
|
||||
).start()
|
||||
|
||||
res = gitea_reconcile_merged_cleanups(
|
||||
dry_run=True,
|
||||
pr_number=848,
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
)
|
||||
self.assertTrue(res.get("success"))
|
||||
self.assertFalse(res.get("performed"))
|
||||
self.assertEqual(res.get("selection_mode"), "exact_pr")
|
||||
self.assertEqual(res.get("selected_pr_number"), 848)
|
||||
entries = res.get("entries") or []
|
||||
self.assertEqual(len(entries), 1, entries)
|
||||
self.assertEqual(entries[0].get("pr_number"), 848)
|
||||
self.assertEqual(
|
||||
entries[0].get("head_branch"),
|
||||
"fix/issue-844-exclude-epic-containers",
|
||||
)
|
||||
# No other PR appears in plan.
|
||||
self.assertEqual(list((res.get("planned_execution_orders") or {}).keys()), ["848"])
|
||||
plan = (res.get("planned_execution_orders") or {}).get("848") or []
|
||||
actions = [s.get("action") for s in plan]
|
||||
self.assertEqual(
|
||||
actions,
|
||||
[
|
||||
"remove_local_worktree",
|
||||
"reassess_branch_ownership",
|
||||
"delete_remote_branch",
|
||||
],
|
||||
)
|
||||
# Prove we never scanned the multi-PR closed batch.
|
||||
self.assertFalse(any("state=closed" in (u or "") for u, _ in batch_fetch_calls))
|
||||
# closed_batch fixture must remain unused (sanity).
|
||||
self.assertEqual(closed_batch[0]["number"], 852)
|
||||
|
||||
def test_exact_pr_execute_only_mutates_selected_pr(self):
|
||||
"""Execute with pr_number must never touch #845/#846/#849/#852."""
|
||||
from mcp_server import gitea_reconcile_merged_cleanups
|
||||
|
||||
pr_848 = self._merged_pr(848, "fix/issue-844-exclude-epic-containers")
|
||||
worktree_path = "/tmp/branches/fix-issue-844-exclude-epic-containers"
|
||||
remove_calls = []
|
||||
delete_api_calls = []
|
||||
ownership_branches = []
|
||||
|
||||
def fake_api(method, url, *args, **kwargs):
|
||||
if method == "GET" and url.rstrip("/").endswith("/pulls/848"):
|
||||
return dict(pr_848)
|
||||
if method == "DELETE":
|
||||
delete_api_calls.append(url)
|
||||
# Forbid foreign PR branch deletion by URL content.
|
||||
for forbidden in ("845", "846", "849", "852"):
|
||||
self.assertNotIn(forbidden, url)
|
||||
return {}
|
||||
|
||||
def fake_remove(project_root, branch, worktree_path=None):
|
||||
remove_calls.append({"branch": branch, "worktree_path": worktree_path})
|
||||
return {
|
||||
"success": True,
|
||||
"performed": True,
|
||||
"message": f"removed {worktree_path}",
|
||||
"worktree_path": worktree_path,
|
||||
}
|
||||
|
||||
def fake_collect(**kwargs):
|
||||
ownership_branches.append(kwargs.get("branch"))
|
||||
return {"records": [], "inventory_error": False}
|
||||
|
||||
def fake_probe(h, o, r, auth, br):
|
||||
return guard.classify_branch_readback_http_status(
|
||||
404, not_found_scope=guard.NOT_FOUND_SCOPE_BRANCH
|
||||
)
|
||||
|
||||
self.mock_api.side_effect = fake_api
|
||||
self.mock_all.side_effect = lambda url, auth, limit=None: []
|
||||
patch("mcp_server._remote_branch_exists", return_value=True).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.build_reconciliation_report",
|
||||
return_value={
|
||||
"entries": [
|
||||
{
|
||||
"pr_number": 848,
|
||||
"head_branch": "fix/issue-844-exclude-epic-containers",
|
||||
"remote_branch": {"safe_to_delete_remote": True},
|
||||
"local_worktree": {
|
||||
"safe_to_remove_worktree": True,
|
||||
"worktree_path": worktree_path,
|
||||
},
|
||||
"planned_execution_order": [
|
||||
{"action": "remove_local_worktree", "phase": 1},
|
||||
{"action": "reassess_branch_ownership", "phase": 2},
|
||||
{"action": "delete_remote_branch", "phase": 3},
|
||||
],
|
||||
}
|
||||
],
|
||||
"reviewer_scratch_entries": [
|
||||
# Foreign scratch must be filtered before report execute loop;
|
||||
# if present here it would still be a test failure if acted on.
|
||||
],
|
||||
"merged_pr_count": 1,
|
||||
},
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.remove_local_worktree",
|
||||
side_effect=fake_remove,
|
||||
).start()
|
||||
patch(
|
||||
"mcp_server._collect_branch_ownership_records",
|
||||
side_effect=fake_collect,
|
||||
).start()
|
||||
patch("mcp_server._probe_remote_branch", side_effect=fake_probe).start()
|
||||
|
||||
res = gitea_reconcile_merged_cleanups(
|
||||
dry_run=False,
|
||||
execute_confirmed=True,
|
||||
pr_number=848,
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
)
|
||||
self.assertTrue(res.get("performed") or res.get("executed"))
|
||||
self.assertEqual(res.get("selection_mode"), "exact_pr")
|
||||
self.assertEqual(res.get("selected_pr_number"), 848)
|
||||
actions = res.get("actions") or []
|
||||
pr_numbers_touched = {
|
||||
a.get("pr_number") for a in actions if a.get("pr_number") is not None
|
||||
}
|
||||
self.assertTrue(pr_numbers_touched.issubset({None, 848}) or not pr_numbers_touched)
|
||||
removes = [a for a in actions if a.get("action") == "remove_local_worktree"]
|
||||
deletes = [a for a in actions if a.get("action") == "delete_remote_branch"]
|
||||
self.assertEqual(len(removes), 1)
|
||||
self.assertEqual(remove_calls[0]["branch"], "fix/issue-844-exclude-epic-containers")
|
||||
self.assertEqual(len(deletes), 1)
|
||||
self.assertTrue(deletes[0].get("success"))
|
||||
self.assertTrue(deletes[0].get("after_worktree_removal"))
|
||||
self.assertEqual(len(delete_api_calls), 1)
|
||||
self.assertEqual(
|
||||
ownership_branches, ["fix/issue-844-exclude-epic-containers"]
|
||||
)
|
||||
|
||||
def test_exact_pr_unknown_fails_closed_without_mutation(self):
|
||||
from mcp_server import gitea_reconcile_merged_cleanups
|
||||
|
||||
def fake_api(method, url, *args, **kwargs):
|
||||
if method == "GET" and "/pulls/99999" in url:
|
||||
raise RuntimeError("HTTP 404 Not Found")
|
||||
raise AssertionError(f"unexpected API call {method} {url}")
|
||||
|
||||
self.mock_api.side_effect = fake_api
|
||||
res = gitea_reconcile_merged_cleanups(
|
||||
dry_run=True,
|
||||
pr_number=99999,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertFalse(res.get("performed"))
|
||||
self.assertEqual(res.get("blocker_kind"), "pr_unresolvable")
|
||||
self.assertIn("99999", " ".join(res.get("reasons") or []))
|
||||
|
||||
def test_exact_pr_not_merged_fails_closed(self):
|
||||
from mcp_server import gitea_reconcile_merged_cleanups
|
||||
|
||||
def fake_api(method, url, *args, **kwargs):
|
||||
if method == "GET" and url.rstrip("/").endswith("/pulls/900"):
|
||||
return {
|
||||
"number": 900,
|
||||
"merged": False,
|
||||
"merged_at": None,
|
||||
"state": "open",
|
||||
"head": {"ref": "feat/x", "sha": "a" * 40},
|
||||
}
|
||||
raise AssertionError(f"unexpected {method} {url}")
|
||||
|
||||
self.mock_api.side_effect = fake_api
|
||||
res = gitea_reconcile_merged_cleanups(
|
||||
dry_run=False,
|
||||
execute_confirmed=True,
|
||||
pr_number=900,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertFalse(res.get("performed"))
|
||||
self.assertEqual(res.get("blocker_kind"), "pr_not_merged")
|
||||
|
||||
def test_exact_pr_invalid_number_fails_closed(self):
|
||||
from mcp_server import gitea_reconcile_merged_cleanups
|
||||
|
||||
res = gitea_reconcile_merged_cleanups(
|
||||
dry_run=True,
|
||||
pr_number=0,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertEqual(res.get("blocker_kind"), "invalid_pr_number")
|
||||
self.mock_api.assert_not_called()
|
||||
|
||||
def test_batch_mode_still_works_without_pr_number(self):
|
||||
"""Unfiltered batch path remains backward compatible."""
|
||||
from mcp_server import gitea_reconcile_merged_cleanups
|
||||
|
||||
self.mock_all.side_effect = lambda url, auth, limit=None: []
|
||||
self.mock_api.side_effect = lambda *a, **k: {}
|
||||
patch(
|
||||
"mcp_server.merged_cleanup_reconcile.build_reconciliation_report",
|
||||
return_value={
|
||||
"entries": [],
|
||||
"reviewer_scratch_entries": [],
|
||||
"merged_pr_count": 0,
|
||||
},
|
||||
).start()
|
||||
res = gitea_reconcile_merged_cleanups(dry_run=True, remote="prgs", limit=10)
|
||||
self.assertTrue(res.get("success"))
|
||||
self.assertEqual(res.get("selection_mode"), "batch")
|
||||
self.assertIsNone(res.get("selected_pr_number"))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -35,7 +35,7 @@ CONFIG = {
|
||||
],
|
||||
"forbidden_operations": [],
|
||||
"execution_profile": "full-author",
|
||||
"allowed_repositories": ["Example-Org/Example-Repo"],
|
||||
"allowed_repositories": ["Scaled-Tech-Consulting/Gitea-Tools", "Example-Org/Example-Repo"],
|
||||
},
|
||||
"reviewer-no-commit": {
|
||||
"enabled": True,
|
||||
@@ -50,7 +50,7 @@ CONFIG = {
|
||||
"gitea.repo.commit", "gitea.pr.create", "gitea.branch.push"
|
||||
],
|
||||
"execution_profile": "reviewer-no-commit",
|
||||
"allowed_repositories": ["Example-Org/Example-Repo"],
|
||||
"allowed_repositories": ["Scaled-Tech-Consulting/Gitea-Tools", "Example-Org/Example-Repo"],
|
||||
},
|
||||
},
|
||||
"rules": {"allow_runtime_switching": False},
|
||||
|
||||
@@ -175,7 +175,7 @@ class TestLauncherSnippets(unittest.TestCase):
|
||||
def test_only_safe_keys_no_secrets(self):
|
||||
entry = gitea_config.launcher_entry("prgs", "/cfg/profiles.json")["gitea-tools"]
|
||||
self.assertEqual(set(entry), {"command", "args", "env"})
|
||||
self.assertEqual(set(entry["env"]), {"GITEA_MCP_CONFIG", "GITEA_MCP_PROFILE"})
|
||||
self.assertEqual(set(entry["env"]), {"GITEA_MCP_CONFIG", "GITEA_MCP_PROFILE", "GITEA_CLIENT_MANAGED"})
|
||||
self.assertEqual(entry["env"]["GITEA_MCP_PROFILE"], "prgs")
|
||||
blob = json.dumps(entry).lower()
|
||||
for word in ("token", "password", "secret"):
|
||||
|
||||
@@ -11,8 +11,10 @@ from datetime import timedelta
|
||||
|
||||
from control_plane_db import (
|
||||
ControlPlaneDB,
|
||||
ControlPlaneError,
|
||||
InvalidWorkKindError,
|
||||
LeaseRequiredError,
|
||||
SCHEMA_VERSION,
|
||||
WORK_KINDS,
|
||||
_ts,
|
||||
_utc_now,
|
||||
@@ -36,7 +38,10 @@ class ControlPlaneDBTest(unittest.TestCase):
|
||||
rows = dict(conn.execute("SELECT key, value FROM schema_meta").fetchall())
|
||||
finally:
|
||||
conn.close()
|
||||
self.assertEqual(rows["schema_version"], "4")
|
||||
# Pinned to the constant, not a literal: every additive migration bumps
|
||||
# SCHEMA_VERSION, and the invariant under test is that the meta row
|
||||
# records the version the code actually wrote (#949 added v6).
|
||||
self.assertEqual(rows["schema_version"], str(SCHEMA_VERSION))
|
||||
self.assertIn("DB coordinates", rows["architecture"])
|
||||
self.assertIn("bridge", rows["architecture"].lower())
|
||||
|
||||
@@ -820,5 +825,229 @@ class ControlPlaneDBTest(unittest.TestCase):
|
||||
self.assertEqual(n, 1)
|
||||
|
||||
|
||||
class SessionCheckpointTest(unittest.TestCase):
|
||||
"""Durable MCP session checkpoint schema, redaction, and reconcile (#660)."""
|
||||
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.db_path = os.path.join(self._tmp.name, "cp.sqlite3")
|
||||
self.db = ControlPlaneDB(self.db_path)
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _write(self, **overrides):
|
||||
base = dict(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
session_id="prgs-author-1-abc",
|
||||
role="author",
|
||||
work_kind="issue",
|
||||
work_number=660,
|
||||
branch="feat/issue-660-session-checkpoint-schema",
|
||||
head_sha="deadbeef",
|
||||
lease_id="lease-1",
|
||||
workflow_stage="implementing",
|
||||
last_completed_action="wrote schema",
|
||||
next_valid_action="write tests",
|
||||
recovery_instructions="re-lock #660 then continue tests",
|
||||
)
|
||||
base.update(overrides)
|
||||
return self.db.write_session_checkpoint(**base)
|
||||
|
||||
# AC1 — schema documented and versioned.
|
||||
def test_table_exists_and_row_carries_schema_version(self) -> None:
|
||||
import sqlite3
|
||||
|
||||
conn = sqlite3.connect(self.db_path)
|
||||
try:
|
||||
names = {
|
||||
r[0]
|
||||
for r in conn.execute(
|
||||
"SELECT name FROM sqlite_master WHERE type='table'"
|
||||
).fetchall()
|
||||
}
|
||||
finally:
|
||||
conn.close()
|
||||
self.assertIn("session_checkpoints", names)
|
||||
record = self._write()
|
||||
self.assertEqual(record["checkpoint_schema_version"], SCHEMA_VERSION)
|
||||
|
||||
# AC2 — checkpoints written for multi-role session fixtures.
|
||||
def test_multi_role_fixtures_each_get_a_row(self) -> None:
|
||||
roles = [
|
||||
("prgs-author-1", "author", "issue", 660),
|
||||
("prgs-reviewer-2", "reviewer", "pr", 795),
|
||||
("prgs-merger-3", "merger", "pr", 862),
|
||||
("prgs-controller-4", "controller", "issue", 653),
|
||||
]
|
||||
for session_id, role, kind, number in roles:
|
||||
self._write(
|
||||
session_id=session_id,
|
||||
role=role,
|
||||
work_kind=kind,
|
||||
work_number=number,
|
||||
lease_id=f"lease-{session_id}",
|
||||
)
|
||||
rows = self.db.list_session_checkpoints(remote="prgs")
|
||||
self.assertEqual(len(rows), 4)
|
||||
self.assertEqual(
|
||||
{r["role"] for r in rows},
|
||||
{"author", "reviewer", "merger", "controller"},
|
||||
)
|
||||
|
||||
def test_upsert_is_current_state_and_audits_stage_change(self) -> None:
|
||||
first = self._write(workflow_stage="implementing")
|
||||
second = self._write(workflow_stage="testing")
|
||||
self.assertEqual(first["checkpoint_id"], second["checkpoint_id"])
|
||||
rows = self.db.list_session_checkpoints(
|
||||
remote="prgs", session_id="prgs-author-1-abc"
|
||||
)
|
||||
self.assertEqual(len(rows), 1)
|
||||
self.assertEqual(rows[0]["workflow_stage"], "testing")
|
||||
|
||||
import sqlite3
|
||||
|
||||
conn = sqlite3.connect(self.db_path)
|
||||
try:
|
||||
n = conn.execute(
|
||||
"SELECT COUNT(*) FROM events "
|
||||
"WHERE event_type = 'session_checkpoint_stage_change'"
|
||||
).fetchone()[0]
|
||||
finally:
|
||||
conn.close()
|
||||
self.assertEqual(n, 1)
|
||||
|
||||
def test_get_and_roundtrip_json_fields(self) -> None:
|
||||
self._write(
|
||||
capabilities=["gitea.repo.commit", "gitea.pr.create"],
|
||||
evidence={"tests": "4 passing"},
|
||||
pending_mutation={"op": "commit_files", "files": ["control_plane_db.py"]},
|
||||
)
|
||||
got = self.db.get_session_checkpoint(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
session_id="prgs-author-1-abc",
|
||||
work_kind="issue",
|
||||
work_number=660,
|
||||
)
|
||||
self.assertIsNotNone(got)
|
||||
self.assertEqual(got["capabilities"], ["gitea.repo.commit", "gitea.pr.create"])
|
||||
self.assertEqual(got["evidence"], {"tests": "4 passing"})
|
||||
self.assertEqual(got["pending_mutation"]["op"], "commit_files")
|
||||
|
||||
# AC4 — no secrets in stored records.
|
||||
def test_secrets_are_redacted_before_storage(self) -> None:
|
||||
self._write(
|
||||
recovery_instructions=(
|
||||
"resume with Authorization: Bearer sk-supersecrettoken then retry"
|
||||
),
|
||||
evidence={"authorization": "Bearer sk-anothersecret"},
|
||||
pending_mutation={"url": "https://user:[email protected]/repo.git"},
|
||||
)
|
||||
import sqlite3
|
||||
|
||||
conn = sqlite3.connect(self.db_path)
|
||||
try:
|
||||
row = conn.execute(
|
||||
"SELECT recovery_instructions, evidence, pending_mutation "
|
||||
"FROM session_checkpoints"
|
||||
).fetchone()
|
||||
finally:
|
||||
conn.close()
|
||||
blob = " ".join(str(v) for v in row)
|
||||
self.assertNotIn("sk-supersecrettoken", blob)
|
||||
self.assertNotIn("sk-anothersecret", blob)
|
||||
self.assertNotIn("password", blob)
|
||||
self.assertIn("REDACTED", blob)
|
||||
|
||||
# AC3 — reconcile detects stale head / lease mismatch.
|
||||
def test_reconcile_flags_stale_head(self) -> None:
|
||||
record = self._write(head_sha="aaaa1111")
|
||||
result = self.db.reconcile_session_checkpoint(
|
||||
record, live_head_sha="bbbb2222", live_lease_active=True,
|
||||
live_lease_id="lease-1",
|
||||
)
|
||||
self.assertTrue(result["stale"])
|
||||
self.assertTrue(result["head_mismatch"])
|
||||
self.assertFalse(result["lease_mismatch"])
|
||||
self.assertEqual(result["reconcile_action"], "reconcile_required")
|
||||
|
||||
def test_reconcile_flags_dead_lease(self) -> None:
|
||||
record = self._write(lease_id="lease-1", head_sha="aaaa1111")
|
||||
result = self.db.reconcile_session_checkpoint(
|
||||
record, live_head_sha="aaaa1111", live_lease_active=False,
|
||||
)
|
||||
self.assertTrue(result["stale"])
|
||||
self.assertFalse(result["head_mismatch"])
|
||||
self.assertTrue(result["lease_mismatch"])
|
||||
|
||||
def test_reconcile_reassigned_lease_is_stale(self) -> None:
|
||||
record = self._write(lease_id="lease-1")
|
||||
result = self.db.reconcile_session_checkpoint(
|
||||
record, live_lease_active=True, live_lease_id="lease-999",
|
||||
)
|
||||
self.assertTrue(result["lease_mismatch"])
|
||||
|
||||
def test_reconcile_clean_state_is_safe_to_resume(self) -> None:
|
||||
record = self._write(head_sha="aaaa1111", lease_id="lease-1")
|
||||
result = self.db.reconcile_session_checkpoint(
|
||||
record, live_head_sha="aaaa1111", live_lease_active=True,
|
||||
live_lease_id="lease-1",
|
||||
)
|
||||
self.assertFalse(result["stale"])
|
||||
self.assertEqual(result["reconcile_action"], "safe_to_resume")
|
||||
|
||||
def test_unknown_live_state_never_flags_mismatch(self) -> None:
|
||||
record = self._write(head_sha="aaaa1111", lease_id="lease-1")
|
||||
result = self.db.reconcile_session_checkpoint(record)
|
||||
self.assertFalse(result["stale"])
|
||||
|
||||
# Drain gate — fail closed when a checkpoint is incomplete.
|
||||
def test_drain_requires_complete_checkpoint(self) -> None:
|
||||
with self.assertRaises(ControlPlaneError):
|
||||
self.db.write_session_checkpoint(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
session_id="prgs-author-1-abc",
|
||||
role="author",
|
||||
workflow_stage="draining",
|
||||
# next_valid_action + recovery_instructions intentionally absent
|
||||
require_complete=True,
|
||||
)
|
||||
# Nothing was written.
|
||||
rows = self.db.list_session_checkpoints(remote="prgs")
|
||||
self.assertEqual(rows, [])
|
||||
|
||||
def test_drain_write_succeeds_when_complete(self) -> None:
|
||||
record = self._write(require_complete=True)
|
||||
self.assertEqual(record["status"], "active")
|
||||
self.assertEqual(self.db.checkpoint_completeness(record), [])
|
||||
|
||||
def test_missing_session_id_fails_closed(self) -> None:
|
||||
with self.assertRaises(ControlPlaneError):
|
||||
self.db.write_session_checkpoint(
|
||||
remote="prgs", org="o", repo="r", session_id="",
|
||||
)
|
||||
|
||||
def test_session_level_checkpoint_uses_sentinel_key(self) -> None:
|
||||
# No work unit -> ('', 0) sentinel; a second session-level write upserts.
|
||||
self.db.write_session_checkpoint(
|
||||
remote="prgs", org="o", repo="r", session_id="s-sess",
|
||||
workflow_stage="idle",
|
||||
)
|
||||
self.db.write_session_checkpoint(
|
||||
remote="prgs", org="o", repo="r", session_id="s-sess",
|
||||
workflow_stage="booting",
|
||||
)
|
||||
rows = self.db.list_session_checkpoints(remote="prgs", session_id="s-sess")
|
||||
self.assertEqual(len(rows), 1)
|
||||
self.assertEqual(rows[0]["work_kind"], "")
|
||||
self.assertEqual(rows[0]["work_number"], 0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -0,0 +1,282 @@
|
||||
"""Control-plane MCP server runtime registry (#949).
|
||||
|
||||
The registry is the half of the fleet evidence a caller cannot supply: each
|
||||
server writes exactly one row about itself at native transport bind. These tests
|
||||
pin the storage contract — additive migration, deterministic ordering, and
|
||||
PID-reuse/retention pruning that cannot hide a live duplicate.
|
||||
"""
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from unittest import mock
|
||||
|
||||
import control_plane_db
|
||||
import mcp_fleet_inventory as mfi
|
||||
|
||||
|
||||
def _stamp(delta_seconds=0):
|
||||
moment = datetime.now(timezone.utc) + timedelta(seconds=delta_seconds)
|
||||
return moment.replace(microsecond=0).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
|
||||
|
||||
class _DBCase(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db_path = os.path.join(self._tmp.name, "control_plane.sqlite3")
|
||||
self.db = control_plane_db.ControlPlaneDB(db_path=self.db_path)
|
||||
|
||||
def record(self, namespace, profile, role, pid, **overrides):
|
||||
base = mfi.build_process_runtime_record(
|
||||
namespace=namespace,
|
||||
profile=profile,
|
||||
role=role,
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
repository_root="/checkout/Gitea-Tools",
|
||||
pid=pid,
|
||||
startup_head="82d71b77028a7abd4f8ab4a4e4d89658a187f73d",
|
||||
transport="stdio",
|
||||
client_provenance="client_managed",
|
||||
env={mfi.COHORT_ID_ENV: "ppid:40990"},
|
||||
boot_id=f"boot{pid}",
|
||||
)
|
||||
base.update(overrides)
|
||||
return base
|
||||
|
||||
|
||||
class TestSchema(_DBCase):
|
||||
def test_schema_version_is_bumped(self):
|
||||
self.assertGreaterEqual(control_plane_db.SCHEMA_VERSION, 6)
|
||||
|
||||
def test_runtime_table_exists_on_a_fresh_database(self):
|
||||
self.assertEqual(self.db.list_mcp_server_runtimes(), [])
|
||||
|
||||
def test_migration_is_additive_and_idempotent(self):
|
||||
"""Re-opening an existing DB must not disturb the other tables."""
|
||||
self.db.upsert_session(session_id="s-a", role="author", profile="prgs-author")
|
||||
self.db.register_mcp_server_runtime(
|
||||
self.record("gitea-author", "prgs-author", "author", 41000)
|
||||
)
|
||||
reopened = control_plane_db.ControlPlaneDB(db_path=self.db_path)
|
||||
self.assertEqual(len(reopened.list_sessions()), 1)
|
||||
self.assertEqual(len(reopened.list_mcp_server_runtimes()), 1)
|
||||
|
||||
|
||||
class TestRegistration(_DBCase):
|
||||
def test_registration_round_trips_every_field(self):
|
||||
record = self.record("gitea-controller", "prgs-controller", "controller", 41001)
|
||||
stored = self.db.register_mcp_server_runtime(record)
|
||||
for key in (
|
||||
"runtime_id",
|
||||
"namespace",
|
||||
"profile",
|
||||
"role",
|
||||
"remote",
|
||||
"org",
|
||||
"repo",
|
||||
"repository_root",
|
||||
"pid",
|
||||
"cohort_id",
|
||||
"cohort_source",
|
||||
"client_provenance",
|
||||
"boot_id",
|
||||
"startup_head",
|
||||
"transport",
|
||||
"status",
|
||||
):
|
||||
self.assertEqual(stored[key], record[key], key)
|
||||
|
||||
def test_registration_requires_a_runtime_id(self):
|
||||
record = self.record("gitea-author", "prgs-author", "author", 41002)
|
||||
record["runtime_id"] = ""
|
||||
with self.assertRaises(ValueError):
|
||||
self.db.register_mcp_server_runtime(record)
|
||||
|
||||
def test_registration_requires_a_namespace(self):
|
||||
record = self.record("gitea-author", "prgs-author", "author", 41003)
|
||||
record["namespace"] = " "
|
||||
with self.assertRaises(ValueError):
|
||||
self.db.register_mcp_server_runtime(record)
|
||||
|
||||
def test_registration_requires_a_pid(self):
|
||||
record = self.record("gitea-author", "prgs-author", "author", 41004)
|
||||
record["pid"] = None
|
||||
with self.assertRaises(ValueError):
|
||||
self.db.register_mcp_server_runtime(record)
|
||||
|
||||
def test_five_members_register_independently(self):
|
||||
for index, entry in enumerate(mfi.EXPECTED_PRGS_FLEET):
|
||||
self.db.register_mcp_server_runtime(
|
||||
self.record(
|
||||
entry["namespace"], entry["profile"], entry["role"], 41100 + index
|
||||
)
|
||||
)
|
||||
rows = self.db.list_mcp_server_runtimes()
|
||||
self.assertEqual(len(rows), 5)
|
||||
self.assertEqual(
|
||||
sorted(r["profile"] for r in rows),
|
||||
[
|
||||
"prgs-author",
|
||||
"prgs-controller",
|
||||
"prgs-merger",
|
||||
"prgs-reconciler",
|
||||
"prgs-reviewer",
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
class TestPruning(_DBCase):
|
||||
def test_reusing_a_pid_replaces_the_stale_row(self):
|
||||
first = self.record("gitea-author", "prgs-author", "author", 41200)
|
||||
self.db.register_mcp_server_runtime(first)
|
||||
second = self.record("gitea-reviewer", "prgs-reviewer", "reviewer", 41200)
|
||||
self.db.register_mcp_server_runtime(second)
|
||||
rows = self.db.list_mcp_server_runtimes()
|
||||
self.assertEqual(len(rows), 1)
|
||||
self.assertEqual(rows[0]["runtime_id"], second["runtime_id"])
|
||||
|
||||
def test_pruning_never_removes_a_live_duplicate_on_another_pid(self):
|
||||
"""The defect this must not have: hiding a second running server."""
|
||||
first = self.record("gitea-author", "prgs-author", "author", 41300)
|
||||
self.db.register_mcp_server_runtime(first)
|
||||
second = self.record("gitea-author", "prgs-author", "author", 41301)
|
||||
self.db.register_mcp_server_runtime(
|
||||
second, retention_seconds=mfi.RUNTIME_RETENTION_SECONDS
|
||||
)
|
||||
rows = self.db.list_mcp_server_runtimes()
|
||||
self.assertEqual(len(rows), 2)
|
||||
self.assertEqual(sorted(r["pid"] for r in rows), [41300, 41301])
|
||||
|
||||
def test_retention_drops_rows_older_than_the_window(self):
|
||||
old = self.record(
|
||||
"gitea-merger",
|
||||
"prgs-merger",
|
||||
"merger",
|
||||
41400,
|
||||
registered_at=_stamp(-(mfi.RUNTIME_RETENTION_SECONDS + 3600)),
|
||||
)
|
||||
self.db.register_mcp_server_runtime(old)
|
||||
fresh = self.record("gitea-author", "prgs-author", "author", 41401)
|
||||
self.db.register_mcp_server_runtime(
|
||||
fresh, retention_seconds=mfi.RUNTIME_RETENTION_SECONDS
|
||||
)
|
||||
rows = self.db.list_mcp_server_runtimes()
|
||||
self.assertEqual([r["pid"] for r in rows], [41401])
|
||||
|
||||
def test_retention_is_skipped_when_not_requested(self):
|
||||
old = self.record(
|
||||
"gitea-merger",
|
||||
"prgs-merger",
|
||||
"merger",
|
||||
41500,
|
||||
registered_at=_stamp(-(mfi.RUNTIME_RETENTION_SECONDS + 3600)),
|
||||
)
|
||||
self.db.register_mcp_server_runtime(old)
|
||||
self.db.register_mcp_server_runtime(
|
||||
self.record("gitea-author", "prgs-author", "author", 41501)
|
||||
)
|
||||
self.assertEqual(len(self.db.list_mcp_server_runtimes()), 2)
|
||||
|
||||
|
||||
class TestListing(_DBCase):
|
||||
def test_listing_is_deterministically_ordered(self):
|
||||
for index in range(5):
|
||||
self.db.register_mcp_server_runtime(
|
||||
self.record(
|
||||
"gitea-author",
|
||||
"prgs-author",
|
||||
"author",
|
||||
41600 + index,
|
||||
registered_at="2026-07-28T02:00:00Z",
|
||||
)
|
||||
)
|
||||
first = [r["runtime_id"] for r in self.db.list_mcp_server_runtimes()]
|
||||
second = [r["runtime_id"] for r in self.db.list_mcp_server_runtimes()]
|
||||
self.assertEqual(first, second)
|
||||
self.assertEqual(first, sorted(first))
|
||||
|
||||
def test_status_filter_selects_running_rows(self):
|
||||
running = self.record("gitea-author", "prgs-author", "author", 41700)
|
||||
self.db.register_mcp_server_runtime(running)
|
||||
stopped = self.record("gitea-merger", "prgs-merger", "merger", 41701)
|
||||
self.db.register_mcp_server_runtime(stopped)
|
||||
self.db.mark_mcp_server_runtime_stopped(stopped["runtime_id"])
|
||||
rows = self.db.list_mcp_server_runtimes(statuses=("running",))
|
||||
self.assertEqual([r["pid"] for r in rows], [41700])
|
||||
|
||||
def test_listing_without_a_filter_returns_stopped_rows_too(self):
|
||||
stopped = self.record("gitea-merger", "prgs-merger", "merger", 41800)
|
||||
self.db.register_mcp_server_runtime(stopped)
|
||||
self.db.mark_mcp_server_runtime_stopped(stopped["runtime_id"])
|
||||
rows = self.db.list_mcp_server_runtimes()
|
||||
self.assertEqual(len(rows), 1)
|
||||
self.assertEqual(rows[0]["status"], "stopped")
|
||||
|
||||
|
||||
class TestHeartbeat(_DBCase):
|
||||
def test_heartbeat_advances_only_the_timestamp(self):
|
||||
record = self.record(
|
||||
"gitea-author",
|
||||
"prgs-author",
|
||||
"author",
|
||||
41900,
|
||||
last_heartbeat_at="2026-07-28T02:00:00Z",
|
||||
)
|
||||
self.db.register_mcp_server_runtime(record)
|
||||
self.db.heartbeat_mcp_server_runtime(record["runtime_id"])
|
||||
row = self.db.list_mcp_server_runtimes()[0]
|
||||
self.assertNotEqual(row["last_heartbeat_at"], "2026-07-28T02:00:00Z")
|
||||
self.assertEqual(row["status"], "running")
|
||||
self.assertEqual(row["pid"], 41900)
|
||||
|
||||
def test_heartbeat_for_an_unknown_runtime_is_a_no_op(self):
|
||||
self.db.heartbeat_mcp_server_runtime("does-not-exist")
|
||||
self.assertEqual(self.db.list_mcp_server_runtimes(), [])
|
||||
|
||||
|
||||
class TestEndToEndSnapshot(_DBCase):
|
||||
"""Registry rows feed the classifier without reshaping."""
|
||||
|
||||
def test_registered_fleet_classifies_as_healthy(self):
|
||||
pids = []
|
||||
for index, entry in enumerate(mfi.EXPECTED_PRGS_FLEET):
|
||||
pid = 42000 + index
|
||||
pids.append(pid)
|
||||
self.db.register_mcp_server_runtime(
|
||||
self.record(
|
||||
entry["namespace"],
|
||||
entry["profile"],
|
||||
entry["role"],
|
||||
pid,
|
||||
registered_at="2026-07-28T02:00:00Z",
|
||||
)
|
||||
)
|
||||
rows = self.db.list_mcp_server_runtimes(statuses=("running",))
|
||||
scan = {
|
||||
"available": True,
|
||||
"processes": [
|
||||
{"pid": pid, "started_at": "2026-07-28T01:00:00Z"} for pid in pids
|
||||
],
|
||||
"reason": None,
|
||||
}
|
||||
with mock.patch.object(mfi, "probe_pid_alive", return_value=True):
|
||||
result = mfi.classify_fleet_inventory(
|
||||
runtime_rows=rows,
|
||||
process_scan=scan,
|
||||
expected_binding={
|
||||
"remote": "prgs",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
},
|
||||
)
|
||||
self.assertTrue(result["inventory_complete"])
|
||||
self.assertTrue(result["mutation_gate_satisfied"])
|
||||
self.assertEqual(result["running_member_count"], 5)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,483 @@
|
||||
"""Synthetic regression coverage for dirty orphaned worktree recovery (#860).
|
||||
|
||||
Modeled on the #850 / #855 shape without mutating their real state.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest import mock
|
||||
|
||||
import dirty_orphan_worktree_recovery as dorec
|
||||
import issue_lock_store
|
||||
|
||||
|
||||
DEAD_PID = 999_999_999
|
||||
LIVE_PID = os.getpid()
|
||||
BRANCH = "fix/issue-901-dirty-orphan"
|
||||
SOURCE_WT = "/repo/branches/issue-901-dirty-orphan"
|
||||
RECOVERY_WT_NAME = "recovery-issue-901-dirty-orphan"
|
||||
LOCAL_HEAD = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
REMOTE_HEAD = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||
OTHER_HEAD = "cccccccccccccccccccccccccccccccccccccccc"
|
||||
FP_A = dorec.sha256_bytes(b"dirty-a")
|
||||
FP_B = dorec.sha256_bytes(b"dirty-b")
|
||||
FP_C = dorec.sha256_bytes(b"dirty-c-conflict")
|
||||
|
||||
|
||||
def durable_lock(**overrides):
|
||||
"""#850-shaped PID-less malformed same-claimant lock."""
|
||||
lock = {
|
||||
"issue_number": 901,
|
||||
"branch_name": BRANCH,
|
||||
"worktree_path": SOURCE_WT,
|
||||
"remote": "prgs",
|
||||
"org": "Example-Org",
|
||||
"repo": "Example-Repo",
|
||||
# intentionally no pid / session_pid / work_lease expiry
|
||||
"claimant": {"username": "author-user", "profile": "prgs-author"},
|
||||
}
|
||||
lock.update(overrides)
|
||||
return lock
|
||||
|
||||
|
||||
def base_kwargs(**overrides):
|
||||
kwargs = {
|
||||
"issue_number": 901,
|
||||
"branch_name": BRANCH,
|
||||
"source_worktree_path": SOURCE_WT,
|
||||
"remote": "prgs",
|
||||
"org": "Example-Org",
|
||||
"repo": "Example-Repo",
|
||||
"identity": "author-user",
|
||||
"profile": "prgs-author",
|
||||
"expected_local_head": LOCAL_HEAD,
|
||||
"expected_remote_head": REMOTE_HEAD,
|
||||
"expected_dirty_fingerprints": {"a.py": FP_A, "b.py": FP_B},
|
||||
"current_branch": BRANCH,
|
||||
"porcelain_status": " M a.py\n M b.py\n",
|
||||
"observed_local_head": LOCAL_HEAD,
|
||||
"observed_remote_head": REMOTE_HEAD,
|
||||
"observed_dirty_fingerprints": {"a.py": FP_A, "b.py": FP_B},
|
||||
"competing_live_locks": [],
|
||||
"competing_live_sessions": [],
|
||||
"workflow_lease_active": False,
|
||||
"workflow_lease_expired": True,
|
||||
"canonical_repo_root": "/repo",
|
||||
"worktree_registered": True,
|
||||
"current_pid": LIVE_PID,
|
||||
}
|
||||
kwargs.update(overrides)
|
||||
return kwargs
|
||||
|
||||
|
||||
def assess(lock=None, **overrides):
|
||||
return dorec.assess_dirty_orphan_recovery(
|
||||
durable_lock() if lock is None else lock, **base_kwargs(**overrides)
|
||||
)
|
||||
|
||||
|
||||
class FreshnessPidLess(unittest.TestCase):
|
||||
def test_pid_less_lock_is_not_live(self):
|
||||
freshness = issue_lock_store.assess_lock_freshness(durable_lock())
|
||||
self.assertFalse(freshness["live"])
|
||||
self.assertTrue(freshness.get("pid_missing"))
|
||||
self.assertEqual(freshness["status"], "malformed")
|
||||
|
||||
def test_pid_less_with_far_future_expiry_still_not_live(self):
|
||||
lock = durable_lock(
|
||||
work_lease={
|
||||
"operation_type": "author_issue_work",
|
||||
"expires_at": "2999-01-01T00:00:00Z",
|
||||
"last_heartbeat_at": "2999-01-01T00:00:00Z",
|
||||
}
|
||||
)
|
||||
freshness = issue_lock_store.assess_lock_freshness(lock)
|
||||
self.assertFalse(freshness["live"])
|
||||
self.assertTrue(freshness.get("pid_missing"))
|
||||
|
||||
|
||||
class EligibilityGranted(unittest.TestCase):
|
||||
def test_dead_same_claimant_pid_less_dirty(self):
|
||||
result = assess()
|
||||
self.assertEqual(result["outcome"], dorec.ELIGIBLE)
|
||||
self.assertTrue(result["eligible"])
|
||||
|
||||
def test_expired_workflow_lease_corroboration(self):
|
||||
result = assess(workflow_lease_active=False, workflow_lease_expired=True)
|
||||
self.assertTrue(result["eligible"])
|
||||
|
||||
def test_older_local_newer_remote_heads(self):
|
||||
result = assess()
|
||||
self.assertTrue(result["evidence"].get("heads_diverged"))
|
||||
self.assertTrue(result["eligible"])
|
||||
|
||||
|
||||
class EligibilityRefused(unittest.TestCase):
|
||||
def test_active_owner_with_pid(self):
|
||||
lock = durable_lock(pid=LIVE_PID, session_pid=LIVE_PID)
|
||||
result = assess(lock=lock, owner_process_alive_override=True)
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
self.assertFalse(result["eligible"])
|
||||
self.assertTrue(any("alive" in r for r in result["reasons"]))
|
||||
|
||||
def test_foreign_claimant(self):
|
||||
result = assess(identity="other-user")
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
self.assertTrue(any("foreign claimant identity" in r for r in result["reasons"]))
|
||||
|
||||
def test_foreign_profile(self):
|
||||
result = assess(profile="prgs-reviewer")
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
|
||||
def test_fingerprint_mismatch(self):
|
||||
result = assess(observed_dirty_fingerprints={"a.py": "0" * 64, "b.py": FP_B})
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
self.assertTrue(any("fingerprint mismatch" in r for r in result["reasons"]))
|
||||
|
||||
def test_head_mismatch(self):
|
||||
result = assess(observed_local_head=OTHER_HEAD)
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
|
||||
def test_remote_head_mismatch(self):
|
||||
result = assess(observed_remote_head=OTHER_HEAD)
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
|
||||
def test_path_not_under_branches(self):
|
||||
result = assess(
|
||||
source_worktree_path="/tmp/branches/evil",
|
||||
# lock path also changed so worktree agreement holds
|
||||
lock=durable_lock(worktree_path="/tmp/branches/evil"),
|
||||
)
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
self.assertTrue(any("canonical branches" in r for r in result["reasons"]))
|
||||
|
||||
def test_unregistered_worktree(self):
|
||||
result = assess(worktree_registered=False)
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
|
||||
def test_active_workflow_lease(self):
|
||||
result = assess(workflow_lease_active=True, workflow_lease_expired=False)
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
|
||||
def test_unsafe_dirty_path_pin(self):
|
||||
result = assess(
|
||||
expected_dirty_fingerprints={"../etc/passwd": FP_A},
|
||||
observed_dirty_fingerprints={"../etc/passwd": FP_A},
|
||||
)
|
||||
self.assertEqual(result["outcome"], dorec.REFUSED)
|
||||
|
||||
def test_symlink_escape_rejected_by_ancestry(self):
|
||||
ok, reasons = dorec.is_path_under_canonical_branches(
|
||||
"/tmp/branches/evil", canonical_repo_root="/repo"
|
||||
)
|
||||
self.assertFalse(ok)
|
||||
self.assertTrue(reasons)
|
||||
|
||||
|
||||
class ConflictDetection(unittest.TestCase):
|
||||
def test_overlapping_upstream_change(self):
|
||||
conflicts = dorec.detect_path_conflicts(
|
||||
dirty_paths=["c.py"],
|
||||
local_head_contents={"c.py": b"local-base"},
|
||||
remote_head_contents={"c.py": b"remote-changed"},
|
||||
dirty_contents={"c.py": b"dirty-c-conflict"},
|
||||
)
|
||||
self.assertEqual(len(conflicts), 1)
|
||||
self.assertEqual(conflicts[0]["path"], "c.py")
|
||||
|
||||
def test_unchanged_upstream_no_conflict(self):
|
||||
conflicts = dorec.detect_path_conflicts(
|
||||
dirty_paths=["a.py"],
|
||||
local_head_contents={"a.py": b"same"},
|
||||
remote_head_contents={"a.py": b"same"},
|
||||
dirty_contents={"a.py": b"dirty-a"},
|
||||
)
|
||||
self.assertEqual(conflicts, [])
|
||||
|
||||
|
||||
class CrashSafeRecovery(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.tmp = tempfile.mkdtemp(prefix="dirty-orphan-")
|
||||
self.repo = os.path.join(self.tmp, "repo")
|
||||
self.branches = os.path.join(self.repo, "branches")
|
||||
self.source = os.path.join(self.branches, "issue-901-dirty-orphan")
|
||||
self.recovery = os.path.join(self.branches, RECOVERY_WT_NAME)
|
||||
os.makedirs(self.source, exist_ok=True)
|
||||
os.makedirs(self.branches, exist_ok=True)
|
||||
# seed dirty files in source
|
||||
with open(os.path.join(self.source, "a.py"), "wb") as fh:
|
||||
fh.write(b"dirty-a")
|
||||
with open(os.path.join(self.source, "b.py"), "wb") as fh:
|
||||
fh.write(b"dirty-b")
|
||||
self.journal_dir = os.path.join(self.tmp, "journals")
|
||||
self.lock = durable_lock(worktree_path=self.source)
|
||||
self.assessment = dorec.assess_dirty_orphan_recovery(
|
||||
self.lock,
|
||||
**base_kwargs(
|
||||
source_worktree_path=self.source,
|
||||
canonical_repo_root=self.repo,
|
||||
),
|
||||
)
|
||||
|
||||
class FakeGit(dorec.GitOps):
|
||||
def __init__(self, recovery_path, head):
|
||||
self.recovery_path = recovery_path
|
||||
self.head = head
|
||||
self.calls = []
|
||||
|
||||
def run(self, args, *, cwd):
|
||||
self.calls.append((args, cwd))
|
||||
if args[:3] == ["git", "worktree", "add"]:
|
||||
os.makedirs(self.recovery_path, exist_ok=True)
|
||||
return mock.Mock(returncode=0, stdout="", stderr="")
|
||||
if args[:2] == ["git", "checkout"]:
|
||||
return mock.Mock(returncode=0, stdout="", stderr="")
|
||||
if args[:2] == ["git", "rev-parse"]:
|
||||
return mock.Mock(returncode=0, stdout=self.head + "\n", stderr="")
|
||||
return mock.Mock(returncode=0, stdout="", stderr="")
|
||||
|
||||
self.git = FakeGit(self.recovery, REMOTE_HEAD)
|
||||
self.written_locks = []
|
||||
|
||||
def lock_writer(record):
|
||||
self.written_locks.append(record)
|
||||
|
||||
self.lock_writer = lock_writer
|
||||
|
||||
def tearDown(self):
|
||||
shutil.rmtree(self.tmp, ignore_errors=True)
|
||||
|
||||
def _run(self, **overrides):
|
||||
kwargs = {
|
||||
"assessment": self.assessment,
|
||||
"existing_lock": self.lock,
|
||||
"issue_number": 901,
|
||||
"branch_name": BRANCH,
|
||||
"source_worktree_path": self.source,
|
||||
"recovery_worktree_path": self.recovery,
|
||||
"remote": "prgs",
|
||||
"org": "Example-Org",
|
||||
"repo": "Example-Repo",
|
||||
"identity": "author-user",
|
||||
"profile": "prgs-author",
|
||||
"expected_local_head": LOCAL_HEAD,
|
||||
"expected_remote_head": REMOTE_HEAD,
|
||||
"expected_dirty_fingerprints": {"a.py": FP_A, "b.py": FP_B},
|
||||
"dirty_contents": {"a.py": b"dirty-a", "b.py": b"dirty-b"},
|
||||
"local_head_contents": {"a.py": b"base-a", "b.py": b"base-b"},
|
||||
"remote_head_contents": {"a.py": b"base-a", "b.py": b"base-b"},
|
||||
"canonical_repo_root": self.repo,
|
||||
"bind_lock": True,
|
||||
"lock_writer": self.lock_writer,
|
||||
"git_ops": self.git,
|
||||
"journal_dir": self.journal_dir,
|
||||
"session_pid": LIVE_PID,
|
||||
}
|
||||
kwargs.update(overrides)
|
||||
return dorec.run_dirty_orphan_recovery(**kwargs)
|
||||
|
||||
def test_success_preserves_dirty_bytes_and_source(self):
|
||||
result = self._run()
|
||||
self.assertTrue(result["success"])
|
||||
self.assertEqual(result["outcome"], dorec.RECOVERY_COMPLETED)
|
||||
self.assertTrue(os.path.isdir(self.source))
|
||||
with open(os.path.join(self.source, "a.py"), "rb") as fh:
|
||||
self.assertEqual(fh.read(), b"dirty-a")
|
||||
with open(os.path.join(self.recovery, "a.py"), "rb") as fh:
|
||||
self.assertEqual(fh.read(), b"dirty-a")
|
||||
with open(os.path.join(self.recovery, "b.py"), "rb") as fh:
|
||||
self.assertEqual(fh.read(), b"dirty-b")
|
||||
self.assertEqual(len(self.written_locks), 1)
|
||||
rec = self.written_locks[0]
|
||||
self.assertEqual(rec["session_pid"], LIVE_PID)
|
||||
self.assertTrue(rec["dirty_orphan_recovery"]["recovered"])
|
||||
self.assertTrue(rec["dirty_orphan_recovery"]["source_frozen"])
|
||||
|
||||
def test_conflict_leaves_governed_state(self):
|
||||
result = self._run(
|
||||
expected_dirty_fingerprints={"c.py": FP_C},
|
||||
dirty_contents={"c.py": b"dirty-c-conflict"},
|
||||
local_head_contents={"c.py": b"local-base"},
|
||||
remote_head_contents={"c.py": b"remote-changed"},
|
||||
)
|
||||
# #860 F4: session binding is NOT finalized while conflicts remain
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], dorec.CONFLICTS_PRESENT)
|
||||
sidecar = os.path.join(self.recovery, "c.py.recovered-dirty")
|
||||
self.assertTrue(os.path.isfile(sidecar))
|
||||
state = os.path.join(
|
||||
self.recovery, dorec.CONFLICT_STATE_DIR, dorec.CONFLICT_STATE_FILE
|
||||
)
|
||||
self.assertTrue(os.path.isfile(state))
|
||||
with open(state, "r", encoding="utf-8") as fh:
|
||||
payload = json.load(fh)
|
||||
self.assertEqual(payload["resolution"], "author_edit_required")
|
||||
|
||||
def test_interrupt_before_journal_no_artifacts(self):
|
||||
result = self._run(interrupt_after_phase=dorec.PHASE_ELIGIBILITY)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], "INTERRUPTED")
|
||||
self.assertFalse(os.path.isdir(self.recovery))
|
||||
|
||||
def test_interrupt_after_journal_then_retry_idempotent(self):
|
||||
first = self._run(interrupt_after_phase=dorec.PHASE_JOURNAL_PERSISTED)
|
||||
self.assertEqual(first["outcome"], "INTERRUPTED")
|
||||
self.assertTrue(first["journal"]["artifacts_created"]["journal"])
|
||||
second = self._run()
|
||||
self.assertTrue(second["success"])
|
||||
# source still recoverable
|
||||
with open(os.path.join(self.source, "a.py"), "rb") as fh:
|
||||
self.assertEqual(fh.read(), b"dirty-a")
|
||||
|
||||
def test_interrupt_after_worktree_then_retry(self):
|
||||
first = self._run(interrupt_after_phase=dorec.PHASE_RECOVERY_WORKTREE)
|
||||
self.assertEqual(first["outcome"], "INTERRUPTED")
|
||||
self.assertTrue(os.path.isdir(self.recovery))
|
||||
second = self._run()
|
||||
self.assertTrue(second["success"])
|
||||
|
||||
def test_interrupt_after_binding_then_retry_complete(self):
|
||||
first = self._run(interrupt_after_phase=dorec.PHASE_BINDING)
|
||||
self.assertEqual(first["outcome"], "INTERRUPTED")
|
||||
second = self._run()
|
||||
self.assertTrue(second["success"])
|
||||
# completed journal makes further retries no-ops
|
||||
third = self._run()
|
||||
self.assertEqual(third["outcome"], dorec.RECOVERY_RESUMED)
|
||||
|
||||
def test_source_worktree_never_deleted(self):
|
||||
self._run()
|
||||
self.assertTrue(os.path.isdir(self.source))
|
||||
self.assertTrue(os.path.isfile(os.path.join(self.source, "a.py")))
|
||||
|
||||
def test_fingerprint_drift_refuses_without_mutation(self):
|
||||
result = self._run(dirty_contents={"a.py": b"CHANGED", "b.py": b"dirty-b"})
|
||||
self.assertFalse(result["success"])
|
||||
self.assertFalse(os.path.isdir(self.recovery))
|
||||
|
||||
|
||||
class SessionBindingPreflight(unittest.TestCase):
|
||||
def test_canonical_session_binding_recognized(self):
|
||||
lock = {
|
||||
"worktree_path": "/repo/branches/recovery",
|
||||
"session_pid": LIVE_PID,
|
||||
"dirty_orphan_recovery": {
|
||||
"recovered": True,
|
||||
"conflicts": [],
|
||||
"recovery_worktree_path": "/repo/branches/recovery",
|
||||
"source_worktree_path": SOURCE_WT,
|
||||
"accepted_head": REMOTE_HEAD,
|
||||
},
|
||||
}
|
||||
result = dorec.preflight_recognizes_recovered_provenance(lock)
|
||||
self.assertTrue(result["recognized"])
|
||||
|
||||
def test_conflicts_block_commit_preflight(self):
|
||||
lock = {
|
||||
"worktree_path": "/repo/branches/recovery",
|
||||
"session_pid": LIVE_PID,
|
||||
"dirty_orphan_recovery": {
|
||||
"recovered": True,
|
||||
"conflicts": [{"path": "c.py"}],
|
||||
},
|
||||
}
|
||||
result = dorec.preflight_recognizes_recovered_provenance(lock)
|
||||
self.assertFalse(result["recognized"])
|
||||
|
||||
def test_active_foreign_does_not_mutate(self):
|
||||
# assess-only path: foreign refused before run
|
||||
result = assess(identity="intruder")
|
||||
self.assertFalse(result["eligible"])
|
||||
|
||||
|
||||
class JournalSymlinkRefusal(unittest.TestCase):
|
||||
def test_symlink_journal_path_refused_on_load(self):
|
||||
tmp = tempfile.mkdtemp()
|
||||
try:
|
||||
real = os.path.join(tmp, "real.json")
|
||||
with open(real, "w", encoding="utf-8") as fh:
|
||||
fh.write("{}")
|
||||
link = os.path.join(tmp, "link.json")
|
||||
os.symlink(real, link)
|
||||
key = "symlink-test"
|
||||
jdir = tmp
|
||||
path = dorec._journal_path(key, journal_dir=jdir)
|
||||
with open(path, "w", encoding="utf-8") as fh:
|
||||
json.dump({"idempotency_key": key}, fh)
|
||||
os.remove(path)
|
||||
os.symlink(real, path)
|
||||
with self.assertRaises(ValueError):
|
||||
dorec.load_journal(key, journal_dir=jdir)
|
||||
finally:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
|
||||
class RealGitMultiWorktreeIntegration(unittest.TestCase):
|
||||
def setUp(self):
|
||||
import subprocess
|
||||
self.tmp = tempfile.mkdtemp(prefix="git-integration-")
|
||||
self.repo = os.path.join(self.tmp, "repo")
|
||||
os.makedirs(self.repo, exist_ok=True)
|
||||
subprocess.run(["git", "init"], cwd=self.repo, check=True, capture_output=True)
|
||||
subprocess.run(["git", "config", "user.name", "Test User"], cwd=self.repo, check=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=self.repo, check=True)
|
||||
with open(os.path.join(self.repo, "init.txt"), "w") as fh:
|
||||
fh.write("init")
|
||||
subprocess.run(["git", "add", "."], cwd=self.repo, check=True)
|
||||
subprocess.run(["git", "commit", "-m", "init"], cwd=self.repo, check=True)
|
||||
branch = "fix/issue-999-test"
|
||||
subprocess.run(["git", "branch", branch], cwd=self.repo, check=True)
|
||||
self.branches = os.path.join(self.repo, "branches")
|
||||
self.source = os.path.join(self.branches, "issue-999-test")
|
||||
subprocess.run(["git", "worktree", "add", self.source, branch], cwd=self.repo, check=True)
|
||||
self.dirty_path = os.path.join(self.source, "dirty.txt")
|
||||
with open(self.dirty_path, "w") as fh:
|
||||
fh.write("dirty-data")
|
||||
|
||||
def tearDown(self):
|
||||
shutil.rmtree(self.tmp, ignore_errors=True)
|
||||
|
||||
def test_prepare_recovery_worktree_detached_no_exit_128(self):
|
||||
import subprocess
|
||||
head_sha = subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=self.repo, text=True).strip()
|
||||
rec_wt = os.path.join(self.branches, "recovery-issue-999-test")
|
||||
res = dorec.prepare_recovery_worktree(
|
||||
canonical_repo_root=self.repo,
|
||||
recovery_worktree_path=rec_wt,
|
||||
branch_name="fix/issue-999-test",
|
||||
remote_head=head_sha,
|
||||
)
|
||||
self.assertTrue(res["success"], res.get("reasons"))
|
||||
self.assertTrue(os.path.isdir(rec_wt))
|
||||
|
||||
def test_real_lock_rebind_recovery_sanctioned(self):
|
||||
lock_dir = os.path.join(self.tmp, "locks")
|
||||
rec_wt = os.path.join(self.branches, "recovery-issue-999-test")
|
||||
os.makedirs(rec_wt, exist_ok=True)
|
||||
record = {
|
||||
"remote": "prgs",
|
||||
"org": "Example-Org",
|
||||
"repo": "Example-Repo",
|
||||
"issue_number": 999,
|
||||
"branch_name": "fix/issue-999-test",
|
||||
"worktree_path": rec_wt,
|
||||
"claimant": {"username": "author-user", "profile": "prgs-author"},
|
||||
}
|
||||
record_src = dict(record)
|
||||
record_src["worktree_path"] = self.source
|
||||
issue_lock_store.bind_session_lock(record_src, lock_dir=lock_dir)
|
||||
path = issue_lock_store.bind_session_lock(
|
||||
record,
|
||||
lock_dir=lock_dir,
|
||||
recovery_sanctioned=True,
|
||||
)
|
||||
self.assertTrue(os.path.isfile(path))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,907 @@
|
||||
"""Tests for the pre-restart drain proof and hard gate (#661).
|
||||
|
||||
Covers the acceptance criteria:
|
||||
|
||||
1. Restart apply without a proof fails closed.
|
||||
2. A successful drain produces a verifiable proof.
|
||||
3. An open unsafe mutation makes the proof fail (multi-session fixture).
|
||||
4. Pass / fail / expired verification paths.
|
||||
|
||||
Plus the security posture: forged/tampered proofs are rejected, break-glass is
|
||||
the only bypass and is never silent, a stale blast-radius fingerprint rejects a
|
||||
proof, and no per-process secret ever leaks into a serialized artifact.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
import drain_proof as dp
|
||||
import restart_coordinator as rc
|
||||
|
||||
|
||||
NOW = datetime(2026, 7, 24, 6, 0, 0, tzinfo=timezone.utc)
|
||||
SECRET = b"unit-test-drain-proof-secret-0123456789abcdef"
|
||||
|
||||
|
||||
def _live_pid() -> int:
|
||||
return os.getpid()
|
||||
|
||||
|
||||
def _clean_drain_state() -> dict:
|
||||
"""Every drain action succeeded, no sessions outstanding."""
|
||||
|
||||
return {
|
||||
"assignments_stopped": True,
|
||||
"checkpoints_complete": True,
|
||||
"handoffs_verified": True,
|
||||
"leases_handled": True,
|
||||
"acks": {}, # no other live sessions to acknowledge
|
||||
"ack_timeout_policy_applied": False,
|
||||
}
|
||||
|
||||
|
||||
def _safe_report() -> dict:
|
||||
"""Impact report with no other live work: a restart here is safe."""
|
||||
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
def _unsafe_mutation_report() -> dict:
|
||||
"""Multi-session report: a second session holds a live author mutation."""
|
||||
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-controller-1-req",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
{
|
||||
"session_id": "prgs-author-99",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
]
|
||||
leases = [
|
||||
{
|
||||
"lease_id": "lease-mut",
|
||||
"session_id": "prgs-author-99",
|
||||
"role": "author",
|
||||
"phase": "implementing",
|
||||
"work_kind": "issue",
|
||||
"work_number": 661,
|
||||
"worktree_path": "branches/issue-661",
|
||||
"freshness": {"freshness": "active"},
|
||||
}
|
||||
]
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": leases, "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
class BuildDrainProofTests(unittest.TestCase):
|
||||
def test_clean_drain_produces_verifiable_clean_proof(self):
|
||||
"""AC#2: a successful drain produces a verifiable proof."""
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
self.assertTrue(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [])
|
||||
self.assertEqual(
|
||||
{c.name for c in proof.checks}, set(dp.REQUIRED_CHECKS)
|
||||
)
|
||||
result = dp.verify_drain_proof(
|
||||
proof.as_dict(), now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertTrue(result.valid, result.reasons)
|
||||
self.assertFalse(result.expired)
|
||||
self.assertFalse(result.tampered)
|
||||
|
||||
def test_open_mutation_makes_proof_unclean(self):
|
||||
"""AC#3: an unsafe mutation still in flight fails the proof."""
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_NO_INFLIGHT_MUTATIONS, proof.failed_checks)
|
||||
# Leases-handled also fails: the report still shows a disruptive lease.
|
||||
self.assertIn(dp.CHECK_LEASES_HANDLED, proof.failed_checks)
|
||||
result = dp.verify_drain_proof(proof.as_dict(), now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
|
||||
def test_incomplete_inventory_fails_no_mutations_check(self):
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report={"inventory_complete": False},
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_NO_INFLIGHT_MUTATIONS, proof.failed_checks)
|
||||
|
||||
def test_missing_checkpoint_flag_fails_closed(self):
|
||||
state = _clean_drain_state()
|
||||
del state["checkpoints_complete"]
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_CHECKPOINTS_COMPLETE, proof.failed_checks)
|
||||
|
||||
def test_non_true_flags_fail_closed(self):
|
||||
"""A truthy-but-not-True value (e.g. the string 'yes') must not pass."""
|
||||
|
||||
state = _clean_drain_state()
|
||||
state["assignments_stopped"] = "yes"
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertIn(dp.CHECK_ASSIGNMENTS_STOPPED, proof.failed_checks)
|
||||
|
||||
def test_ack_timeout_policy_satisfies_ack_check(self):
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-author-99": "pending"}
|
||||
state["ack_timeout_policy_applied"] = True
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
names = {c.name: c.passed for c in proof.checks}
|
||||
self.assertTrue(names[dp.CHECK_ACKS_OR_TIMEOUT])
|
||||
|
||||
def test_outstanding_acks_without_timeout_fail(self):
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-author-99": "pending"}
|
||||
state["ack_timeout_policy_applied"] = False
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertIn(dp.CHECK_ACKS_OR_TIMEOUT, proof.failed_checks)
|
||||
|
||||
def test_all_acked_satisfies_ack_check(self):
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-author-99": "acked", "prgs-author-2": "acknowledged"}
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
names = {c.name: c.passed for c in proof.checks}
|
||||
self.assertTrue(names[dp.CHECK_ACKS_OR_TIMEOUT])
|
||||
|
||||
|
||||
class VerifyDrainProofTests(unittest.TestCase):
|
||||
def _clean_proof_dict(self) -> dict:
|
||||
return dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
|
||||
def test_missing_proof_is_invalid(self):
|
||||
result = dp.verify_drain_proof(None, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertIsNone(result.proof_id)
|
||||
|
||||
def test_expired_proof_is_invalid(self):
|
||||
"""AC#4: an expired proof fails verification."""
|
||||
|
||||
proof = self._clean_proof_dict()
|
||||
later = NOW + timedelta(seconds=dp.DEFAULT_PROOF_TTL_SECONDS + 1)
|
||||
result = dp.verify_drain_proof(proof, now=later, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.expired)
|
||||
|
||||
def test_proof_valid_just_before_expiry(self):
|
||||
proof = self._clean_proof_dict()
|
||||
almost = NOW + timedelta(seconds=dp.DEFAULT_PROOF_TTL_SECONDS - 1)
|
||||
result = dp.verify_drain_proof(proof, now=almost, secret=SECRET)
|
||||
self.assertTrue(result.valid, result.reasons)
|
||||
|
||||
def test_wrong_secret_rejected(self):
|
||||
"""A proof minted in a prior process (different secret) will not verify."""
|
||||
|
||||
proof = self._clean_proof_dict()
|
||||
result = dp.verify_drain_proof(proof, now=NOW, secret=b"other-secret")
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.tampered)
|
||||
|
||||
def test_flipping_clean_flag_is_detected(self):
|
||||
"""Forging clean=True on an unclean proof breaks the signature."""
|
||||
|
||||
unclean = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
self.assertFalse(unclean["clean"])
|
||||
unclean["clean"] = True # forge
|
||||
result = dp.verify_drain_proof(unclean, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.tampered)
|
||||
|
||||
def test_tampering_a_check_is_detected(self):
|
||||
unclean = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
for c in unclean["checks"]:
|
||||
if c["name"] == dp.CHECK_NO_INFLIGHT_MUTATIONS:
|
||||
c["passed"] = True # forge the failing check to pass
|
||||
result = dp.verify_drain_proof(unclean, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.tampered)
|
||||
|
||||
def test_missing_required_check_rejected(self):
|
||||
proof = self._clean_proof_dict()
|
||||
proof["checks"] = [
|
||||
c for c in proof["checks"] if c["name"] != dp.CHECK_HANDOFFS_OK
|
||||
]
|
||||
result = dp.verify_drain_proof(proof, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
|
||||
def test_stale_fingerprint_rejected(self):
|
||||
proof = self._clean_proof_dict()
|
||||
result = dp.verify_drain_proof(
|
||||
proof,
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint="deadbeef",
|
||||
)
|
||||
self.assertFalse(result.valid)
|
||||
|
||||
def test_matching_fingerprint_accepted(self):
|
||||
report = _safe_report()
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=report,
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
fp = dp.impact_fingerprint(report)
|
||||
result = dp.verify_drain_proof(
|
||||
proof, now=NOW, secret=SECRET, expected_impact_fingerprint=fp
|
||||
)
|
||||
self.assertTrue(result.valid, result.reasons)
|
||||
|
||||
|
||||
class GateApplyRestartTests(unittest.TestCase):
|
||||
def _clean_proof_dict(self) -> dict:
|
||||
return dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
|
||||
def test_apply_without_proof_denied(self):
|
||||
"""AC#1: restart apply without a proof fails closed + raises incident."""
|
||||
|
||||
decision = dp.gate_apply_restart(proof=None, now=NOW, secret=SECRET)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_DENY)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
self.assertEqual(
|
||||
decision.incident["kind"], "restart_drain_gate_denied"
|
||||
)
|
||||
|
||||
def test_apply_with_valid_proof_allowed(self):
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=self._clean_proof_dict(), now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertTrue(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_ALLOW)
|
||||
self.assertIsNone(decision.incident)
|
||||
|
||||
def test_apply_with_expired_proof_denied_with_incident(self):
|
||||
later = NOW + timedelta(seconds=dp.DEFAULT_PROOF_TTL_SECONDS + 5)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=self._clean_proof_dict(), now=later, secret=SECRET
|
||||
)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
|
||||
def test_apply_with_unclean_proof_denied(self):
|
||||
"""AC#3 at the gate: an unsafe-mutation proof is denied."""
|
||||
|
||||
unclean = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
decision = dp.gate_apply_restart(proof=unclean, now=NOW, secret=SECRET)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
|
||||
def test_break_glass_allows_without_proof_but_records_bypass(self):
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=None, now=NOW, secret=SECRET, break_glass=True
|
||||
)
|
||||
self.assertTrue(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_BREAK_GLASS)
|
||||
self.assertTrue(decision.break_glass)
|
||||
self.assertIsNone(decision.incident)
|
||||
self.assertTrue(decision.audit_record["break_glass"])
|
||||
|
||||
def test_denied_gate_carries_stale_fingerprint_reason(self):
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=self._clean_proof_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint="not-the-fingerprint",
|
||||
)
|
||||
self.assertFalse(decision.allow)
|
||||
|
||||
|
||||
class SecretHygieneTests(unittest.TestCase):
|
||||
def test_secret_never_serialized(self):
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
blob = dp._canonical(proof.as_dict())
|
||||
self.assertNotIn(SECRET.decode(), blob)
|
||||
# The signature is a hex digest, not the raw secret.
|
||||
self.assertNotIn(SECRET.hex(), blob)
|
||||
|
||||
def test_incident_descriptor_has_no_secret(self):
|
||||
decision = dp.gate_apply_restart(proof=None, now=NOW, secret=SECRET)
|
||||
blob = dp._canonical(decision.incident)
|
||||
self.assertNotIn(SECRET.decode(), blob)
|
||||
|
||||
|
||||
def _drained_report_with_live_sessions(count: int) -> dict:
|
||||
"""Report with ``count`` other live sessions but nothing in flight.
|
||||
|
||||
Every other checklist item passes against this report, so a failure
|
||||
isolates the acknowledgement check rather than tripping on mutations.
|
||||
"""
|
||||
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-controller-1-req",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
]
|
||||
for index in range(count):
|
||||
sessions.append(
|
||||
{
|
||||
"session_id": f"prgs-author-{index}",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
)
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
class AcknowledgementFailClosedTests(unittest.TestCase):
|
||||
"""Acknowledgement evidence must fail closed unless explicitly verified.
|
||||
|
||||
Regression cover for the reviewed fail-open on PR #882: an absent ``acks``
|
||||
key collapsed to ``{}`` and was read as "no other live sessions required to
|
||||
acknowledge", so a proof minted clean and the restart gate allowed while the
|
||||
impact report still showed other live sessions.
|
||||
"""
|
||||
|
||||
def _state(self, **overrides) -> dict:
|
||||
state = _clean_drain_state()
|
||||
state.pop("acks", None)
|
||||
state["ack_timeout_policy_applied"] = False
|
||||
state.update(overrides)
|
||||
return state
|
||||
|
||||
def _acks_check(self, proof) -> dp.DrainCheck:
|
||||
return next(c for c in proof.checks if c.name == dp.CHECK_ACKS_OR_TIMEOUT)
|
||||
|
||||
def _build(self, report: dict, state: dict):
|
||||
return dp.build_drain_proof(
|
||||
impact_report=report, drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
|
||||
def assertAcksFailClosed(self, report: dict, state: dict) -> None:
|
||||
proof = self._build(report, state)
|
||||
self.assertFalse(self._acks_check(proof).passed)
|
||||
self.assertIn(dp.CHECK_ACKS_OR_TIMEOUT, proof.failed_checks)
|
||||
self.assertFalse(proof.clean)
|
||||
|
||||
# --- missing / null / empty / malformed ------------------------------
|
||||
|
||||
def test_missing_acks_key_with_live_sessions_fails_closed(self):
|
||||
"""The exact reviewed defect: absent key, three other live sessions."""
|
||||
report = _drained_report_with_live_sessions(3)
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 3)
|
||||
state = self._state()
|
||||
self.assertNotIn("acks", state)
|
||||
proof = self._build(report, state)
|
||||
check = self._acks_check(proof)
|
||||
self.assertFalse(check.passed)
|
||||
self.assertNotIn("no other live sessions", check.detail)
|
||||
self.assertIn("fail closed", check.detail)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [dp.CHECK_ACKS_OR_TIMEOUT])
|
||||
|
||||
def test_none_acks_with_live_sessions_fails_closed(self):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(2), self._state(acks=None)
|
||||
)
|
||||
|
||||
def test_empty_acks_with_live_sessions_fails_closed(self):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(1), self._state(acks={})
|
||||
)
|
||||
|
||||
def test_malformed_acks_fail_closed(self):
|
||||
for malformed in ([], "ack", 7, ("ack",), True):
|
||||
with self.subTest(malformed=malformed):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(1),
|
||||
self._state(acks=malformed),
|
||||
)
|
||||
|
||||
# --- stale / unproven values -----------------------------------------
|
||||
|
||||
def test_stale_or_unproven_ack_values_fail_closed(self):
|
||||
for value in ("pending", "stale", "unknown", "", None, True, 1, NOW):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(1),
|
||||
self._state(acks={"prgs-author-0": value}),
|
||||
)
|
||||
|
||||
def test_partial_coverage_fails_closed(self):
|
||||
"""Fewer acknowledgements than the report's live-session count."""
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(3),
|
||||
self._state(acks={"prgs-author-0": "ack"}),
|
||||
)
|
||||
|
||||
def test_one_unacked_entry_among_many_fails_closed(self):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(2),
|
||||
self._state(acks={"prgs-author-0": "ack", "prgs-author-1": "pending"}),
|
||||
)
|
||||
|
||||
def test_unproven_live_session_count_fails_closed(self):
|
||||
"""A missing or malformed count cannot prove nobody had to acknowledge."""
|
||||
malformed_counts = (
|
||||
None,
|
||||
{},
|
||||
{"sessions_live_other": None},
|
||||
{"sessions_live_other": "3"},
|
||||
{"sessions_live_other": -1},
|
||||
{"sessions_live_other": True},
|
||||
)
|
||||
for counts in malformed_counts:
|
||||
with self.subTest(counts=counts):
|
||||
report = _drained_report_with_live_sessions(0)
|
||||
if counts is None:
|
||||
report.pop("counts", None)
|
||||
else:
|
||||
report["counts"] = counts
|
||||
self.assertAcksFailClosed(report, self._state())
|
||||
|
||||
# --- valid evidence still passes -------------------------------------
|
||||
|
||||
def test_complete_valid_acks_pass(self):
|
||||
report = _drained_report_with_live_sessions(2)
|
||||
state = self._state(
|
||||
acks={"prgs-author-0": "ack", "prgs-author-1": "acknowledged"}
|
||||
)
|
||||
proof = self._build(report, state)
|
||||
self.assertTrue(self._acks_check(proof).passed)
|
||||
self.assertTrue(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [])
|
||||
|
||||
def test_no_other_live_sessions_still_passes(self):
|
||||
"""Intended behavior retained: zero live sessions needs no acks."""
|
||||
report = _drained_report_with_live_sessions(0)
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 0)
|
||||
proof = self._build(report, self._state())
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("sessions_live_other=0", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
# --- timeout policy cannot become a second fail-open ------------------
|
||||
|
||||
def test_unproven_timeout_policy_cannot_open_the_gate(self):
|
||||
for value in (None, "true", "yes", 1, "True", [], {}):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(2),
|
||||
self._state(ack_timeout_policy_applied=value),
|
||||
)
|
||||
|
||||
def test_explicit_timeout_policy_permits(self):
|
||||
proof = self._build(
|
||||
_drained_report_with_live_sessions(2),
|
||||
self._state(ack_timeout_policy_applied=True),
|
||||
)
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("timeout policy", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
# --- the gate itself must deny ---------------------------------------
|
||||
|
||||
def test_failed_ack_check_denies_the_restart_gate(self):
|
||||
report = _drained_report_with_live_sessions(3)
|
||||
proof = self._build(report, self._state())
|
||||
self.assertFalse(proof.clean)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_DENY)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
|
||||
def test_unclean_ack_proof_fails_verification(self):
|
||||
report = _drained_report_with_live_sessions(3)
|
||||
proof = self._build(report, self._state())
|
||||
result = dp.verify_drain_proof(
|
||||
proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertFalse(result.clean)
|
||||
|
||||
|
||||
def _identity_report(*, requester: str, others: tuple[str, ...]) -> dict:
|
||||
"""Report with explicitly named requester and other live sessions.
|
||||
|
||||
Unlike :func:`_drained_report_with_live_sessions`, the session ids are
|
||||
chosen by the caller so a test can supply acknowledgements for the *wrong*
|
||||
identities while keeping the count correct.
|
||||
"""
|
||||
|
||||
sessions = [
|
||||
{
|
||||
"session_id": requester,
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
]
|
||||
for session_id in others:
|
||||
sessions.append(
|
||||
{
|
||||
"session_id": session_id,
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
)
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id=requester,
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
class AcknowledgementIdentityBindingTests(unittest.TestCase):
|
||||
"""Acknowledgement coverage must be bound to session identity, not counted.
|
||||
|
||||
Regression cover for the second reviewed fail-open on PR #882 (review 582,
|
||||
blocker B1): coverage compared ``acked_count`` against
|
||||
``counts.sessions_live_other``, so acknowledgements supplied for the
|
||||
requesting session and for ids that do not exist satisfied the obligations
|
||||
of the live sessions that never answered. The required identities are
|
||||
carried by the report itself — ``ack_state`` keys and ``affected_sessions``
|
||||
filtered on ``live and not is_requester`` — and only an acknowledgement
|
||||
keyed by one of those ids may count for it.
|
||||
"""
|
||||
|
||||
def _state(self, **overrides) -> dict:
|
||||
state = _clean_drain_state()
|
||||
state.pop("acks", None)
|
||||
state["ack_timeout_policy_applied"] = False
|
||||
state.update(overrides)
|
||||
return state
|
||||
|
||||
def _acks_check(self, proof) -> dp.DrainCheck:
|
||||
return next(c for c in proof.checks if c.name == dp.CHECK_ACKS_OR_TIMEOUT)
|
||||
|
||||
def _build(self, report: dict, state: dict):
|
||||
return dp.build_drain_proof(
|
||||
impact_report=report, drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
|
||||
def assertAcksFailClosed(self, report: dict, state: dict) -> dp.DrainCheck:
|
||||
"""Failure must propagate through the check, the proof, and the gate."""
|
||||
|
||||
proof = self._build(report, state)
|
||||
check = self._acks_check(proof)
|
||||
self.assertFalse(check.passed)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_ACKS_OR_TIMEOUT, proof.failed_checks)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertEqual(decision.verdict, dp.GATE_DENY)
|
||||
self.assertFalse(decision.allow)
|
||||
return check
|
||||
|
||||
# --- the reviewer's exact reproduction --------------------------------
|
||||
|
||||
def test_requester_plus_unknown_id_cannot_satisfy_two_live_sessions(self):
|
||||
"""Review 582 B1 verbatim: requester + a nonexistent session.
|
||||
|
||||
``sessions_live_other=2`` with ``ack_state`` naming ``other-0`` and
|
||||
``other-1``; the drain state supplies an acknowledgement from the
|
||||
requesting session itself and from a session that does not exist. The
|
||||
count matches, the identities do not.
|
||||
"""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 2)
|
||||
self.assertEqual(
|
||||
report["ack_state"], {"other-0": "pending", "other-1": "pending"}
|
||||
)
|
||||
state = self._state(acks={"req": "ack", "totally-bogus-session": "ack"})
|
||||
check = self.assertAcksFailClosed(report, state)
|
||||
self.assertIn("other-0", check.detail)
|
||||
self.assertIn("other-1", check.detail)
|
||||
self.assertIn("fail closed", check.detail)
|
||||
|
||||
# --- wrong / unknown / requester identities ---------------------------
|
||||
|
||||
def test_sufficient_count_of_wrong_ids_fails_closed(self):
|
||||
"""Right cardinality, wrong identities: two acks, neither required."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
state = self._state(acks={"ghost-a": "ack", "ghost-b": "ack"})
|
||||
check = self.assertAcksFailClosed(report, state)
|
||||
self.assertIn("do not count", check.detail)
|
||||
|
||||
def test_more_acks_than_required_still_fails_on_wrong_ids(self):
|
||||
"""Coverage cannot be bought with volume: five acks, none required."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
state = self._state(acks={f"ghost-{i}": "acknowledged" for i in range(5)})
|
||||
self.assertAcksFailClosed(report, state)
|
||||
|
||||
def test_partial_identity_match_fails_closed(self):
|
||||
"""One required id acknowledged, the rest padded with unknown ids."""
|
||||
|
||||
report = _identity_report(
|
||||
requester="req", others=("other-0", "other-1", "other-2")
|
||||
)
|
||||
state = self._state(
|
||||
acks={"other-0": "ack", "ghost-1": "ack", "ghost-2": "ack"}
|
||||
)
|
||||
check = self.assertAcksFailClosed(report, state)
|
||||
self.assertIn("other-1", check.detail)
|
||||
self.assertIn("other-2", check.detail)
|
||||
|
||||
def test_requester_ack_never_satisfies_another_sessions_obligation(self):
|
||||
"""The requester is excluded from the required set and stays excluded."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
requester_rows = [s for s in report["affected_sessions"] if s["is_requester"]]
|
||||
self.assertEqual([s["session_id"] for s in requester_rows], ["req"])
|
||||
self.assertNotIn("req", report["ack_state"])
|
||||
check = self.assertAcksFailClosed(report, self._state(acks={"req": "ack"}))
|
||||
self.assertIn("other-0", check.detail)
|
||||
|
||||
def test_fabricated_ids_do_not_count_toward_coverage(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
for bogus in ("", " ", "other-0 extra", "OTHER-0", "other-01", "0"):
|
||||
with self.subTest(bogus=bogus):
|
||||
self.assertAcksFailClosed(report, self._state(acks={bogus: "ack"}))
|
||||
|
||||
# --- per-session state must be explicitly valid ------------------------
|
||||
|
||||
def test_unproven_per_session_states_fail_closed(self):
|
||||
"""A required id present but not explicitly acknowledged fails closed."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
for value in ("pending", "stale", "unknown", "", None, True, 1, NOW):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
report,
|
||||
self._state(acks={"other-0": "ack", "other-1": value}),
|
||||
)
|
||||
|
||||
def test_report_ack_state_placeholder_is_never_read_as_an_ack(self):
|
||||
"""``ack_state`` values are the report's own placeholders, not evidence."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = {"other-0": "ack"}
|
||||
self.assertAcksFailClosed(report, self._state())
|
||||
|
||||
# --- missing / malformed / contradictory identity evidence -------------
|
||||
|
||||
def test_missing_identity_evidence_fails_closed(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
report.pop("affected_sessions", None)
|
||||
check = self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
self.assertIn("no session-identity evidence", check.detail)
|
||||
|
||||
def test_malformed_ack_state_fails_closed(self):
|
||||
for malformed in ([], "other-0", 7, None, ("other-0",)):
|
||||
with self.subTest(malformed=malformed):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = malformed
|
||||
self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack"})
|
||||
)
|
||||
|
||||
def test_non_string_ack_state_key_fails_closed(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = {7: "pending"}
|
||||
self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
|
||||
def test_malformed_affected_sessions_fails_closed(self):
|
||||
for malformed in ("sessions", 7, {"session_id": "other-0"}, [None], [7]):
|
||||
with self.subTest(malformed=malformed):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
report["affected_sessions"] = malformed
|
||||
self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack"})
|
||||
)
|
||||
|
||||
def test_affected_sessions_without_explicit_booleans_fails_closed(self):
|
||||
"""``live``/``is_requester`` must be real booleans, never inferred."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
for row in report["affected_sessions"]:
|
||||
if row["session_id"] == "other-0":
|
||||
row["is_requester"] = "false"
|
||||
self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
|
||||
def test_affected_sessions_missing_live_flag_fails_closed(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
for row in report["affected_sessions"]:
|
||||
row.pop("live", None)
|
||||
self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
|
||||
def test_contradictory_ack_state_and_affected_sessions_fails_closed(self):
|
||||
"""Both views present and disagreeing is unresolvable, not a tie-break."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
report["ack_state"] = {"other-0": "pending", "other-9": "pending"}
|
||||
check = self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack", "other-9": "ack"})
|
||||
)
|
||||
self.assertIn("contradicts itself", check.detail)
|
||||
|
||||
def test_identity_count_mismatch_fails_closed(self):
|
||||
"""Identity evidence that cannot be reconciled with the count denies."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
report["counts"] = dict(report["counts"], sessions_live_other=1)
|
||||
check = self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack", "other-1": "ack"})
|
||||
)
|
||||
self.assertIn("cannot be reconciled", check.detail)
|
||||
|
||||
def test_broken_identity_evidence_outranks_timeout_policy(self):
|
||||
"""The sanctioned timeout path cannot paper over an unreadable report."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = "not-a-mapping"
|
||||
self.assertAcksFailClosed(report, self._state(ack_timeout_policy_applied=True))
|
||||
|
||||
# --- legitimate success is preserved -----------------------------------
|
||||
|
||||
def test_every_required_session_acknowledged_passes(self):
|
||||
report = _identity_report(
|
||||
requester="req", others=("other-0", "other-1", "other-2")
|
||||
)
|
||||
state = self._state(
|
||||
acks={
|
||||
"other-0": "ack",
|
||||
"other-1": "acked",
|
||||
"other-2": "acknowledged",
|
||||
}
|
||||
)
|
||||
proof = self._build(report, state)
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertTrue(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [])
|
||||
self.assertIn("acknowledged by identity", check.detail)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertEqual(decision.verdict, dp.GATE_ALLOW)
|
||||
self.assertTrue(decision.allow)
|
||||
|
||||
def test_required_session_ack_tolerates_surrounding_whitespace(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
proof = self._build(report, self._state(acks={" other-0 ": " ACK "}))
|
||||
self.assertTrue(self._acks_check(proof).passed)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
def test_no_other_live_sessions_still_passes_with_identity_evidence(self):
|
||||
report = _identity_report(requester="req", others=())
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 0)
|
||||
self.assertEqual(report["ack_state"], {})
|
||||
proof = self._build(report, self._state())
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("sessions_live_other=0", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
def test_explicit_timeout_policy_retains_intended_behavior(self):
|
||||
"""Valid, correctly typed timeout evidence still permits the check."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
proof = self._build(report, self._state(ack_timeout_policy_applied=True))
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("timeout policy", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
def test_timeout_policy_still_strictly_typed_under_identity_binding(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
for value in (None, "true", "True", 1, [], {}):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
report, self._state(ack_timeout_policy_applied=value)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,337 @@
|
||||
"""Native exposure of the fleet inventory capability (#949).
|
||||
|
||||
Covers the parts the pure classifier cannot: that the tool is registered and
|
||||
documented, that it is gated on ``gitea.read`` rather than a role, that it
|
||||
accepts no caller-supplied evidence, that it fails closed on an unreadable
|
||||
registry, and that the workflow documentation explains gate consumption.
|
||||
"""
|
||||
|
||||
import inspect
|
||||
import os
|
||||
import pathlib
|
||||
import unittest
|
||||
from unittest import mock
|
||||
|
||||
import gitea_mcp_server
|
||||
import mcp_fleet_inventory as mfi
|
||||
import task_capability_map
|
||||
|
||||
|
||||
REPO_ROOT = pathlib.Path(__file__).resolve().parent.parent
|
||||
TOOL_NAME = "gitea_assess_fleet_inventory"
|
||||
|
||||
|
||||
class TestCapabilityMapExposure(unittest.TestCase):
|
||||
"""AC14: the invariant is provable through sanctioned native calls."""
|
||||
|
||||
def test_task_resolves_to_a_read_permission(self):
|
||||
self.assertEqual(
|
||||
task_capability_map.required_permission("assess_fleet_inventory"),
|
||||
"gitea.read",
|
||||
)
|
||||
self.assertEqual(
|
||||
task_capability_map.required_permission(TOOL_NAME), "gitea.read"
|
||||
)
|
||||
|
||||
def test_task_and_tool_keys_agree(self):
|
||||
self.assertEqual(
|
||||
task_capability_map.TASK_CAPABILITY_MAP["assess_fleet_inventory"],
|
||||
task_capability_map.TASK_CAPABILITY_MAP[TOOL_NAME],
|
||||
)
|
||||
|
||||
def test_task_is_not_role_exclusive(self):
|
||||
"""Controller and reconciler both need it; so does any read-only gate."""
|
||||
self.assertNotIn(
|
||||
"assess_fleet_inventory", task_capability_map.ROLE_EXCLUSIVE_TASKS
|
||||
)
|
||||
self.assertNotIn(TOOL_NAME, task_capability_map.ROLE_EXCLUSIVE_TASKS)
|
||||
|
||||
def test_task_is_not_registered_as_an_issue_mutation(self):
|
||||
self.assertNotIn(TOOL_NAME, task_capability_map.ISSUE_MUTATION_TOOL_TASKS)
|
||||
|
||||
def test_unknown_task_still_fails_closed(self):
|
||||
with self.assertRaises(KeyError):
|
||||
task_capability_map.required_permission("assess_fleet_inventory_typo")
|
||||
|
||||
|
||||
class TestToolRegistration(unittest.TestCase):
|
||||
def test_tool_is_registered_on_the_server(self):
|
||||
self.assertTrue(hasattr(gitea_mcp_server, TOOL_NAME))
|
||||
|
||||
def test_tool_accepts_no_caller_supplied_evidence(self):
|
||||
"""The defect #949 names: evidence parameters the caller controls."""
|
||||
signature = inspect.signature(gitea_mcp_server.gitea_assess_fleet_inventory)
|
||||
self.assertEqual(sorted(signature.parameters), ["host", "org", "remote", "repo"])
|
||||
for forbidden in ("process", "probe_result", "registered_tools", "processes"):
|
||||
self.assertNotIn(forbidden, signature.parameters)
|
||||
|
||||
def test_tool_is_documented_in_the_canonical_inventory(self):
|
||||
doc = (REPO_ROOT / "docs" / "mcp-tool-inventory.md").read_text()
|
||||
self.assertIn(f"`{TOOL_NAME}`", doc)
|
||||
|
||||
def test_docstring_states_the_read_only_guarantee(self):
|
||||
doc = (gitea_mcp_server.gitea_assess_fleet_inventory.__doc__ or "").lower()
|
||||
self.assertIn("read-only", doc)
|
||||
self.assertIn("fails closed", doc)
|
||||
|
||||
|
||||
class TestNamespaceResolution(unittest.TestCase):
|
||||
"""Every configured profile must resolve to its real fleet namespace.
|
||||
|
||||
``role_namespace_gate.infer_mcp_namespace`` recognises only author and
|
||||
reviewer and echoes the profile name for the rest, which would label the
|
||||
controller, merger, and reconciler members with namespaces that do not
|
||||
exist — and then flag all three as role mismatches.
|
||||
"""
|
||||
|
||||
def test_every_expected_profile_maps_to_its_namespace(self):
|
||||
for entry in mfi.EXPECTED_PRGS_FLEET:
|
||||
self.assertEqual(
|
||||
mfi.namespace_for_profile(entry["profile"]),
|
||||
entry["namespace"],
|
||||
entry["profile"],
|
||||
)
|
||||
|
||||
def test_server_helper_agrees_with_the_roster(self):
|
||||
for entry in mfi.EXPECTED_PRGS_FLEET:
|
||||
self.assertEqual(
|
||||
gitea_mcp_server._fleet_namespace_for_profile(entry["profile"]),
|
||||
entry["namespace"],
|
||||
entry["profile"],
|
||||
)
|
||||
|
||||
def test_unknown_profile_falls_back_to_the_supplied_default(self):
|
||||
self.assertEqual(
|
||||
mfi.namespace_for_profile("prgs-shadow", default="gitea-shadow"),
|
||||
"gitea-shadow",
|
||||
)
|
||||
|
||||
def test_unknown_profile_without_a_default_returns_the_profile(self):
|
||||
self.assertEqual(mfi.namespace_for_profile("prgs-shadow"), "prgs-shadow")
|
||||
|
||||
def test_registered_row_uses_the_roster_namespace(self):
|
||||
fake_db = mock.Mock()
|
||||
with mock.patch.object(
|
||||
gitea_mcp_server, "_control_plane_db_or_error", return_value=(fake_db, [])
|
||||
), mock.patch.object(
|
||||
gitea_mcp_server, "get_profile", return_value={"allowed_operations": []}
|
||||
), mock.patch.object(
|
||||
gitea_mcp_server.gitea_config,
|
||||
"selected_profile_name",
|
||||
return_value="prgs-reconciler",
|
||||
):
|
||||
record = gitea_mcp_server._register_fleet_runtime(transport="stdio")
|
||||
self.assertEqual(record["namespace"], "gitea-reconciler")
|
||||
|
||||
|
||||
class TestToolBehavior(unittest.TestCase):
|
||||
"""The tool wires the registry and the process scan into the classifier."""
|
||||
|
||||
@staticmethod
|
||||
def _healthy_rows():
|
||||
return [
|
||||
{
|
||||
"runtime_id": f"{entry['namespace']}:{43000 + index}:boot",
|
||||
"namespace": entry["namespace"],
|
||||
"profile": entry["profile"],
|
||||
"role": entry["role"],
|
||||
"remote": "prgs",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"repository_root": "/checkout/Gitea-Tools",
|
||||
"pid": 43000 + index,
|
||||
"cohort_id": "ppid:40990",
|
||||
"cohort_source": "parent_process",
|
||||
"client_provenance": "client_managed",
|
||||
"boot_id": "boot",
|
||||
"startup_head": "82d71b77028a7abd4f8ab4a4e4d89658a187f73d",
|
||||
"daemon_start_head": "82d71b77028a7abd4f8ab4a4e4d89658a187f73d",
|
||||
"transport": "stdio",
|
||||
"registered_at": "2026-07-28T02:00:00Z",
|
||||
"last_heartbeat_at": "2026-07-28T02:00:00Z",
|
||||
"status": "running",
|
||||
}
|
||||
for index, entry in enumerate(mfi.EXPECTED_PRGS_FLEET)
|
||||
]
|
||||
|
||||
def _call(self, rows, *, scan=None, db=None, db_errors=None):
|
||||
fake_db = db
|
||||
if fake_db is None and db_errors is None:
|
||||
fake_db = mock.Mock()
|
||||
fake_db.list_mcp_server_runtimes.return_value = rows
|
||||
scan = scan or {
|
||||
"available": True,
|
||||
"processes": [
|
||||
{"pid": r["pid"], "started_at": "2026-07-28T01:00:00Z"} for r in rows
|
||||
],
|
||||
"reason": None,
|
||||
}
|
||||
with mock.patch.object(
|
||||
gitea_mcp_server, "_profile_operation_gate", return_value=[]
|
||||
), mock.patch.object(
|
||||
gitea_mcp_server,
|
||||
"_control_plane_db_or_error",
|
||||
return_value=(fake_db, db_errors or []),
|
||||
), mock.patch.object(
|
||||
gitea_mcp_server, "_active_profile_name", return_value="prgs-controller"
|
||||
), mock.patch.object(
|
||||
mfi, "scan_mcp_server_processes", return_value=scan
|
||||
), mock.patch.object(
|
||||
mfi, "probe_pid_alive", return_value=True
|
||||
):
|
||||
return gitea_mcp_server.gitea_assess_fleet_inventory(
|
||||
remote="prgs", org="Scaled-Tech-Consulting", repo="Gitea-Tools"
|
||||
)
|
||||
|
||||
def test_healthy_fleet_satisfies_the_gate_through_the_tool(self):
|
||||
result = self._call(self._healthy_rows())
|
||||
self.assertTrue(result["inventory_complete"])
|
||||
self.assertTrue(result["mutation_gate_satisfied"])
|
||||
self.assertEqual(result["running_member_count"], 5)
|
||||
self.assertEqual(result["mutations_performed"], [])
|
||||
|
||||
def test_tool_reports_the_expected_repository_binding(self):
|
||||
result = self._call(self._healthy_rows())
|
||||
self.assertEqual(
|
||||
result["expected_repository_binding"],
|
||||
{"remote": "prgs", "org": "Scaled-Tech-Consulting", "repo": "Gitea-Tools"},
|
||||
)
|
||||
|
||||
def test_tool_reads_only_running_rows(self):
|
||||
rows = self._healthy_rows()
|
||||
fake_db = mock.Mock()
|
||||
fake_db.list_mcp_server_runtimes.return_value = rows
|
||||
self._call(rows, db=fake_db)
|
||||
fake_db.list_mcp_server_runtimes.assert_called_once_with(statuses=("running",))
|
||||
|
||||
def test_tool_never_writes_to_the_registry(self):
|
||||
rows = self._healthy_rows()
|
||||
fake_db = mock.Mock()
|
||||
fake_db.list_mcp_server_runtimes.return_value = rows
|
||||
self._call(rows, db=fake_db)
|
||||
fake_db.register_mcp_server_runtime.assert_not_called()
|
||||
fake_db.heartbeat_mcp_server_runtime.assert_not_called()
|
||||
fake_db.mark_mcp_server_runtime_stopped.assert_not_called()
|
||||
|
||||
def test_unavailable_control_plane_fails_closed(self):
|
||||
result = self._call([], db_errors=["control-plane DB substrate unavailable"])
|
||||
self.assertFalse(result["inventory_complete"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertFalse(result["registry"]["available"])
|
||||
self.assertIn("unavailable", result["blocked_reason"])
|
||||
|
||||
def test_registry_read_failure_fails_closed(self):
|
||||
rows = self._healthy_rows()
|
||||
fake_db = mock.Mock()
|
||||
fake_db.list_mcp_server_runtimes.side_effect = RuntimeError("db locked")
|
||||
result = self._call(rows, db=fake_db)
|
||||
self.assertFalse(result["inventory_complete"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertIn("could not be read", result["registry"]["error"])
|
||||
|
||||
def test_missing_read_permission_blocks_without_touching_the_registry(self):
|
||||
fake_db = mock.Mock()
|
||||
with mock.patch.object(
|
||||
gitea_mcp_server,
|
||||
"_profile_operation_gate",
|
||||
return_value=["profile may not read"],
|
||||
), mock.patch.object(
|
||||
gitea_mcp_server, "_control_plane_db_or_error", return_value=(fake_db, [])
|
||||
):
|
||||
result = gitea_mcp_server.gitea_assess_fleet_inventory(remote="prgs")
|
||||
self.assertFalse(result["success"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertIn("permission_report", result)
|
||||
self.assertEqual(result["mutations_performed"], [])
|
||||
fake_db.list_mcp_server_runtimes.assert_not_called()
|
||||
|
||||
def test_answering_namespace_is_reported(self):
|
||||
result = self._call(self._healthy_rows())
|
||||
self.assertEqual(result["answering_namespace"], "gitea-controller")
|
||||
|
||||
def test_summary_is_present(self):
|
||||
result = self._call(self._healthy_rows())
|
||||
self.assertIn("5 of 5", result["summary"])
|
||||
|
||||
|
||||
class TestStartupRegistration(unittest.TestCase):
|
||||
"""The registry row is written by the process it describes."""
|
||||
|
||||
def test_registration_helper_exists_on_the_entrypoint_module(self):
|
||||
self.assertTrue(hasattr(gitea_mcp_server, "_register_fleet_runtime"))
|
||||
|
||||
def test_registration_writes_one_row_for_this_process(self):
|
||||
fake_db = mock.Mock()
|
||||
with mock.patch.object(
|
||||
gitea_mcp_server, "_control_plane_db_or_error", return_value=(fake_db, [])
|
||||
), mock.patch.object(
|
||||
gitea_mcp_server, "get_profile", return_value={"allowed_operations": []}
|
||||
):
|
||||
record = gitea_mcp_server._register_fleet_runtime(transport="stdio")
|
||||
self.assertIsNotNone(record)
|
||||
fake_db.register_mcp_server_runtime.assert_called_once()
|
||||
written = fake_db.register_mcp_server_runtime.call_args.args[0]
|
||||
self.assertEqual(written["pid"], os.getpid())
|
||||
self.assertEqual(written["transport"], "stdio")
|
||||
|
||||
def test_registration_failure_never_blocks_startup(self):
|
||||
with mock.patch.object(
|
||||
gitea_mcp_server,
|
||||
"_control_plane_db_or_error",
|
||||
side_effect=RuntimeError("boom"),
|
||||
):
|
||||
self.assertIsNone(gitea_mcp_server._register_fleet_runtime())
|
||||
|
||||
def test_registration_is_skipped_when_the_control_plane_is_unavailable(self):
|
||||
with mock.patch.object(
|
||||
gitea_mcp_server,
|
||||
"_control_plane_db_or_error",
|
||||
return_value=(None, ["unavailable"]),
|
||||
):
|
||||
self.assertIsNone(gitea_mcp_server._register_fleet_runtime())
|
||||
|
||||
def test_entrypoint_registers_after_binding_native_transport(self):
|
||||
"""Order matters: only a transport-bound process may claim a row."""
|
||||
source = (REPO_ROOT / "gitea_mcp_server.py").read_text()
|
||||
bind_at = source.index('bind_native_mcp_transport(transport="stdio")')
|
||||
register_at = source.index('_register_fleet_runtime(transport="stdio")')
|
||||
run_at = source.index('mcp.run(transport="stdio")')
|
||||
self.assertLess(bind_at, register_at)
|
||||
self.assertLess(register_at, run_at)
|
||||
|
||||
|
||||
class TestWorkflowDocumentation(unittest.TestCase):
|
||||
"""AC13: documentation explains how the gates consume the result."""
|
||||
|
||||
def setUp(self):
|
||||
self.doc = (REPO_ROOT / "docs" / "mcp-fleet-inventory.md").read_text()
|
||||
self.skill = (
|
||||
REPO_ROOT / "skills" / "llm-project-workflow" / "SKILL.md"
|
||||
).read_text()
|
||||
|
||||
def test_dedicated_document_exists(self):
|
||||
self.assertIn("# Authoritative MCP fleet inventory", self.doc)
|
||||
|
||||
def test_document_names_both_consuming_namespaces(self):
|
||||
self.assertIn("gitea-controller", self.doc)
|
||||
self.assertIn("gitea-reconciler", self.doc)
|
||||
|
||||
def test_document_explains_gate_consumption(self):
|
||||
self.assertIn("mutation_gate_satisfied", self.doc)
|
||||
self.assertIn("blocked_reason", self.doc)
|
||||
|
||||
def test_document_states_the_non_inferences(self):
|
||||
self.assertIn("Configuration is not existence", self.doc)
|
||||
self.assertIn("Matching revisions are not a cohort", self.doc)
|
||||
|
||||
def test_document_preserves_the_neighbouring_issue_boundaries(self):
|
||||
for issue in ("#950", "#951", "#952"):
|
||||
self.assertIn(issue, self.doc)
|
||||
|
||||
def test_canonical_workflow_skill_links_the_document(self):
|
||||
self.assertIn("docs/mcp-fleet-inventory.md", self.skill)
|
||||
self.assertIn(TOOL_NAME, self.skill)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,199 @@
|
||||
"""Regression tests for #628 building blocks (child scope only).
|
||||
|
||||
Honest scope: unit coverage of pre-existing APIs used by umbrella #628.
|
||||
This module does **not** implement or verify all 21 umbrella acceptance
|
||||
criteria, automatic handoff store/retrieve, multi-worker product wiring,
|
||||
or end-to-end orchestration.
|
||||
|
||||
Covered building blocks:
|
||||
- CTH format / parse / assess (`format_cth_body`, `parse_cth_comment`,
|
||||
`assess_cth_comment`)
|
||||
- Exclusive-ownership skip classification (`classify_skip` with
|
||||
OWNERSHIP_FOREIGN vs OWNERSHIP_OWN)
|
||||
- Durable dependency edges (`upsert_dependency_edge` / list) and skip
|
||||
when dependency_unmet
|
||||
- Edge state transition UNMET -> MET
|
||||
|
||||
Parent umbrella remains #628; this slice is a scoped child issue only.
|
||||
"""
|
||||
|
||||
import unittest
|
||||
from unittest.mock import MagicMock, patch
|
||||
import os
|
||||
import json
|
||||
import tempfile
|
||||
|
||||
from canonical_thread_handoff import (
|
||||
format_cth_body,
|
||||
parse_cth_comment,
|
||||
assess_cth_comment,
|
||||
)
|
||||
import dependency_graph
|
||||
from control_plane_db import ControlPlaneDB
|
||||
from allocator_service import (
|
||||
WorkCandidate,
|
||||
classify_skip,
|
||||
ROLE_AUTHOR,
|
||||
ROLE_REVIEWER,
|
||||
ROLE_MERGER,
|
||||
ROLE_RECONCILER,
|
||||
OWNERSHIP_OWN,
|
||||
OWNERSHIP_FOREIGN,
|
||||
)
|
||||
|
||||
|
||||
class TestIssue628Orchestration(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.db_path = os.path.join(self._tmp.name, "cp.sqlite3")
|
||||
self.db = ControlPlaneDB(self.db_path)
|
||||
|
||||
def tearDown(self):
|
||||
self._tmp.cleanup()
|
||||
|
||||
def test_canonical_handoff_serialization_and_retrieval(self):
|
||||
"""Building block: format/parse/assess a CTH body (not full AC1/AC2 product path)."""
|
||||
handoff = format_cth_body(
|
||||
cth_type="Author Handoff",
|
||||
status="completed",
|
||||
next_owner="reviewer",
|
||||
current_blocker="none",
|
||||
decision="Implementation complete, tests passing",
|
||||
proof="pytest tests/test_issue_628_orchestration.py passed",
|
||||
next_action="Review PR and run reviewer pre-flight",
|
||||
ready_to_paste_prompt="Review PR for child issue #878 (parent #628)",
|
||||
)
|
||||
self.assertIn("CTH: Author Handoff", handoff)
|
||||
|
||||
parsed = parse_cth_comment(handoff)
|
||||
self.assertIsNotNone(parsed)
|
||||
self.assertEqual(parsed["cth_type"], "Author Handoff")
|
||||
|
||||
assessment = assess_cth_comment(handoff)
|
||||
self.assertFalse(assessment["block"])
|
||||
|
||||
def test_exclusive_task_unit_single_owner(self):
|
||||
"""Building block: classify_skip foreign vs own ownership."""
|
||||
candidate = WorkCandidate(
|
||||
kind="issue",
|
||||
number=878,
|
||||
title="Child #878 ownership classify candidate",
|
||||
state="open",
|
||||
labels=["status:in-progress"],
|
||||
blocked=False,
|
||||
dependency_unmet=False,
|
||||
)
|
||||
# Foreign ownership MUST be skipped
|
||||
skip_foreign = classify_skip(
|
||||
c=candidate,
|
||||
role=ROLE_AUTHOR,
|
||||
terminal_pr=None,
|
||||
claim_ownership=OWNERSHIP_FOREIGN,
|
||||
)
|
||||
self.assertIsNotNone(skip_foreign)
|
||||
self.assertIn("active lease", skip_foreign)
|
||||
|
||||
# Own/Self claim remains selectable for session resumption
|
||||
skip_self = classify_skip(
|
||||
c=candidate,
|
||||
role=ROLE_AUTHOR,
|
||||
terminal_pr=None,
|
||||
claim_ownership=OWNERSHIP_OWN,
|
||||
)
|
||||
self.assertIsNone(skip_self)
|
||||
|
||||
def test_durable_dependency_graph_blocking(self):
|
||||
"""Building block: unmet dependency edge + classify_skip on dependency_unmet."""
|
||||
# Upsert a blocking dependency edge between issue 878 and blocker 601
|
||||
self.db.upsert_dependency_edge(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
source_kind="issue",
|
||||
source_number=878,
|
||||
target_kind="issue",
|
||||
target_number=601,
|
||||
edge_type=dependency_graph.EDGE_ISSUE_BLOCKED_BY_ISSUE,
|
||||
state=dependency_graph.STATE_UNMET,
|
||||
blocking_condition="Target issue #601 is not closed",
|
||||
completion_condition="Target issue #601 is closed",
|
||||
evidence={"source": "unit_test"},
|
||||
)
|
||||
|
||||
edges = self.db.list_dependency_edges(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
source_kind="issue",
|
||||
source_number=878,
|
||||
)
|
||||
self.assertEqual(len(edges), 1)
|
||||
self.assertEqual(edges[0]["state"], "unmet")
|
||||
self.assertEqual(edges[0]["target_number"], 601)
|
||||
|
||||
# When dependency is unmet, candidate is blocked from selection
|
||||
candidate = WorkCandidate(
|
||||
kind="issue",
|
||||
number=878,
|
||||
title="Blocked candidate",
|
||||
state="open",
|
||||
labels=[],
|
||||
blocked=False,
|
||||
dependency_unmet=True,
|
||||
dependency_reason="issue#878 is blocked by unmet dependency issue#601",
|
||||
)
|
||||
skip_reason = classify_skip(
|
||||
c=candidate,
|
||||
role=ROLE_AUTHOR,
|
||||
terminal_pr=None,
|
||||
claim_ownership=OWNERSHIP_OWN,
|
||||
)
|
||||
self.assertIsNotNone(skip_reason)
|
||||
self.assertIn("issue#601", skip_reason)
|
||||
|
||||
def test_dependency_completion_reevaluation(self):
|
||||
"""Building block: dependency edge state can transition UNMET -> MET."""
|
||||
self.db.upsert_dependency_edge(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
source_kind="issue",
|
||||
source_number=878,
|
||||
target_kind="issue",
|
||||
target_number=601,
|
||||
edge_type=dependency_graph.EDGE_ISSUE_BLOCKED_BY_ISSUE,
|
||||
state=dependency_graph.STATE_UNMET,
|
||||
blocking_condition="Target issue #601 is open",
|
||||
completion_condition="Target issue #601 is closed",
|
||||
evidence={"source": "unit_test"},
|
||||
)
|
||||
|
||||
# Mark edge as met upon target issue closure
|
||||
self.db.upsert_dependency_edge(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
source_kind="issue",
|
||||
source_number=878,
|
||||
target_kind="issue",
|
||||
target_number=601,
|
||||
edge_type=dependency_graph.EDGE_ISSUE_BLOCKED_BY_ISSUE,
|
||||
state=dependency_graph.STATE_MET,
|
||||
blocking_condition="Target issue #601 is open",
|
||||
completion_condition="Target issue #601 is closed",
|
||||
evidence={"source": "target_closed_event"},
|
||||
)
|
||||
|
||||
edges = self.db.list_dependency_edges(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
source_kind="issue",
|
||||
source_number=878,
|
||||
)
|
||||
self.assertEqual(len(edges), 1)
|
||||
self.assertEqual(edges[0]["state"], "met")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,249 @@
|
||||
"""Tests for post-restart MCP reconciliation and completion proof (#662)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import unittest
|
||||
from datetime import datetime, timezone
|
||||
|
||||
import post_restart_reconcile as prr
|
||||
|
||||
NOW = datetime(2026, 7, 24, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def _base_inventory(**overrides):
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"incomplete_reasons": [],
|
||||
"service_health": {"healthy": True},
|
||||
"clients": [{"session_id": "c1", "connected": True}],
|
||||
"sessions": [
|
||||
{
|
||||
"session_id": "s-live",
|
||||
"status": "active",
|
||||
"pid": os.getpid(),
|
||||
"role": "author",
|
||||
}
|
||||
],
|
||||
"leases": [],
|
||||
"checkpoints_available": False,
|
||||
"worktree_bindings": [{"path": "/tmp/wt", "exists": True}],
|
||||
"pending_mutations": [],
|
||||
"capabilities": {"stale": False},
|
||||
"boot_head_sha": "a" * 40,
|
||||
"current_head_sha": "a" * 40,
|
||||
"queue_state": {"safe_to_resume": True},
|
||||
}
|
||||
inv.update(overrides)
|
||||
return inv
|
||||
|
||||
|
||||
class IncompleteInventoryTest(unittest.TestCase):
|
||||
def test_incomplete_inventory_fails_closed(self) -> None:
|
||||
proof = prr.reconcile_after_restart(
|
||||
{
|
||||
"inventory_complete": False,
|
||||
"incomplete_reasons": ["control-plane DB unavailable"],
|
||||
},
|
||||
now=NOW,
|
||||
mode=prr.MODE_ENFORCE,
|
||||
reconcile_id="test-incomplete",
|
||||
)
|
||||
self.assertEqual(proof.overall_status, prr.STATUS_FAILED)
|
||||
self.assertTrue(proof.mutation_hold)
|
||||
self.assertFalse(proof.inventory_complete)
|
||||
self.assertTrue(proof.proposed_follow_ups)
|
||||
self.assertIn("control-plane DB unavailable", proof.incomplete_reasons)
|
||||
|
||||
|
||||
class HappyPathTest(unittest.TestCase):
|
||||
def test_clean_restart_is_complete_without_mutation_hold(self) -> None:
|
||||
proof = prr.reconcile_after_restart(
|
||||
_base_inventory(),
|
||||
now=NOW,
|
||||
mode=prr.MODE_ENFORCE,
|
||||
reconcile_id="test-clean",
|
||||
)
|
||||
self.assertEqual(proof.overall_status, prr.STATUS_COMPLETE)
|
||||
self.assertFalse(proof.mutation_hold)
|
||||
self.assertEqual(proof.unresolved_count, 0)
|
||||
cp = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS)
|
||||
self.assertEqual(cp.status, prr.ITEM_SKIPPED)
|
||||
links = proof.as_dict()["links"]
|
||||
self.assertEqual(links["umbrella"], 655)
|
||||
self.assertEqual(links["issue"], 662)
|
||||
self.assertEqual(links["vision"], 652)
|
||||
self.assertEqual(links["roadmap"], 653)
|
||||
|
||||
|
||||
class InterruptedMutationTest(unittest.TestCase):
|
||||
def test_mutating_lease_with_dead_owner_is_unresolved(self) -> None:
|
||||
proof = prr.reconcile_after_restart(
|
||||
_base_inventory(
|
||||
leases=[
|
||||
{
|
||||
"lease_id": "lease-mut",
|
||||
"session_id": "s-dead",
|
||||
"phase": "implementing",
|
||||
"work_kind": "issue",
|
||||
"work_number": 662,
|
||||
"worktree_path": "/tmp/wt-662",
|
||||
"freshness": {"freshness": "stale_dead_process"},
|
||||
}
|
||||
]
|
||||
),
|
||||
now=NOW,
|
||||
mode=prr.MODE_ENFORCE,
|
||||
)
|
||||
mut = next(i for i in proof.items if i.dimension == prr.DIM_MUTATIONS)
|
||||
self.assertEqual(mut.status, prr.ITEM_UNRESOLVED)
|
||||
self.assertTrue(mut.follow_up_required)
|
||||
interrupted = mut.details["interrupted"]
|
||||
self.assertEqual(len(interrupted), 1)
|
||||
self.assertFalse(interrupted[0]["resume_allowed"])
|
||||
self.assertTrue(proof.mutation_hold)
|
||||
self.assertTrue(
|
||||
any(f.dimension == prr.DIM_MUTATIONS for f in proof.proposed_follow_ups)
|
||||
)
|
||||
|
||||
def test_explicit_pending_mutation_inventory(self) -> None:
|
||||
proof = prr.reconcile_after_restart(
|
||||
_base_inventory(
|
||||
pending_mutations=[
|
||||
{
|
||||
"session_id": "s1",
|
||||
"phase": "publishing",
|
||||
"work_kind": "pr",
|
||||
"work_number": 856,
|
||||
"reason": "push interrupted mid-flight",
|
||||
}
|
||||
]
|
||||
),
|
||||
now=NOW,
|
||||
mode=prr.MODE_LOG_ONLY,
|
||||
)
|
||||
mut = next(i for i in proof.items if i.dimension == prr.DIM_MUTATIONS)
|
||||
self.assertEqual(mut.status, prr.ITEM_UNRESOLVED)
|
||||
# log_only never holds mutations even when unresolved
|
||||
self.assertFalse(proof.mutation_hold)
|
||||
self.assertEqual(proof.overall_status, prr.STATUS_DEGRADED)
|
||||
|
||||
|
||||
class DuplicateClaimsTest(unittest.TestCase):
|
||||
def test_duplicate_live_claims_flagged(self) -> None:
|
||||
proof = prr.reconcile_after_restart(
|
||||
_base_inventory(
|
||||
leases=[
|
||||
{
|
||||
"lease_id": "l1",
|
||||
"session_id": "s1",
|
||||
"phase": "allocated",
|
||||
"work_kind": "issue",
|
||||
"work_number": 100,
|
||||
"freshness": {"freshness": "active"},
|
||||
},
|
||||
{
|
||||
"lease_id": "l2",
|
||||
"session_id": "s2",
|
||||
"phase": "allocated",
|
||||
"work_kind": "issue",
|
||||
"work_number": 100,
|
||||
"freshness": {"freshness": "active"},
|
||||
},
|
||||
]
|
||||
),
|
||||
now=NOW,
|
||||
mode=prr.MODE_ENFORCE,
|
||||
)
|
||||
dups = next(i for i in proof.items if i.dimension == prr.DIM_DUPLICATES)
|
||||
self.assertEqual(dups.status, prr.ITEM_UNRESOLVED)
|
||||
self.assertEqual(dups.details["duplicates"][0]["claim_count"], 2)
|
||||
self.assertTrue(proof.mutation_hold)
|
||||
|
||||
|
||||
class OrphanSessionTest(unittest.TestCase):
|
||||
def test_active_session_dead_pid_is_unresolved(self) -> None:
|
||||
proof = prr.reconcile_after_restart(
|
||||
_base_inventory(
|
||||
sessions=[
|
||||
{
|
||||
"session_id": "ghost",
|
||||
"status": "active",
|
||||
"pid": 2_000_000_000,
|
||||
"role": "author",
|
||||
}
|
||||
]
|
||||
),
|
||||
now=NOW,
|
||||
mode=prr.MODE_ENFORCE,
|
||||
)
|
||||
sess = next(i for i in proof.items if i.dimension == prr.DIM_SESSIONS)
|
||||
self.assertEqual(sess.status, prr.ITEM_UNRESOLVED)
|
||||
self.assertIn("ghost", sess.details["orphan_session_ids"])
|
||||
|
||||
|
||||
class CapabilityStaleTest(unittest.TestCase):
|
||||
def test_stale_runtime_unresolved(self) -> None:
|
||||
proof = prr.reconcile_after_restart(
|
||||
_base_inventory(capabilities={"stale": True, "startup_head": "aaa"}),
|
||||
now=NOW,
|
||||
mode=prr.MODE_ENFORCE,
|
||||
)
|
||||
caps = next(i for i in proof.items if i.dimension == prr.DIM_CAPABILITIES)
|
||||
self.assertEqual(caps.status, prr.ITEM_UNRESOLVED)
|
||||
self.assertTrue(proof.mutation_hold)
|
||||
|
||||
|
||||
class MutationsAllowedHelperTest(unittest.TestCase):
|
||||
def test_mutations_allowed_respects_hold(self) -> None:
|
||||
held = prr.reconcile_after_restart(
|
||||
_base_inventory(
|
||||
pending_mutations=[{"phase": "merging", "session_id": "x"}]
|
||||
),
|
||||
now=NOW,
|
||||
mode=prr.MODE_ENFORCE,
|
||||
)
|
||||
self.assertFalse(prr.mutations_allowed(held))
|
||||
self.assertFalse(prr.mutations_allowed(held.as_dict()))
|
||||
clean = prr.reconcile_after_restart(
|
||||
_base_inventory(), now=NOW, mode=prr.MODE_ENFORCE
|
||||
)
|
||||
self.assertTrue(prr.mutations_allowed(clean))
|
||||
|
||||
|
||||
class CheckpointSoftDependencyTest(unittest.TestCase):
|
||||
def test_checkpoints_when_schema_present(self) -> None:
|
||||
proof = prr.reconcile_after_restart(
|
||||
_base_inventory(
|
||||
checkpoints_available=True,
|
||||
checkpoints=[{"session_id": "s1", "stale": False}],
|
||||
),
|
||||
now=NOW,
|
||||
)
|
||||
cp = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS)
|
||||
self.assertEqual(cp.status, prr.ITEM_RESOLVED)
|
||||
|
||||
def test_stale_checkpoints_unresolved(self) -> None:
|
||||
proof = prr.reconcile_after_restart(
|
||||
_base_inventory(
|
||||
checkpoints_available=True,
|
||||
checkpoints=[{"session_id": "s1", "stale": True}],
|
||||
),
|
||||
now=NOW,
|
||||
mode=prr.MODE_ENFORCE,
|
||||
)
|
||||
cp = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS)
|
||||
self.assertEqual(cp.status, prr.ITEM_UNRESOLVED)
|
||||
|
||||
|
||||
class ProofSerializationTest(unittest.TestCase):
|
||||
def test_as_dict_is_json_friendly(self) -> None:
|
||||
proof = prr.reconcile_after_restart(_base_inventory(), now=NOW)
|
||||
blob = json.dumps(proof.as_dict())
|
||||
self.assertIn("reconcile_id", blob)
|
||||
self.assertIn("proposed_follow_ups", blob)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,323 @@
|
||||
"""Tests for sanctioned Codex MCP reconnect request surface (#678)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from unittest import mock
|
||||
|
||||
import mcp_client_reconnect as mcr
|
||||
|
||||
|
||||
class NormalizeReasonTests(unittest.TestCase):
|
||||
def test_stale_runtime_aliases(self):
|
||||
self.assertEqual(mcr.normalize_reason("stale-runtime"), mcr.REASON_STALE_RUNTIME)
|
||||
self.assertEqual(mcr.normalize_reason("stale_runtime"), mcr.REASON_STALE_RUNTIME)
|
||||
self.assertEqual(mcr.normalize_reason("STALE"), mcr.REASON_STALE_RUNTIME)
|
||||
|
||||
def test_transport_eof_aliases(self):
|
||||
self.assertEqual(mcr.normalize_reason("transport_eof"), mcr.REASON_TRANSPORT_EOF)
|
||||
self.assertEqual(mcr.normalize_reason("EOF"), mcr.REASON_TRANSPORT_EOF)
|
||||
self.assertEqual(
|
||||
mcr.normalize_reason("client_is_closing"), mcr.REASON_TRANSPORT_EOF
|
||||
)
|
||||
|
||||
def test_missing_namespace(self):
|
||||
self.assertEqual(
|
||||
mcr.normalize_reason("missing_namespace"), mcr.REASON_MISSING_NAMESPACE
|
||||
)
|
||||
|
||||
def test_empty_is_unspecified(self):
|
||||
self.assertEqual(mcr.normalize_reason(None), mcr.REASON_UNSPECIFIED)
|
||||
self.assertEqual(mcr.normalize_reason(""), mcr.REASON_UNSPECIFIED)
|
||||
|
||||
|
||||
class BoundaryClassificationTests(unittest.TestCase):
|
||||
def test_clean_when_shas_match(self):
|
||||
self.assertEqual(
|
||||
mcr.classify_boundary_status(
|
||||
startup_sha="abc", current_master_sha="abc"
|
||||
),
|
||||
mcr.BOUNDARY_CLEAN,
|
||||
)
|
||||
|
||||
def test_mismatch_when_shas_differ(self):
|
||||
self.assertEqual(
|
||||
mcr.classify_boundary_status(
|
||||
startup_sha="aaa", current_master_sha="bbb"
|
||||
),
|
||||
mcr.BOUNDARY_MISMATCH,
|
||||
)
|
||||
|
||||
def test_stale_when_live_stale(self):
|
||||
self.assertEqual(
|
||||
mcr.classify_boundary_status(
|
||||
startup_sha="aaa",
|
||||
current_master_sha="aaa",
|
||||
live_stale=True,
|
||||
),
|
||||
mcr.BOUNDARY_STALE,
|
||||
)
|
||||
|
||||
|
||||
class BuildReconnectRequestTests(unittest.TestCase):
|
||||
def test_stale_runtime_returns_typed_blocker_with_codex_steps(self):
|
||||
result = mcr.build_reconnect_request(
|
||||
namespace="gitea-author",
|
||||
profile="prgs-author",
|
||||
pid=1234,
|
||||
session_id="sess-1",
|
||||
startup_sha="aaa111",
|
||||
current_master_sha="bbb222",
|
||||
reason="stale-runtime",
|
||||
client="codex",
|
||||
restart_required=True,
|
||||
stop_required=True,
|
||||
)
|
||||
self.assertTrue(result["success"])
|
||||
self.assertTrue(result["read_only"])
|
||||
self.assertFalse(result["reconnect_performed"])
|
||||
self.assertFalse(result["mutation_performed"])
|
||||
self.assertTrue(result["reconnect_needed"])
|
||||
self.assertEqual(result["namespace"], "gitea-author")
|
||||
self.assertEqual(result["profile"], "prgs-author")
|
||||
self.assertEqual(result["pid"], 1234)
|
||||
self.assertEqual(result["session_id"], "sess-1")
|
||||
self.assertEqual(result["startup_sha"], "aaa111")
|
||||
self.assertEqual(result["current_master_sha"], "bbb222")
|
||||
self.assertEqual(result["boundary_status"], mcr.BOUNDARY_MISMATCH)
|
||||
self.assertEqual(result["blocker_kind"], mcr.BLOCKER_OPERATOR_RECONNECT)
|
||||
self.assertIsNotNone(result["typed_blocker"])
|
||||
blocker = result["typed_blocker"]
|
||||
self.assertEqual(blocker["namespaces"], ["gitea-author"])
|
||||
self.assertEqual(blocker["why_reconnect_required"], mcr.REASON_STALE_RUNTIME)
|
||||
self.assertTrue(any("Codex" in s or "Reload" in s for s in blocker["operator_ui_steps"]))
|
||||
self.assertIn("pkill", " ".join(result["forbidden_recovery_paths"]).lower())
|
||||
self.assertTrue(
|
||||
mcr.reasons_never_suggest_forbidden(result["exact_safe_next_action"] or "")
|
||||
)
|
||||
# Must not recommend forbidden recovery.
|
||||
for step in blocker["operator_ui_steps"]:
|
||||
self.assertTrue(mcr.reasons_never_suggest_forbidden(step), step)
|
||||
|
||||
def test_transport_eof_typed_blocker(self):
|
||||
result = mcr.build_reconnect_request(
|
||||
namespace="gitea-reviewer",
|
||||
reason="transport_eof",
|
||||
client="claude_code",
|
||||
)
|
||||
self.assertTrue(result["reconnect_needed"])
|
||||
self.assertEqual(result["reason"], mcr.REASON_TRANSPORT_EOF)
|
||||
self.assertEqual(result["client"], "claude_code")
|
||||
steps = " ".join(result["operator_ui_steps"]).lower()
|
||||
self.assertIn("/mcp", steps)
|
||||
|
||||
def test_missing_namespace_typed_blocker(self):
|
||||
result = mcr.build_reconnect_request(
|
||||
namespace="gitea-merger",
|
||||
reason="missing_namespace",
|
||||
client="codex",
|
||||
)
|
||||
self.assertTrue(result["reconnect_needed"])
|
||||
self.assertEqual(result["reason"], mcr.REASON_MISSING_NAMESPACE)
|
||||
self.assertEqual(
|
||||
result["typed_blocker"]["blocker_kind"], mcr.BLOCKER_OPERATOR_RECONNECT
|
||||
)
|
||||
|
||||
def test_healthy_not_required(self):
|
||||
result = mcr.build_reconnect_request(
|
||||
namespace="gitea-tools",
|
||||
startup_sha="deadbeef",
|
||||
current_master_sha="deadbeef",
|
||||
reason="not_required",
|
||||
client="codex",
|
||||
in_parity=True,
|
||||
restart_required=False,
|
||||
stop_required=False,
|
||||
)
|
||||
self.assertFalse(result["reconnect_needed"])
|
||||
self.assertEqual(result["blocker_kind"], mcr.BLOCKER_NONE)
|
||||
self.assertIsNone(result["typed_blocker"])
|
||||
self.assertFalse(result["stop_required"])
|
||||
self.assertFalse(result["restart_required"])
|
||||
self.assertIn("not required", (result["exact_safe_next_action"] or "").lower())
|
||||
|
||||
def test_successful_reconnect_report_fields_present(self):
|
||||
"""AC2: reconnect result reports required fields (even when needed)."""
|
||||
result = mcr.build_reconnect_request(
|
||||
namespace="gitea-controller",
|
||||
profile="prgs-controller",
|
||||
pid=99,
|
||||
session_id="sid",
|
||||
startup_sha="s" * 40,
|
||||
current_master_sha="c" * 40,
|
||||
reason="stale-runtime",
|
||||
)
|
||||
for key in (
|
||||
"namespace",
|
||||
"profile",
|
||||
"pid",
|
||||
"session_id",
|
||||
"startup_sha",
|
||||
"current_master_sha",
|
||||
"boundary_status",
|
||||
):
|
||||
self.assertIn(key, result)
|
||||
self.assertIsNotNone(result[key], key)
|
||||
|
||||
|
||||
class ToolSurfaceTests(unittest.TestCase):
|
||||
"""Exercise gitea_request_mcp_reconnect with a stubbed server context."""
|
||||
|
||||
def test_tool_is_registered_and_side_effect_free(self):
|
||||
import gitea_mcp_server as srv
|
||||
|
||||
self.assertTrue(hasattr(srv, "gitea_request_mcp_reconnect"))
|
||||
with mock.patch.object(srv, "_profile_operation_gate", return_value=None):
|
||||
with mock.patch.object(
|
||||
srv,
|
||||
"get_profile",
|
||||
return_value={
|
||||
"profile_name": "prgs-author",
|
||||
"role_kind": "author",
|
||||
"role": "author",
|
||||
},
|
||||
):
|
||||
with mock.patch.object(
|
||||
srv,
|
||||
"_current_master_parity",
|
||||
return_value={
|
||||
"startup_head": "a" * 40,
|
||||
"current_head": "a" * 40,
|
||||
"daemon_start_head": "a" * 40,
|
||||
"local_head": "a" * 40,
|
||||
"in_parity": True,
|
||||
"stale": False,
|
||||
"restart_required": False,
|
||||
"determinable": True,
|
||||
"live_stale": False,
|
||||
"live_known": True,
|
||||
"reasons": [],
|
||||
},
|
||||
):
|
||||
with mock.patch.object(
|
||||
srv.master_parity_gate,
|
||||
"format_parity",
|
||||
return_value="in parity",
|
||||
):
|
||||
with mock.patch.object(
|
||||
srv.role_namespace_gate,
|
||||
"infer_mcp_namespace",
|
||||
return_value="gitea-author",
|
||||
):
|
||||
with mock.patch.object(
|
||||
srv.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": "prgs-author"},
|
||||
):
|
||||
result = srv.gitea_request_mcp_reconnect(
|
||||
namespace="gitea-author",
|
||||
reason="not_required",
|
||||
client="codex",
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertTrue(result.get("success"))
|
||||
self.assertFalse(result.get("reconnect_performed"))
|
||||
self.assertFalse(result.get("mutation_performed"))
|
||||
self.assertEqual(result.get("namespace"), "gitea-author")
|
||||
self.assertEqual(result.get("profile"), "prgs-author")
|
||||
self.assertEqual(result.get("pid"), os.getpid())
|
||||
self.assertIn("startup_sha", result)
|
||||
self.assertIn("current_master_sha", result)
|
||||
self.assertIn("boundary_status", result)
|
||||
self.assertTrue(
|
||||
mcr.reasons_never_suggest_forbidden(
|
||||
result.get("exact_safe_next_action") or ""
|
||||
)
|
||||
)
|
||||
|
||||
def test_tool_stale_returns_typed_blocker(self):
|
||||
import gitea_mcp_server as srv
|
||||
|
||||
with mock.patch.object(srv, "_profile_operation_gate", return_value=None):
|
||||
with mock.patch.object(
|
||||
srv,
|
||||
"get_profile",
|
||||
return_value={
|
||||
"profile_name": "prgs-reconciler",
|
||||
"role_kind": "reconciler",
|
||||
"role": "reconciler",
|
||||
},
|
||||
):
|
||||
with mock.patch.object(
|
||||
srv,
|
||||
"_current_master_parity",
|
||||
return_value={
|
||||
"startup_head": "a" * 40,
|
||||
"current_head": "b" * 40,
|
||||
"daemon_start_head": "a" * 40,
|
||||
"local_head": "b" * 40,
|
||||
"in_parity": False,
|
||||
"stale": True,
|
||||
"restart_required": True,
|
||||
"determinable": True,
|
||||
"live_stale": True,
|
||||
"live_known": True,
|
||||
"reasons": ["stale"],
|
||||
},
|
||||
):
|
||||
with mock.patch.object(
|
||||
srv.master_parity_gate,
|
||||
"format_parity",
|
||||
return_value="stale",
|
||||
):
|
||||
with mock.patch.object(
|
||||
srv.role_namespace_gate,
|
||||
"infer_mcp_namespace",
|
||||
return_value="gitea-reconciler",
|
||||
):
|
||||
with mock.patch.object(
|
||||
srv.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={},
|
||||
):
|
||||
result = srv.gitea_request_mcp_reconnect(
|
||||
reason="stale-runtime",
|
||||
client="codex",
|
||||
)
|
||||
self.assertTrue(result["reconnect_needed"])
|
||||
self.assertEqual(
|
||||
result["blocker_kind"], mcr.BLOCKER_OPERATOR_RECONNECT
|
||||
)
|
||||
self.assertIsNotNone(result["typed_blocker"])
|
||||
self.assertIn("gitea-reconciler", result["typed_blocker"]["namespaces"])
|
||||
self.assertTrue(result["stop_required"])
|
||||
self.assertTrue(result["restart_required"])
|
||||
self.assertTrue(
|
||||
mcr.reasons_never_suggest_forbidden(
|
||||
result.get("exact_safe_next_action") or ""
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
class InventoryRegistrationTests(unittest.TestCase):
|
||||
def test_reconnect_path_in_restart_inventory(self):
|
||||
import mcp_restart_paths as mrp
|
||||
|
||||
ids = {p.path_id for p in mrp.iter_restart_paths()}
|
||||
self.assertIn("codex_client_reconnect_request", ids)
|
||||
self.assertIn("ide_client_reconnect", ids)
|
||||
|
||||
def test_tool_name_in_documented_inventory(self):
|
||||
import mcp_tool_inventory as inv
|
||||
|
||||
doc_path = os.path.join(
|
||||
os.path.dirname(os.path.dirname(__file__)), inv.INVENTORY_DOC_PATH
|
||||
)
|
||||
with open(doc_path, encoding="utf-8") as handle:
|
||||
documented = inv.parse_documented_inventory(handle.read())
|
||||
self.assertIn("gitea_request_mcp_reconnect", documented)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,72 @@
|
||||
"""Starlette TestClient uses httpx2 without deprecation warning (#682)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import sys
|
||||
import unittest
|
||||
import warnings
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
|
||||
class TestHttpx2Installed(unittest.TestCase):
|
||||
def test_httpx2_importable(self) -> None:
|
||||
httpx2 = importlib.import_module("httpx2")
|
||||
self.assertTrue(hasattr(httpx2, "Client") or hasattr(httpx2, "AsyncClient"))
|
||||
|
||||
|
||||
class TestNoStarletteTestClientDeprecation(unittest.TestCase):
|
||||
def test_starlette_testclient_import_does_not_warn(self) -> None:
|
||||
# Force a fresh import path so a previous httpx-fallback warning cannot
|
||||
# hide a regression after httpx2 is present. Restore sys.modules afterwards.
|
||||
saved = {
|
||||
name: sys.modules[name]
|
||||
for name in list(sys.modules)
|
||||
if name == "starlette.testclient" or name.startswith("starlette.testclient.")
|
||||
}
|
||||
try:
|
||||
for name in list(saved):
|
||||
del sys.modules[name]
|
||||
|
||||
with warnings.catch_warnings(record=True) as caught:
|
||||
warnings.simplefilter("always")
|
||||
import starlette.testclient as testclient # noqa: F401
|
||||
|
||||
dep = [
|
||||
w
|
||||
for w in caught
|
||||
if "httpx" in str(w.message).lower()
|
||||
and "deprecated" in str(w.message).lower()
|
||||
]
|
||||
self.assertEqual(
|
||||
dep,
|
||||
[],
|
||||
f"expected no Starlette httpx deprecation warning; got: "
|
||||
f"{[str(w.message) for w in dep]}",
|
||||
)
|
||||
finally:
|
||||
sys.modules.update(saved)
|
||||
|
||||
def test_canonical_webui_testclient_import(self) -> None:
|
||||
with warnings.catch_warnings(record=True) as caught:
|
||||
warnings.simplefilter("always")
|
||||
from tests.webui_testclient import TestClient
|
||||
from webui.app import create_app
|
||||
|
||||
client = TestClient(create_app())
|
||||
response = client.get("/")
|
||||
self.assertIn(response.status_code, {200, 302, 404})
|
||||
|
||||
dep = [
|
||||
w
|
||||
for w in caught
|
||||
if "httpx" in str(w.message).lower()
|
||||
and "deprecated" in str(w.message).lower()
|
||||
]
|
||||
self.assertEqual(dep, [], [str(w.message) for w in dep])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,139 @@
|
||||
"""Tests for Issue #686: Detect and reject manually launched duplicate MCP role servers."""
|
||||
import os
|
||||
import unittest
|
||||
from unittest.mock import patch, MagicMock
|
||||
from datetime import datetime
|
||||
|
||||
import gitea_config
|
||||
import gitea_mcp_server
|
||||
import mcp_namespace_health
|
||||
|
||||
|
||||
class TestIssue686ManualMcpProvenance(unittest.TestCase):
|
||||
|
||||
def test_client_managed_process_detection(self):
|
||||
"""Test _is_client_managed_process correctly detects provenance markers."""
|
||||
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "1"}, clear=True):
|
||||
self.assertTrue(gitea_mcp_server._is_client_managed_process())
|
||||
|
||||
with patch.dict(os.environ, {"GITEA_MCP_CLIENT_MANAGED": "true"}, clear=True):
|
||||
self.assertTrue(gitea_mcp_server._is_client_managed_process())
|
||||
|
||||
with patch.dict(os.environ, {"GITEA_SERVER_PROVENANCE": "client_managed"}, clear=True):
|
||||
self.assertTrue(gitea_mcp_server._is_client_managed_process())
|
||||
|
||||
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "0"}, clear=True):
|
||||
self.assertFalse(gitea_mcp_server._is_client_managed_process())
|
||||
|
||||
def test_unconsumed_gitea_env_overrides(self):
|
||||
"""Test surfacing of unsupported GITEA_* env overrides (e.g. GITEA_DUMMY)."""
|
||||
env = {
|
||||
"GITEA_MCP_PROFILE": "prgs-author",
|
||||
"GITEA_CLIENT_MANAGED": "1",
|
||||
"GITEA_DUMMY": "2",
|
||||
"GITEA_UNKNOWN_FLAG": "abc",
|
||||
}
|
||||
unconsumed = gitea_config.get_unconsumed_gitea_env_overrides(env)
|
||||
self.assertIn("GITEA_DUMMY", unconsumed)
|
||||
self.assertEqual(unconsumed["GITEA_DUMMY"], "2")
|
||||
self.assertIn("GITEA_UNKNOWN_FLAG", unconsumed)
|
||||
self.assertNotIn("GITEA_MCP_PROFILE", unconsumed)
|
||||
self.assertNotIn("GITEA_CLIENT_MANAGED", unconsumed)
|
||||
|
||||
def test_manual_server_mutation_fail_closed(self):
|
||||
"""AC 2: Mutating tools on a server without client-managed provenance fail closed with a typed blocker."""
|
||||
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "0"}, clear=True):
|
||||
block = gitea_mcp_server._provenance_mutation_block(task="create_issue")
|
||||
self.assertIsNotNone(block)
|
||||
self.assertFalse(block["success"])
|
||||
self.assertFalse(block["performed"])
|
||||
self.assertEqual(block["blocker_kind"], "unsupported_manual_launch")
|
||||
self.assertEqual(block["provenance"], "manual_launch")
|
||||
self.assertTrue(any("mutation denied: server process was launched manually" in r for r in block["reasons"]))
|
||||
self.assertIn("BLOCKED + RECONNECT", block["exact_next_action"])
|
||||
|
||||
def test_client_managed_server_mutation_passes_provenance_gate(self):
|
||||
"""AC 3: Clean client-managed baseline passes the provenance gate."""
|
||||
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "1"}, clear=True):
|
||||
block = gitea_mcp_server._provenance_mutation_block(task="create_issue")
|
||||
self.assertIsNone(block)
|
||||
|
||||
@patch("subprocess.run")
|
||||
@patch("os.path.getmtime")
|
||||
@patch("os.path.exists")
|
||||
@patch("os.getpid")
|
||||
def test_manual_duplicate_does_not_mask_stale_runtime(
|
||||
self, mock_getpid, mock_exists, mock_getmtime, mock_run
|
||||
):
|
||||
"""AC 1 & AC 3: Staleness detection ignores manual duplicates and reports stale supported runtimes."""
|
||||
mock_getpid.return_value = 12345
|
||||
mock_exists.return_value = True
|
||||
|
||||
code_time = datetime(2026, 7, 8, 14, 0, 0)
|
||||
mock_getmtime.return_value = code_time.timestamp()
|
||||
|
||||
# PID 12345: stale client-managed process (started at 13:00)
|
||||
# PID 99999: fresh manual duplicate process (started at 15:00, no GITEA_CLIENT_MANAGED)
|
||||
ps_output = (
|
||||
" PID LSTART COMMAND\n"
|
||||
"12345 Wed Jul 8 13:00:00 2026 /path/to/python mcp_server.py\n"
|
||||
"99999 Wed Jul 8 15:00:00 2026 /path/to/python mcp_server.py\n"
|
||||
)
|
||||
|
||||
mock_run_ps = MagicMock()
|
||||
mock_run_ps.stdout = ps_output
|
||||
|
||||
mock_env_12345 = MagicMock()
|
||||
mock_env_12345.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_CLIENT_MANAGED=1"
|
||||
|
||||
mock_env_99999 = MagicMock()
|
||||
mock_env_99999.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_DUMMY=2"
|
||||
|
||||
def side_effect(args, **kwargs):
|
||||
if args[0] == "ps" and "eww" in args:
|
||||
pid = args[2]
|
||||
if pid == "12345":
|
||||
return mock_env_12345
|
||||
elif pid == "99999":
|
||||
return mock_env_99999
|
||||
elif args[0] == "ps":
|
||||
return mock_run_ps
|
||||
raise ValueError(f"Unexpected args: {args}")
|
||||
|
||||
mock_run.side_effect = side_effect
|
||||
|
||||
reasons = gitea_mcp_server._check_mcp_runtimes_diagnostics("create_issue", ["prgs-author"])
|
||||
|
||||
# Manual duplicate process must be flagged
|
||||
self.assertTrue(any("Duplicate MCP server process(es) detected" in r for r in reasons))
|
||||
# Unsupported env override (GITEA_DUMMY=2) must be flagged
|
||||
self.assertTrue(any("unsupported-env: Unsupported GITEA_* environment variable override(s) detected: GITEA_DUMMY=2" in r for r in reasons))
|
||||
# Stale runtime must NOT be masked by fresh manual process 99999!
|
||||
self.assertTrue(any("All matching profiles for task 'create_issue' (['prgs-author']) are running but stale" in r for r in reasons))
|
||||
|
||||
def test_namespace_health_classification_includes_provenance(self):
|
||||
"""AC 1 & 4: mcp_namespace_health diagnostics include provenance and unconsumed_gitea_env."""
|
||||
process = {
|
||||
"pid": 5555,
|
||||
"profile": "prgs-author",
|
||||
"env": {
|
||||
"GITEA_MCP_PROFILE": "prgs-author",
|
||||
"GITEA_DUMMY": "99",
|
||||
},
|
||||
}
|
||||
res = mcp_namespace_health.classify_namespace_probe(
|
||||
"gitea-author",
|
||||
configured=True,
|
||||
registered_tools=["gitea_whoami"],
|
||||
probe_result={"success": True},
|
||||
process=process,
|
||||
probe_source="client_namespace",
|
||||
)
|
||||
self.assertEqual(res["provenance"], "manual_launch")
|
||||
self.assertFalse(res["is_client_managed"])
|
||||
self.assertEqual(res["unconsumed_gitea_env"], {"GITEA_DUMMY": "99"})
|
||||
self.assertEqual(res["diagnostics"]["provenance"], "manual_launch")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,444 @@
|
||||
"""Task heartbeat through the native MCP author path (#790 Slice A, AC-N6).
|
||||
|
||||
Assessor-level coverage is not sufficient here, and this project has already
|
||||
paid for learning that: in review #499 on PR #791 the #760 renewal waiver was
|
||||
computed correctly and then *discarded* at two later gates, so every real
|
||||
renewal still failed while the unit suite stayed green. AC-N6 exists because of
|
||||
that, and requires driving the real tools against a real git repository and a
|
||||
real durable lock file, composing the gates in production order.
|
||||
|
||||
These tests therefore call ``gitea_lock_issue`` and
|
||||
``gitea_heartbeat_issue_lock`` themselves and assert on what lands on disk,
|
||||
never on an assessor's return value alone.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
from mutation_profile_fixture import shared_mutation_env # noqa: E402
|
||||
|
||||
import issue_lock_provenance # noqa: E402
|
||||
import issue_lock_store # noqa: E402
|
||||
import lease_policy # noqa: E402
|
||||
import mcp_server # noqa: E402
|
||||
|
||||
ISSUE = 9791
|
||||
BRANCH = f"fix/issue-{ISSUE}-heartbeat-mcp"
|
||||
IDENTITY = "example-user"
|
||||
PROFILE = "test-author-prgs"
|
||||
ORG = "Scaled-Tech-Consulting"
|
||||
REPO = "Gitea-Tools"
|
||||
|
||||
|
||||
def _ts(moment: datetime) -> str:
|
||||
return (
|
||||
moment.astimezone(timezone.utc)
|
||||
.replace(microsecond=0)
|
||||
.isoformat()
|
||||
.replace("+00:00", "Z")
|
||||
)
|
||||
|
||||
|
||||
class _HeartbeatMcpBase(unittest.TestCase):
|
||||
"""Real git repo plus a real durable lock, driven through the real tools."""
|
||||
|
||||
def setUp(self):
|
||||
self.lock_dir = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self.lock_dir.cleanup)
|
||||
self.repo = tempfile.mkdtemp(prefix="issue790-mcp-")
|
||||
self.addCleanup(lambda: subprocess.run(["rm", "-rf", self.repo], check=False))
|
||||
self._init_worktree()
|
||||
self.remotes = patch.dict(
|
||||
mcp_server.REMOTES,
|
||||
{"prgs": {"host": "gitea.prgs.cc", "org": ORG, "repo": REPO}},
|
||||
)
|
||||
self.remotes.start()
|
||||
self.addCleanup(patch.stopall)
|
||||
mcp_server._IDENTITY_CACHE.clear()
|
||||
|
||||
def _git(self, *args):
|
||||
return subprocess.run(
|
||||
["git", "-C", self.repo, *args], capture_output=True, text=True, check=True
|
||||
)
|
||||
|
||||
def _init_worktree(self):
|
||||
self._git("init", "-q", "-b", "master")
|
||||
self._git("config", "user.email", "[email protected]")
|
||||
self._git("config", "user.name", "Test")
|
||||
with open(os.path.join(self.repo, "seed.txt"), "w") as fh:
|
||||
fh.write("seed\n")
|
||||
self._git("add", "seed.txt")
|
||||
self._git("commit", "-q", "-m", "seed")
|
||||
self.base_sha = self._git("rev-parse", "HEAD").stdout.strip()
|
||||
# A fresh claim starts base-equivalent, which is the ordinary first-lock
|
||||
# shape and exercises assess_issue_lock_worktree on its normal path.
|
||||
self._git("checkout", "-q", "-b", BRANCH)
|
||||
self.head_sha = self.base_sha
|
||||
self.worktree = os.path.realpath(self.repo)
|
||||
|
||||
def _lock_path(self):
|
||||
return issue_lock_store.lock_file_path(
|
||||
remote="prgs",
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
issue_number=ISSUE,
|
||||
lock_dir=self.lock_dir.name,
|
||||
)
|
||||
|
||||
def _tool_env(self):
|
||||
env = shared_mutation_env(
|
||||
PROFILE, include_example_repo=True, GITEA_ISSUE_LOCK_DIR=self.lock_dir.name
|
||||
)
|
||||
env["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||
return env
|
||||
|
||||
def _git_state(self, *, porcelain="", base_equivalent=True):
|
||||
return {
|
||||
"current_branch": BRANCH,
|
||||
"porcelain_status": porcelain,
|
||||
"base_equivalent": base_equivalent,
|
||||
"head_sha": self.head_sha,
|
||||
"inspected_git_root": self.worktree,
|
||||
"base_branch": "master",
|
||||
}
|
||||
|
||||
def run_lock_issue(
|
||||
self,
|
||||
*,
|
||||
branch_entries=None,
|
||||
open_prs=None,
|
||||
git_state=None,
|
||||
identity=IDENTITY,
|
||||
profile=PROFILE,
|
||||
):
|
||||
branch_entries = branch_entries if branch_entries is not None else []
|
||||
open_prs = open_prs if open_prs is not None else []
|
||||
git_state = git_state or self._git_state()
|
||||
env = self._tool_env()
|
||||
with patch(
|
||||
"mcp_server.api_get_all", return_value=list(branch_entries)
|
||||
), patch(
|
||||
"mcp_server._list_open_pulls", return_value=list(open_prs)
|
||||
), patch(
|
||||
"mcp_server.get_auth_header", return_value="token x"
|
||||
), patch(
|
||||
"mcp_server._work_lease_claimant",
|
||||
return_value={"username": identity, "profile": profile},
|
||||
), patch(
|
||||
"mcp_server.issue_lock_worktree.read_worktree_git_state",
|
||||
return_value=git_state,
|
||||
), patch(
|
||||
"mcp_server.issue_duplicate_context_fetcher",
|
||||
side_effect=lambda h, o, r, auth, issue_number: (
|
||||
list(open_prs),
|
||||
[b.get("name") for b in branch_entries if isinstance(b, dict)],
|
||||
{"status": "not_claimed"},
|
||||
),
|
||||
), patch.dict(os.environ, env, clear=True):
|
||||
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||
return mcp_server.gitea_lock_issue(
|
||||
issue_number=ISSUE,
|
||||
branch_name=BRANCH,
|
||||
remote="prgs",
|
||||
worktree_path=self.worktree,
|
||||
)
|
||||
|
||||
def run_heartbeat(
|
||||
self, *, task_session_id, identity=IDENTITY, profile=PROFILE, **kwargs
|
||||
):
|
||||
env = self._tool_env()
|
||||
with patch(
|
||||
"mcp_server._work_lease_claimant",
|
||||
return_value={"username": identity, "profile": profile},
|
||||
), patch("mcp_server.get_auth_header", return_value="token x"), patch.dict(
|
||||
os.environ, env, clear=True
|
||||
):
|
||||
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||
return mcp_server.gitea_heartbeat_issue_lock(
|
||||
issue_number=ISSUE,
|
||||
branch_name=kwargs.pop("branch_name", BRANCH),
|
||||
task_session_id=task_session_id,
|
||||
remote="prgs",
|
||||
worktree_path=kwargs.pop("worktree_path", self.worktree),
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
def write_legacy_lock(self, *, hours_old: float = 3.0, ttl_hours: float = 4.0):
|
||||
"""A durable lock in the shape the store wrote before this slice."""
|
||||
now = datetime.now(timezone.utc)
|
||||
claimant = {"username": IDENTITY, "profile": PROFILE}
|
||||
created = now - timedelta(hours=hours_old)
|
||||
record = {
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": BRANCH,
|
||||
"remote": "prgs",
|
||||
"org": ORG,
|
||||
"repo": REPO,
|
||||
"worktree_path": self.worktree,
|
||||
"session_pid": os.getpid(),
|
||||
"pid": os.getpid(),
|
||||
"lock_generation": 1,
|
||||
"work_lease": {
|
||||
"operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE,
|
||||
"issue_number": ISSUE,
|
||||
"pr_number": None,
|
||||
"branch": BRANCH,
|
||||
"worktree_path": self.worktree,
|
||||
"claimant": claimant,
|
||||
"created_at": _ts(created),
|
||||
# The legacy signature: never advanced past creation.
|
||||
"last_heartbeat_at": _ts(created),
|
||||
"expires_at": _ts(created + timedelta(hours=ttl_hours)),
|
||||
},
|
||||
"lock_provenance": issue_lock_provenance.build_sanctioned_lock_provenance(
|
||||
tool="gitea_lock_issue", claimant=claimant
|
||||
),
|
||||
}
|
||||
path = self._lock_path()
|
||||
record["lock_file_path"] = path
|
||||
issue_lock_store.save_lock_file(path, record)
|
||||
return record
|
||||
|
||||
|
||||
class TestLockIssueMintsTheLifecycle(_HeartbeatMcpBase):
|
||||
"""Durable lock creation and read-back through the real tool."""
|
||||
|
||||
def test_native_lock_writes_the_marker_and_a_task_session_id(self):
|
||||
result = self.run_lock_issue()
|
||||
self.assertTrue(result["success"], result)
|
||||
|
||||
written = issue_lock_store.read_lock_file(result["lock_file_path"])
|
||||
lease = written["work_lease"]
|
||||
self.assertEqual(
|
||||
lease["lifecycle_version"], lease_policy.LIFECYCLE_HEARTBEAT_V1
|
||||
)
|
||||
self.assertTrue(lease["task_session_id"])
|
||||
self.assertFalse(issue_lock_store.is_legacy_lease(written))
|
||||
# AC-N1: the ownership key is not the daemon pid, which is recorded
|
||||
# separately as evidence.
|
||||
self.assertNotIn(str(written["session_pid"]), lease["task_session_id"])
|
||||
self.assertEqual(written["session_pid"], os.getpid())
|
||||
|
||||
def test_native_lease_uses_the_policy_window_not_four_hours(self):
|
||||
result = self.run_lock_issue()
|
||||
lease = result["work_lease"]
|
||||
created = datetime.fromisoformat(lease["created_at"].replace("Z", "+00:00"))
|
||||
expires = datetime.fromisoformat(lease["expires_at"].replace("Z", "+00:00"))
|
||||
policy = lease_policy.policy_for(lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK)
|
||||
self.assertEqual(
|
||||
(expires - created).total_seconds() / 60.0, policy.initial_ttl_minutes
|
||||
)
|
||||
|
||||
def test_freshness_of_a_new_native_lock_is_live(self):
|
||||
result = self.run_lock_issue()
|
||||
self.assertEqual(
|
||||
result["lock_freshness"]["status"], issue_lock_store.STATUS_LIVE
|
||||
)
|
||||
self.assertTrue(result["lock_freshness"]["live"])
|
||||
|
||||
|
||||
class TestHeartbeatThroughTheTool(_HeartbeatMcpBase):
|
||||
def _lock_and_session(self):
|
||||
result = self.run_lock_issue()
|
||||
self.assertTrue(result["success"], result)
|
||||
return result, result["work_lease"]["task_session_id"]
|
||||
|
||||
def test_heartbeat_slides_the_lease_and_advances_the_generation(self):
|
||||
locked, session = self._lock_and_session()
|
||||
before = issue_lock_store.read_lock_file(locked["lock_file_path"])
|
||||
|
||||
beat = self.run_heartbeat(task_session_id=session)
|
||||
|
||||
self.assertTrue(beat["success"], beat)
|
||||
self.assertEqual(beat["operation"], "heartbeat")
|
||||
after = issue_lock_store.read_lock_file(locked["lock_file_path"])
|
||||
self.assertGreater(
|
||||
issue_lock_store.lock_generation(after),
|
||||
issue_lock_store.lock_generation(before),
|
||||
)
|
||||
self.assertGreaterEqual(
|
||||
after["work_lease"]["expires_at"], before["work_lease"]["expires_at"]
|
||||
)
|
||||
self.assertEqual(after["work_lease"]["heartbeat_count"], 2)
|
||||
|
||||
def test_heartbeat_evidence_survives_the_downstream_mutation_gate(self):
|
||||
"""The #499 F2 lesson, applied.
|
||||
|
||||
A sanction that is computed and then discarded downstream is worthless.
|
||||
After a heartbeat the lock must still satisfy the gate every author
|
||||
mutation runs through.
|
||||
"""
|
||||
locked, session = self._lock_and_session()
|
||||
self.run_heartbeat(task_session_id=session)
|
||||
|
||||
written = issue_lock_store.read_lock_file(locked["lock_file_path"])
|
||||
verdict = issue_lock_store.verify_lock_for_mutation(
|
||||
written,
|
||||
issue_number=ISSUE,
|
||||
branch_name=BRANCH,
|
||||
worktree_path=self.worktree,
|
||||
)
|
||||
self.assertTrue(verdict["proven"], verdict)
|
||||
self.assertFalse(verdict["block"])
|
||||
|
||||
def _duplicate_gate(self, *, open_prs, branches):
|
||||
env = self._tool_env()
|
||||
with patch("mcp_server.get_auth_header", return_value="token x"), patch(
|
||||
"mcp_server.issue_duplicate_context_fetcher",
|
||||
side_effect=lambda h, o, r, auth, issue_number: (
|
||||
list(open_prs),
|
||||
list(branches),
|
||||
{"status": "not_claimed"},
|
||||
),
|
||||
), patch.dict(os.environ, env, clear=True):
|
||||
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||
return mcp_server.gitea_assess_work_issue_duplicate(
|
||||
issue_number=ISSUE, branch_name=BRANCH, remote="prgs"
|
||||
)
|
||||
|
||||
def test_heartbeat_does_not_change_the_duplicate_gate_verdict(self):
|
||||
"""The gate must be invariant under heartbeating.
|
||||
|
||||
The point is not that the gate passes — with a linked open PR at the
|
||||
lock phase it correctly blocks (#400), heartbeat or not. The property
|
||||
that matters is that sliding a lease neither loosens the gate nor
|
||||
corrupts the lock state it reads: the verdict before and after a
|
||||
heartbeat must be identical, for both the clear and the blocking shape.
|
||||
"""
|
||||
_, session = self._lock_and_session()
|
||||
linked = [{"number": 4242, "head": {"ref": BRANCH, "sha": self.head_sha}}]
|
||||
|
||||
clear_before = self._duplicate_gate(open_prs=[], branches=[])
|
||||
blocked_before = self._duplicate_gate(open_prs=linked, branches=[BRANCH])
|
||||
|
||||
self.assertTrue(self.run_heartbeat(task_session_id=session)["success"])
|
||||
|
||||
clear_after = self._duplicate_gate(open_prs=[], branches=[])
|
||||
blocked_after = self._duplicate_gate(open_prs=linked, branches=[BRANCH])
|
||||
|
||||
self.assertEqual(clear_before["outcome"], clear_after["outcome"])
|
||||
self.assertFalse(clear_after["block"])
|
||||
self.assertEqual(blocked_before["outcome"], blocked_after["outcome"])
|
||||
self.assertTrue(blocked_after["block"])
|
||||
self.assertEqual(blocked_after["linked_open_pr"], 4242)
|
||||
|
||||
def test_foreign_session_id_is_refused_through_the_tool(self):
|
||||
self._lock_and_session()
|
||||
beat = self.run_heartbeat(task_session_id="author_issue_work-ffffffffffffffff")
|
||||
self.assertFalse(beat["success"])
|
||||
self.assertIn("task_session_id does not match", " ".join(beat["reasons"]))
|
||||
|
||||
def test_stale_generation_is_refused_through_the_tool(self):
|
||||
locked, session = self._lock_and_session()
|
||||
current = issue_lock_store.lock_generation(
|
||||
issue_lock_store.read_lock_file(locked["lock_file_path"])
|
||||
)
|
||||
beat = self.run_heartbeat(
|
||||
task_session_id=session, expected_generation=current + 5
|
||||
)
|
||||
self.assertFalse(beat["success"])
|
||||
self.assertIn("generation changed", beat["reasons"][0])
|
||||
|
||||
def test_foreign_claimant_is_refused_through_the_tool(self):
|
||||
_, session = self._lock_and_session()
|
||||
beat = self.run_heartbeat(task_session_id=session, identity="someone-else")
|
||||
self.assertFalse(beat["success"])
|
||||
|
||||
def test_heartbeat_cannot_acquire_a_missing_lock(self):
|
||||
beat = self.run_heartbeat(task_session_id="author_issue_work-000000000000")
|
||||
self.assertFalse(beat["success"])
|
||||
self.assertIn("no durable lock", beat["reasons"][0])
|
||||
|
||||
def test_alive_pid_alone_does_not_keep_a_lease_live_through_the_tool(self):
|
||||
"""PID-only refusal, end to end.
|
||||
|
||||
The recorded pid is this live process. The lock is aged past its grace
|
||||
with no heartbeat, so the tool must refuse to slide it and the durable
|
||||
record must classify as a missed heartbeat rather than as live.
|
||||
"""
|
||||
locked, session = self._lock_and_session()
|
||||
record = issue_lock_store.read_lock_file(locked["lock_file_path"])
|
||||
record["work_lease"]["last_heartbeat_at"] = _ts(
|
||||
datetime.now(timezone.utc) - timedelta(minutes=30)
|
||||
)
|
||||
record["work_lease"]["expires_at"] = _ts(
|
||||
datetime.now(timezone.utc) + timedelta(hours=2)
|
||||
)
|
||||
issue_lock_store.save_lock_file(locked["lock_file_path"], record)
|
||||
|
||||
self.assertTrue(issue_lock_store.is_process_alive(record["session_pid"]))
|
||||
fresh = issue_lock_store.assess_lock_freshness(record)
|
||||
self.assertEqual(
|
||||
fresh["status"], issue_lock_store.STATUS_STALE_MISSED_HEARTBEAT
|
||||
)
|
||||
self.assertTrue(fresh["pid_alive"])
|
||||
|
||||
beat = self.run_heartbeat(task_session_id=session)
|
||||
self.assertFalse(beat["success"])
|
||||
self.assertIn("reclaimed", " ".join(beat["reasons"]))
|
||||
|
||||
|
||||
class TestLegacyLocksThroughTheTool(_HeartbeatMcpBase):
|
||||
"""AC-N8 end to end: protected on deployment, and rebindable."""
|
||||
|
||||
def test_legacy_lock_stays_protected_after_deployment(self):
|
||||
record = self.write_legacy_lock(hours_old=3.0, ttl_hours=4.0)
|
||||
fresh = issue_lock_store.assess_lock_freshness(record)
|
||||
self.assertEqual(fresh["status"], issue_lock_store.STATUS_LIVE)
|
||||
self.assertTrue(fresh["legacy_lease"])
|
||||
self.assertTrue(fresh["legacy_expiry_preserved"])
|
||||
# It had never heartbeated, so under the new grace alone it would be
|
||||
# long gone; the preserved absolute expiry is what protects it.
|
||||
self.assertEqual(
|
||||
record["work_lease"]["created_at"],
|
||||
record["work_lease"]["last_heartbeat_at"],
|
||||
)
|
||||
|
||||
def test_tool_rebinds_a_legacy_lock_and_mints_a_first_heartbeat(self):
|
||||
self.write_legacy_lock(hours_old=3.0, ttl_hours=4.0)
|
||||
|
||||
result = self.run_heartbeat(task_session_id=None)
|
||||
|
||||
self.assertTrue(result["success"], result)
|
||||
self.assertEqual(result["operation"], "legacy_rebind")
|
||||
self.assertTrue(result["task_session_id"])
|
||||
|
||||
written = issue_lock_store.read_lock_file(self._lock_path())
|
||||
lease = written["work_lease"]
|
||||
self.assertEqual(
|
||||
lease["lifecycle_version"], lease_policy.LIFECYCLE_HEARTBEAT_V1
|
||||
)
|
||||
self.assertEqual(lease["heartbeat_count"], 1)
|
||||
self.assertNotEqual(
|
||||
lease["created_at"],
|
||||
written["legacy_rebind"]["legacy_origin"]["created_at"],
|
||||
)
|
||||
self.assertFalse(issue_lock_store.is_legacy_lease(written))
|
||||
|
||||
def test_rebound_lock_then_heartbeats_through_the_tool(self):
|
||||
self.write_legacy_lock(hours_old=3.0, ttl_hours=4.0)
|
||||
rebound = self.run_heartbeat(task_session_id=None)
|
||||
beat = self.run_heartbeat(task_session_id=rebound["task_session_id"])
|
||||
self.assertTrue(beat["success"], beat)
|
||||
self.assertEqual(beat["operation"], "heartbeat")
|
||||
self.assertEqual(beat["heartbeat_count"], 2)
|
||||
|
||||
def test_rebind_refuses_a_foreign_owner_through_the_tool(self):
|
||||
self.write_legacy_lock(hours_old=3.0, ttl_hours=4.0)
|
||||
result = self.run_heartbeat(task_session_id=None, identity="someone-else")
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["operation"], "legacy_rebind")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,594 @@
|
||||
"""Central lease policy and load-bearing heartbeat freshness (#790 Slice A).
|
||||
|
||||
Before this slice, ``issue_lock_store.assess_lock_freshness`` parsed
|
||||
``last_heartbeat_at`` and then never consulted it: liveness was decided by an
|
||||
absolute four-hour ``expires_at`` and by PID liveness. Because the recorded PID
|
||||
is the long-lived MCP daemon rather than the authoring task, an abandoned claim
|
||||
stayed "live" for the full four hours, and a claim whose work had already landed
|
||||
blocked reconciliation for just as long (Issue #787 / PR #789, and again Issue
|
||||
#760 / PR #791).
|
||||
|
||||
These tests pin the corrected semantics, including the two asymmetries that are
|
||||
easy to lose in a refactor:
|
||||
|
||||
* an **alive** PID must never make anything live (AC-N2), while
|
||||
* a **dead** PID must still mark a lease stale, because #753 dead-session
|
||||
recovery keys on exactly that classification.
|
||||
|
||||
Durable-state helpers here write real lock files through the real flock path;
|
||||
they are not mocks of the store.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
import issue_lock_store # noqa: E402
|
||||
import lease_policy # noqa: E402
|
||||
import pr_work_lease # noqa: E402
|
||||
import reviewer_pr_lease # noqa: E402
|
||||
|
||||
ISSUE = 9790
|
||||
BRANCH = f"fix/issue-{ISSUE}-heartbeat"
|
||||
IDENTITY = "example-user"
|
||||
PROFILE = "test-author-prgs"
|
||||
ORG = "Example-Org"
|
||||
REPO = "Example-Repo"
|
||||
REMOTE = "prgs"
|
||||
DEAD_PID = 2**22 # far above any live pid on a test host
|
||||
|
||||
|
||||
def _ts(moment: datetime) -> str:
|
||||
return (
|
||||
moment.astimezone(timezone.utc)
|
||||
.replace(microsecond=0)
|
||||
.isoformat()
|
||||
.replace("+00:00", "Z")
|
||||
)
|
||||
|
||||
|
||||
class _LockFixture(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.lock_dir = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self.lock_dir.cleanup)
|
||||
self.now = datetime.now(timezone.utc)
|
||||
self.worktree = os.path.realpath(tempfile.mkdtemp(prefix="issue790-"))
|
||||
self.addCleanup(patch.stopall)
|
||||
|
||||
def _path(self):
|
||||
return issue_lock_store.lock_file_path(
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
issue_number=ISSUE,
|
||||
lock_dir=self.lock_dir.name,
|
||||
)
|
||||
|
||||
def write_lock(
|
||||
self,
|
||||
*,
|
||||
lifecycle: str | None = lease_policy.LIFECYCLE_HEARTBEAT_V1,
|
||||
created_delta: timedelta = timedelta(minutes=1),
|
||||
heartbeat_delta: timedelta = timedelta(minutes=1),
|
||||
expires_delta: timedelta = timedelta(minutes=9),
|
||||
pid: int | None = None,
|
||||
task_session_id: str | None = "author_issue_work-aaaabbbbccccdddd",
|
||||
generation: int = 1,
|
||||
identity: str = IDENTITY,
|
||||
profile: str = PROFILE,
|
||||
branch: str = BRANCH,
|
||||
worktree: str | None = None,
|
||||
) -> dict:
|
||||
"""Write a real durable lock and return the record.
|
||||
|
||||
Deltas are relative to ``self.now``; ``expires_delta`` is added, the
|
||||
others subtracted, so "in the past" reads naturally at each call site.
|
||||
"""
|
||||
lease: dict = {
|
||||
"operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE,
|
||||
"issue_number": ISSUE,
|
||||
"pr_number": None,
|
||||
"branch": branch,
|
||||
"worktree_path": worktree or self.worktree,
|
||||
"claimant": {"username": identity, "profile": profile},
|
||||
"created_at": _ts(self.now - created_delta),
|
||||
"last_heartbeat_at": _ts(self.now - heartbeat_delta),
|
||||
"expires_at": _ts(self.now + expires_delta),
|
||||
}
|
||||
if lifecycle is not None:
|
||||
lease["lifecycle_version"] = lifecycle
|
||||
if task_session_id is not None:
|
||||
lease["task_session_id"] = task_session_id
|
||||
pid_value = os.getpid() if pid is None else pid
|
||||
record = {
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": branch,
|
||||
"remote": REMOTE,
|
||||
"org": ORG,
|
||||
"repo": REPO,
|
||||
"worktree_path": worktree or self.worktree,
|
||||
"session_pid": pid_value,
|
||||
"pid": pid_value,
|
||||
"lock_generation": generation,
|
||||
"work_lease": lease,
|
||||
}
|
||||
path = self._path()
|
||||
record["lock_file_path"] = path
|
||||
issue_lock_store.save_lock_file(path, record)
|
||||
return record
|
||||
|
||||
|
||||
class TestPolicyIsTheSingleSource(unittest.TestCase):
|
||||
"""AC-N7: one authoritative configuration source for every duration."""
|
||||
|
||||
def test_author_policy_carries_the_agreed_values(self):
|
||||
policy = lease_policy.policy_for(lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK)
|
||||
self.assertEqual(policy.initial_ttl_minutes, 10.0)
|
||||
self.assertEqual(policy.heartbeat_cadence_minutes, 2.0)
|
||||
self.assertEqual(policy.stale_warning_minutes, 5.0)
|
||||
self.assertEqual(policy.missed_heartbeat_grace_minutes, 10.0)
|
||||
self.assertEqual(policy.absolute_cap_hours, 8.0)
|
||||
self.assertEqual(policy.recovery_grace_minutes, 10.0)
|
||||
self.assertEqual(policy.terminal_race_drain_minutes, 2.0)
|
||||
self.assertTrue(policy.terminal_retirement_eligible)
|
||||
self.assertTrue(policy.heartbeat_lifecycle_active)
|
||||
|
||||
def test_the_four_hour_author_ttl_literal_is_gone(self):
|
||||
"""The duplicated literal AC-N7 exists to remove."""
|
||||
self.assertFalse(hasattr(issue_lock_store, "WORK_LEASE_TTL_HOURS"))
|
||||
import gitea_mcp_server
|
||||
|
||||
self.assertFalse(hasattr(gitea_mcp_server, "WORK_LEASE_TTL_HOURS"))
|
||||
|
||||
def test_declared_reviewer_values_match_the_module_still_using_them(self):
|
||||
"""Slice A declares reviewer/merger numbers without rewiring them.
|
||||
|
||||
Recording a value in two places is only safe if drift is detectable, so
|
||||
this asserts the declaration still equals the constants #747 owns. When
|
||||
Slice C migrates those call sites, this test becomes the proof the
|
||||
migration changed nothing.
|
||||
"""
|
||||
policy = lease_policy.policy_for(lease_policy.TASK_CLASS_REVIEWER_PR)
|
||||
self.assertEqual(
|
||||
policy.initial_ttl_minutes, float(reviewer_pr_lease.LEASE_TTL_MINUTES)
|
||||
)
|
||||
self.assertEqual(
|
||||
policy.stale_warning_minutes,
|
||||
float(reviewer_pr_lease.STALE_WARNING_MINUTES),
|
||||
)
|
||||
self.assertFalse(policy.heartbeat_lifecycle_active)
|
||||
|
||||
def test_declared_conflict_fix_value_matches_its_module(self):
|
||||
policy = lease_policy.policy_for(lease_policy.TASK_CLASS_CONFLICT_FIX)
|
||||
self.assertEqual(
|
||||
policy.initial_ttl_minutes,
|
||||
float(pr_work_lease.DEFAULT_CONFLICT_FIX_TTL_MINUTES),
|
||||
)
|
||||
self.assertFalse(policy.heartbeat_lifecycle_active)
|
||||
|
||||
def test_environment_override_applies(self):
|
||||
var = lease_policy.env_var_name(
|
||||
lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK, "initial_ttl_minutes"
|
||||
)
|
||||
with patch.dict(os.environ, {var: "7"}):
|
||||
self.assertEqual(
|
||||
lease_policy.policy_for(
|
||||
lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK
|
||||
).initial_ttl_minutes,
|
||||
7.0,
|
||||
)
|
||||
|
||||
def test_unusable_override_falls_back_instead_of_minting_a_zero_lease(self):
|
||||
"""A typo must not make every claim instantly reclaimable."""
|
||||
var = lease_policy.env_var_name(
|
||||
lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK, "initial_ttl_minutes"
|
||||
)
|
||||
for bad in ("0", "-5", "not-a-number", " "):
|
||||
with self.subTest(value=bad), patch.dict(os.environ, {var: bad}):
|
||||
self.assertEqual(
|
||||
lease_policy.policy_for(
|
||||
lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK
|
||||
).initial_ttl_minutes,
|
||||
10.0,
|
||||
)
|
||||
|
||||
def test_unknown_task_class_does_not_raise(self):
|
||||
policy = lease_policy.policy_for("something-new")
|
||||
self.assertEqual(policy.task_class, lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK)
|
||||
|
||||
|
||||
class TestLifecycleDiscrimination(_LockFixture):
|
||||
"""AC-N8: the marker, never a timestamp, decides legacy vs heartbeat."""
|
||||
|
||||
def test_missing_marker_reads_as_legacy(self):
|
||||
record = self.write_lock(lifecycle=None)
|
||||
self.assertTrue(issue_lock_store.is_legacy_lease(record))
|
||||
self.assertEqual(
|
||||
issue_lock_store.lease_lifecycle_version(record),
|
||||
lease_policy.LIFECYCLE_LEGACY,
|
||||
)
|
||||
|
||||
def test_marker_present_reads_as_heartbeat_lifecycle(self):
|
||||
record = self.write_lock()
|
||||
self.assertFalse(issue_lock_store.is_legacy_lease(record))
|
||||
|
||||
def test_equal_created_and_heartbeat_never_implies_a_fresh_heartbeat(self):
|
||||
"""The exact inversion AC-N8 forbids.
|
||||
|
||||
A legacy lock has ``last_heartbeat_at == created_at`` forever because
|
||||
nothing ever advanced it. Reading that equality as "recently
|
||||
heartbeated" would classify every never-heartbeated lock as fresh.
|
||||
"""
|
||||
legacy = self.write_lock(
|
||||
lifecycle=None,
|
||||
created_delta=timedelta(hours=3),
|
||||
heartbeat_delta=timedelta(hours=3),
|
||||
)
|
||||
lease = legacy["work_lease"]
|
||||
self.assertEqual(lease["created_at"], lease["last_heartbeat_at"])
|
||||
self.assertTrue(issue_lock_store.is_legacy_lease(legacy))
|
||||
|
||||
# A brand-new heartbeat lease has them equal too, so the equality
|
||||
# carries no information in either direction.
|
||||
fresh = self.write_lock(
|
||||
created_delta=timedelta(seconds=0), heartbeat_delta=timedelta(seconds=0)
|
||||
)
|
||||
self.assertEqual(
|
||||
fresh["work_lease"]["created_at"],
|
||||
fresh["work_lease"]["last_heartbeat_at"],
|
||||
)
|
||||
self.assertFalse(issue_lock_store.is_legacy_lease(fresh))
|
||||
|
||||
def test_minted_session_id_contains_no_pid(self):
|
||||
"""AC-N1: the ownership key must not be derived from the daemon pid."""
|
||||
minted = issue_lock_store.mint_task_session_id()
|
||||
self.assertNotIn(str(os.getpid()), minted)
|
||||
self.assertNotEqual(minted, issue_lock_store.mint_task_session_id())
|
||||
|
||||
|
||||
class TestFreshnessIsHeartbeatDriven(_LockFixture):
|
||||
"""AC-N2 and the new bands."""
|
||||
|
||||
def test_fresh_heartbeat_is_live(self):
|
||||
record = self.write_lock(heartbeat_delta=timedelta(minutes=1))
|
||||
fresh = issue_lock_store.assess_lock_freshness(record, now=self.now)
|
||||
self.assertEqual(fresh["status"], issue_lock_store.STATUS_LIVE)
|
||||
self.assertTrue(fresh["live"])
|
||||
self.assertFalse(fresh["heartbeat_warning"])
|
||||
|
||||
def test_heartbeat_past_warning_is_still_live_but_flagged(self):
|
||||
record = self.write_lock(heartbeat_delta=timedelta(minutes=6))
|
||||
fresh = issue_lock_store.assess_lock_freshness(record, now=self.now)
|
||||
self.assertEqual(fresh["status"], issue_lock_store.STATUS_LIVE)
|
||||
self.assertTrue(fresh["heartbeat_warning"])
|
||||
|
||||
def test_missed_heartbeat_past_grace_is_classified_explicitly(self):
|
||||
record = self.write_lock(
|
||||
heartbeat_delta=timedelta(minutes=11),
|
||||
expires_delta=timedelta(minutes=30),
|
||||
)
|
||||
fresh = issue_lock_store.assess_lock_freshness(record, now=self.now)
|
||||
self.assertEqual(
|
||||
fresh["status"], issue_lock_store.STATUS_STALE_MISSED_HEARTBEAT
|
||||
)
|
||||
self.assertFalse(fresh["live"])
|
||||
self.assertTrue(fresh["stale"])
|
||||
|
||||
def test_alive_pid_never_establishes_freshness(self):
|
||||
"""The defect in one assertion.
|
||||
|
||||
The recorded PID is this very process, so it is unambiguously alive —
|
||||
and the lease is still not live, because the task stopped heartbeating.
|
||||
"""
|
||||
record = self.write_lock(
|
||||
pid=os.getpid(),
|
||||
heartbeat_delta=timedelta(hours=4),
|
||||
expires_delta=timedelta(hours=4),
|
||||
)
|
||||
fresh = issue_lock_store.assess_lock_freshness(record, now=self.now)
|
||||
self.assertTrue(fresh["pid_alive"])
|
||||
self.assertFalse(fresh["live"])
|
||||
self.assertEqual(
|
||||
fresh["status"], issue_lock_store.STATUS_STALE_MISSED_HEARTBEAT
|
||||
)
|
||||
|
||||
def test_dead_pid_still_marks_stale_for_issue_753(self):
|
||||
"""The opposite asymmetry: dead-PID corroboration is preserved."""
|
||||
record = self.write_lock(pid=DEAD_PID, heartbeat_delta=timedelta(minutes=1))
|
||||
fresh = issue_lock_store.assess_lock_freshness(record, now=self.now)
|
||||
self.assertEqual(fresh["status"], issue_lock_store.STATUS_STALE)
|
||||
self.assertFalse(fresh["live"])
|
||||
self.assertIn("not alive", fresh["reason"])
|
||||
|
||||
def test_absolute_cap_requires_readoption(self):
|
||||
record = self.write_lock(
|
||||
created_delta=timedelta(hours=9), heartbeat_delta=timedelta(minutes=1)
|
||||
)
|
||||
fresh = issue_lock_store.assess_lock_freshness(record, now=self.now)
|
||||
self.assertEqual(fresh["status"], issue_lock_store.STATUS_STALE_ABSOLUTE_CAP)
|
||||
self.assertIn("re-adoption", fresh["reason"])
|
||||
|
||||
def test_heartbeat_lifecycle_without_a_heartbeat_fails_closed(self):
|
||||
record = self.write_lock()
|
||||
del record["work_lease"]["last_heartbeat_at"]
|
||||
issue_lock_store.save_lock_file(self._path(), record)
|
||||
fresh = issue_lock_store.assess_lock_freshness(record, now=self.now)
|
||||
self.assertEqual(
|
||||
fresh["status"], issue_lock_store.STATUS_STALE_MISSED_HEARTBEAT
|
||||
)
|
||||
self.assertIn("fail closed", fresh["reason"])
|
||||
|
||||
def test_absent_lock(self):
|
||||
fresh = issue_lock_store.assess_lock_freshness(None)
|
||||
self.assertEqual(fresh["status"], issue_lock_store.STATUS_ABSENT)
|
||||
self.assertFalse(fresh["stale"])
|
||||
|
||||
|
||||
class TestLegacyLocksStayProtected(_LockFixture):
|
||||
"""AC-N8: deployment must not retroactively shorten an existing claim."""
|
||||
|
||||
def test_legacy_lock_with_a_stale_heartbeat_remains_live(self):
|
||||
"""The deployment-safety case.
|
||||
|
||||
A four-hour legacy lease minted three hours ago has not heartbeated
|
||||
once. Under the new grace it would be long gone; under its preserved
|
||||
absolute expiry it is still live, and must stay that way.
|
||||
"""
|
||||
record = self.write_lock(
|
||||
lifecycle=None,
|
||||
created_delta=timedelta(hours=3),
|
||||
heartbeat_delta=timedelta(hours=3),
|
||||
expires_delta=timedelta(hours=1),
|
||||
)
|
||||
fresh = issue_lock_store.assess_lock_freshness(record, now=self.now)
|
||||
self.assertEqual(fresh["status"], issue_lock_store.STATUS_LIVE)
|
||||
self.assertTrue(fresh["live"])
|
||||
self.assertTrue(fresh["legacy_lease"])
|
||||
self.assertTrue(fresh["legacy_expiry_preserved"])
|
||||
|
||||
def test_legacy_lock_past_its_absolute_expiry_is_expired_as_before(self):
|
||||
record = self.write_lock(
|
||||
lifecycle=None,
|
||||
created_delta=timedelta(hours=5),
|
||||
heartbeat_delta=timedelta(hours=5),
|
||||
expires_delta=timedelta(hours=-1),
|
||||
)
|
||||
fresh = issue_lock_store.assess_lock_freshness(record, now=self.now)
|
||||
self.assertEqual(fresh["status"], issue_lock_store.STATUS_EXPIRED)
|
||||
|
||||
def test_legacy_lock_is_never_reclaimed_by_the_heartbeat_band(self):
|
||||
record = self.write_lock(
|
||||
lifecycle=None,
|
||||
created_delta=timedelta(hours=3),
|
||||
heartbeat_delta=timedelta(hours=3),
|
||||
expires_delta=timedelta(hours=1),
|
||||
)
|
||||
reclaim = issue_lock_store.assess_expired_lock_reclaim(record, now=self.now)
|
||||
self.assertFalse(reclaim["reclaim_allowed"])
|
||||
|
||||
|
||||
class TestReclaimAfterMissedHeartbeat(_LockFixture):
|
||||
def test_missed_heartbeat_makes_ownership_reclaimable(self):
|
||||
record = self.write_lock(
|
||||
pid=os.getpid(),
|
||||
heartbeat_delta=timedelta(minutes=15),
|
||||
expires_delta=timedelta(hours=3),
|
||||
)
|
||||
reclaim = issue_lock_store.assess_expired_lock_reclaim(record, now=self.now)
|
||||
self.assertTrue(reclaim["reclaim_allowed"])
|
||||
self.assertIn("stale_missed_heartbeat", reclaim["reasons"][0])
|
||||
|
||||
def test_live_lease_is_never_reclaimable(self):
|
||||
record = self.write_lock(heartbeat_delta=timedelta(minutes=1))
|
||||
reclaim = issue_lock_store.assess_expired_lock_reclaim(record, now=self.now)
|
||||
self.assertFalse(reclaim["reclaim_allowed"])
|
||||
|
||||
def test_dead_pid_reclaim_path_is_unchanged(self):
|
||||
"""#753 must keep working through its original conditions."""
|
||||
record = self.write_lock(pid=DEAD_PID, heartbeat_delta=timedelta(minutes=1))
|
||||
reclaim = issue_lock_store.assess_expired_lock_reclaim(record, now=self.now)
|
||||
self.assertTrue(reclaim["reclaim_allowed"])
|
||||
self.assertTrue(reclaim["owner_pid_dead"])
|
||||
|
||||
|
||||
class TestHeartbeatWriter(_LockFixture):
|
||||
"""A4: flock + CAS + exact verification, and no revival path."""
|
||||
|
||||
def _heartbeat(self, **kwargs):
|
||||
params = {
|
||||
"remote": REMOTE,
|
||||
"org": ORG,
|
||||
"repo": REPO,
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": BRANCH,
|
||||
"worktree_path": self.worktree,
|
||||
"identity": IDENTITY,
|
||||
"profile": PROFILE,
|
||||
"task_session_id": "author_issue_work-aaaabbbbccccdddd",
|
||||
"lock_dir": self.lock_dir.name,
|
||||
"now": self.now,
|
||||
}
|
||||
params.update(kwargs)
|
||||
return issue_lock_store.heartbeat_session_lock(**params)
|
||||
|
||||
def test_heartbeat_slides_expiry_and_advances_generation(self):
|
||||
self.write_lock(heartbeat_delta=timedelta(minutes=4), generation=5)
|
||||
result = self._heartbeat()
|
||||
self.assertTrue(result["success"], result)
|
||||
self.assertEqual(result["prior_generation"], 5)
|
||||
self.assertEqual(result["lock_generation"], 6)
|
||||
self.assertEqual(result["heartbeat_count"], 1)
|
||||
self.assertEqual(result["last_heartbeat_at"], _ts(self.now))
|
||||
self.assertEqual(result["expires_at"], _ts(self.now + timedelta(minutes=10)))
|
||||
self.assertTrue(result["freshness"]["live"])
|
||||
|
||||
def test_heartbeat_is_durable_and_repeatable(self):
|
||||
self.write_lock(heartbeat_delta=timedelta(minutes=4))
|
||||
self._heartbeat()
|
||||
second = self._heartbeat(now=self.now + timedelta(minutes=1))
|
||||
self.assertTrue(second["success"], second)
|
||||
self.assertEqual(second["heartbeat_count"], 2)
|
||||
written = issue_lock_store.read_lock_file(self._path())
|
||||
self.assertEqual(written["work_lease"]["heartbeat_count"], 2)
|
||||
|
||||
def test_stale_generation_is_refused(self):
|
||||
self.write_lock(generation=5)
|
||||
result = self._heartbeat(expected_generation=4)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertIn("generation changed", result["reasons"][0])
|
||||
|
||||
def test_foreign_session_is_refused(self):
|
||||
self.write_lock()
|
||||
result = self._heartbeat(task_session_id="author_issue_work-ffffffffffffffff")
|
||||
self.assertFalse(result["success"])
|
||||
self.assertIn("task_session_id does not match", " ".join(result["reasons"]))
|
||||
|
||||
def test_missing_session_id_is_refused(self):
|
||||
self.write_lock()
|
||||
result = self._heartbeat(task_session_id="")
|
||||
self.assertFalse(result["success"])
|
||||
|
||||
def test_foreign_claimant_is_refused(self):
|
||||
self.write_lock()
|
||||
for field, value in (
|
||||
("identity", "someone-else"),
|
||||
("profile", "other-profile"),
|
||||
):
|
||||
with self.subTest(field=field):
|
||||
result = self._heartbeat(**{field: value})
|
||||
self.assertFalse(result["success"])
|
||||
|
||||
def test_branch_and_worktree_mismatch_are_refused(self):
|
||||
self.write_lock()
|
||||
wrong_branch = self._heartbeat(branch_name=f"fix/issue-{ISSUE}-other")
|
||||
self.assertFalse(wrong_branch["success"])
|
||||
wrong_worktree = self._heartbeat(worktree_path="/tmp/not-the-worktree")
|
||||
self.assertFalse(wrong_worktree["success"])
|
||||
|
||||
def test_lapsed_lease_cannot_be_heartbeated_back_to_life(self):
|
||||
"""No revival path (A4).
|
||||
|
||||
A session that stopped proving liveness must reclaim under a fresh
|
||||
generation, not restore ownership retroactively.
|
||||
"""
|
||||
self.write_lock(
|
||||
heartbeat_delta=timedelta(minutes=30), expires_delta=timedelta(hours=1)
|
||||
)
|
||||
result = self._heartbeat()
|
||||
self.assertFalse(result["success"])
|
||||
self.assertIn("reclaimed", " ".join(result["reasons"]))
|
||||
|
||||
def test_absent_lock_cannot_be_created_by_heartbeat(self):
|
||||
result = self._heartbeat()
|
||||
self.assertFalse(result["success"])
|
||||
self.assertIn("no durable lock", result["reasons"][0])
|
||||
|
||||
def test_legacy_lock_is_refused_until_rebound(self):
|
||||
self.write_lock(lifecycle=None)
|
||||
result = self._heartbeat()
|
||||
self.assertFalse(result["success"])
|
||||
self.assertTrue(result["legacy_lease"])
|
||||
self.assertIn("rebound", " ".join(result["reasons"]))
|
||||
|
||||
|
||||
class TestLegacyRebind(_LockFixture):
|
||||
"""AC-N8 exit route: canonical exact-owner rebinding."""
|
||||
|
||||
def _rebind(self, **kwargs):
|
||||
params = {
|
||||
"remote": REMOTE,
|
||||
"org": ORG,
|
||||
"repo": REPO,
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": BRANCH,
|
||||
"worktree_path": self.worktree,
|
||||
"identity": IDENTITY,
|
||||
"profile": PROFILE,
|
||||
"lock_dir": self.lock_dir.name,
|
||||
"now": self.now,
|
||||
}
|
||||
params.update(kwargs)
|
||||
return issue_lock_store.rebind_legacy_lock(**params)
|
||||
|
||||
def test_rebind_mints_a_session_and_a_genuine_first_heartbeat(self):
|
||||
self.write_lock(
|
||||
lifecycle=None,
|
||||
created_delta=timedelta(hours=3),
|
||||
heartbeat_delta=timedelta(hours=3),
|
||||
expires_delta=timedelta(hours=1),
|
||||
generation=2,
|
||||
)
|
||||
result = self._rebind()
|
||||
self.assertTrue(result["success"], result)
|
||||
self.assertTrue(result["task_session_id"])
|
||||
self.assertEqual(result["lock_generation"], 3)
|
||||
|
||||
written = issue_lock_store.read_lock_file(self._path())
|
||||
lease = written["work_lease"]
|
||||
self.assertEqual(
|
||||
lease["lifecycle_version"], lease_policy.LIFECYCLE_HEARTBEAT_V1
|
||||
)
|
||||
self.assertEqual(lease["last_heartbeat_at"], _ts(self.now))
|
||||
self.assertEqual(lease["expires_at"], _ts(self.now + timedelta(minutes=10)))
|
||||
self.assertFalse(issue_lock_store.is_legacy_lease(written))
|
||||
# The original claim is preserved for audit rather than overwritten.
|
||||
origin = written["legacy_rebind"]["legacy_origin"]
|
||||
self.assertTrue(origin["created_at"])
|
||||
self.assertEqual(origin["lifecycle"], lease_policy.LIFECYCLE_LEGACY)
|
||||
|
||||
def test_rebound_lock_can_then_heartbeat(self):
|
||||
self.write_lock(
|
||||
lifecycle=None,
|
||||
created_delta=timedelta(hours=3),
|
||||
heartbeat_delta=timedelta(hours=3),
|
||||
expires_delta=timedelta(hours=1),
|
||||
)
|
||||
rebound = self._rebind()
|
||||
beat = issue_lock_store.heartbeat_session_lock(
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
issue_number=ISSUE,
|
||||
branch_name=BRANCH,
|
||||
worktree_path=self.worktree,
|
||||
identity=IDENTITY,
|
||||
profile=PROFILE,
|
||||
task_session_id=rebound["task_session_id"],
|
||||
lock_dir=self.lock_dir.name,
|
||||
now=self.now + timedelta(minutes=1),
|
||||
)
|
||||
self.assertTrue(beat["success"], beat)
|
||||
|
||||
def test_rebind_refuses_a_foreign_owner(self):
|
||||
self.write_lock(lifecycle=None, expires_delta=timedelta(hours=1))
|
||||
result = self._rebind(identity="someone-else")
|
||||
self.assertFalse(result["success"])
|
||||
|
||||
def test_rebind_refuses_a_lock_already_on_the_lifecycle(self):
|
||||
self.write_lock()
|
||||
result = self._rebind()
|
||||
self.assertFalse(result["success"])
|
||||
self.assertFalse(result["legacy_lease"])
|
||||
|
||||
def test_rebind_is_not_a_recovery_path_for_a_lapsed_legacy_lease(self):
|
||||
"""An expired legacy lease belongs to #760 renewal or #601 reclaim."""
|
||||
self.write_lock(
|
||||
lifecycle=None,
|
||||
created_delta=timedelta(hours=5),
|
||||
heartbeat_delta=timedelta(hours=5),
|
||||
expires_delta=timedelta(hours=-1),
|
||||
)
|
||||
result = self._rebind()
|
||||
self.assertFalse(result["success"])
|
||||
self.assertIn("not a recovery path", " ".join(result["reasons"]))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,193 @@
|
||||
"""Conflict gate honors heartbeat-lifecycle non-live reclaim bands (#790 review #502/#516).
|
||||
|
||||
``assess_expired_lock_reclaim`` already permits reclaim for the heartbeat-lifecycle
|
||||
stale bands (``stale_missed_heartbeat``, ``stale_absolute_cap``) without a dead PID.
|
||||
But ``assess_same_issue_lease_conflict`` used to enter that reclaim branch only under
|
||||
``is_lease_expired`` (``expires_at <= now``). For a heartbeat-lifecycle lease that is
|
||||
non-live yet whose ``expires_at`` is still in the future, the acquisition gate fell
|
||||
through to the foreign "already has an active lease" block and never consulted the
|
||||
reclaim assessor — so the load-bearing heartbeat was not load-bearing for foreign
|
||||
reclaim, the exact abandonment scenario #790 exists to fix.
|
||||
|
||||
Two future-``expires_at`` non-live shapes are reachable:
|
||||
|
||||
* ``stale_absolute_cap`` — a session that keeps heartbeating past the 8h absolute cap
|
||||
has ``expires_at = last_heartbeat + TTL`` in the future (default policy).
|
||||
* ``stale_missed_heartbeat`` — under an independent TTL>grace policy the heartbeat
|
||||
grace lapses while ``expires_at`` is still ahead.
|
||||
|
||||
These tests pin: both reclaim from a foreign acquirer; a live lease still blocks a
|
||||
foreign acquirer; the same owner may reclaim its own abandoned heartbeat lease; and a
|
||||
legacy (pre-lifecycle) lock keeps its absolute-``expires_at`` clock — a non-expired
|
||||
legacy lock with a dead PID is *not* reclaimable through this path.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from unittest import mock
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
import issue_lock_store as ils # noqa: E402
|
||||
import lease_policy # noqa: E402
|
||||
|
||||
ISSUE = 790
|
||||
OWNER_BRANCH = f"fix/issue-{ISSUE}-slice-a-heartbeat-policy"
|
||||
OWNER_WORKTREE = "/tmp/wt-790-owner"
|
||||
FOREIGN_BRANCH = f"fix/issue-{ISSUE}-foreign-attempt"
|
||||
FOREIGN_WORKTREE = "/tmp/wt-790-foreign"
|
||||
|
||||
|
||||
def _ts(moment: datetime) -> str:
|
||||
return (
|
||||
moment.astimezone(timezone.utc)
|
||||
.replace(microsecond=0)
|
||||
.isoformat()
|
||||
.replace("+00:00", "Z")
|
||||
)
|
||||
|
||||
|
||||
class _ConflictGateBase(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.now = datetime(2026, 7, 23, 12, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
def _lock(
|
||||
self,
|
||||
*,
|
||||
lifecycle: str | None,
|
||||
created_ago: timedelta,
|
||||
heartbeat_ago: timedelta,
|
||||
expires_in: timedelta,
|
||||
pid: int,
|
||||
) -> dict:
|
||||
lease: dict = {
|
||||
"operation_type": ils.AUTHOR_ISSUE_WORK_LEASE,
|
||||
"issue_number": ISSUE,
|
||||
"branch": OWNER_BRANCH,
|
||||
"worktree_path": OWNER_WORKTREE,
|
||||
"created_at": _ts(self.now - created_ago),
|
||||
"last_heartbeat_at": _ts(self.now - heartbeat_ago),
|
||||
"expires_at": _ts(self.now + expires_in),
|
||||
}
|
||||
if lifecycle is not None:
|
||||
lease["lifecycle_version"] = lifecycle
|
||||
lease["task_session_id"] = "author_issue_work-deadbeefdeadbeef"
|
||||
return {
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": OWNER_BRANCH,
|
||||
"remote": "prgs",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"worktree_path": OWNER_WORKTREE,
|
||||
"session_pid": pid,
|
||||
"pid": pid,
|
||||
"work_lease": lease,
|
||||
}
|
||||
|
||||
def _foreign_conflict(self, existing: dict) -> str | None:
|
||||
return ils.assess_same_issue_lease_conflict(
|
||||
existing,
|
||||
issue_number=ISSUE,
|
||||
branch_name=FOREIGN_BRANCH,
|
||||
worktree_path=FOREIGN_WORKTREE,
|
||||
now=self.now,
|
||||
)
|
||||
|
||||
def _same_owner_conflict(self, existing: dict) -> str | None:
|
||||
return ils.assess_same_issue_lease_conflict(
|
||||
existing,
|
||||
issue_number=ISSUE,
|
||||
branch_name=OWNER_BRANCH,
|
||||
worktree_path=OWNER_WORKTREE,
|
||||
now=self.now,
|
||||
)
|
||||
|
||||
|
||||
class TestHeartbeatNonLiveFutureExpiresReclaim(_ConflictGateBase):
|
||||
def test_stale_absolute_cap_future_expires_allows_foreign_reclaim(self):
|
||||
# Created >8h ago, heartbeated one minute ago, expires 9 min in the FUTURE,
|
||||
# owner PID alive: freshness = stale_absolute_cap, live=False, not expired.
|
||||
existing = self._lock(
|
||||
lifecycle=lease_policy.LIFECYCLE_HEARTBEAT_V1,
|
||||
created_ago=timedelta(hours=9),
|
||||
heartbeat_ago=timedelta(minutes=1),
|
||||
expires_in=timedelta(minutes=9),
|
||||
pid=os.getpid(),
|
||||
)
|
||||
freshness = ils.assess_lock_freshness(existing, now=self.now)
|
||||
self.assertEqual(freshness["status"], ils.STATUS_STALE_ABSOLUTE_CAP)
|
||||
self.assertFalse(freshness["live"])
|
||||
self.assertFalse(ils.is_lease_expired(existing, now=self.now))
|
||||
with mock.patch.object(ils, "is_process_alive", return_value=True):
|
||||
self.assertIsNone(self._foreign_conflict(existing))
|
||||
|
||||
def test_missed_heartbeat_ttl_gt_grace_future_expires_allows_foreign_reclaim(self):
|
||||
# Heartbeat grace (default 10 min) lapsed 5 min ago, but a TTL>grace policy
|
||||
# leaves expires_at 20 min in the FUTURE: stale_missed_heartbeat, not expired.
|
||||
existing = self._lock(
|
||||
lifecycle=lease_policy.LIFECYCLE_HEARTBEAT_V1,
|
||||
created_ago=timedelta(minutes=30),
|
||||
heartbeat_ago=timedelta(minutes=15),
|
||||
expires_in=timedelta(minutes=20),
|
||||
pid=os.getpid(),
|
||||
)
|
||||
freshness = ils.assess_lock_freshness(existing, now=self.now)
|
||||
self.assertEqual(freshness["status"], ils.STATUS_STALE_MISSED_HEARTBEAT)
|
||||
self.assertFalse(freshness["live"])
|
||||
self.assertFalse(ils.is_lease_expired(existing, now=self.now))
|
||||
with mock.patch.object(ils, "is_process_alive", return_value=True):
|
||||
self.assertIsNone(self._foreign_conflict(existing))
|
||||
|
||||
def test_same_owner_may_reclaim_its_own_abandoned_heartbeat_lease(self):
|
||||
existing = self._lock(
|
||||
lifecycle=lease_policy.LIFECYCLE_HEARTBEAT_V1,
|
||||
created_ago=timedelta(hours=9),
|
||||
heartbeat_ago=timedelta(minutes=1),
|
||||
expires_in=timedelta(minutes=9),
|
||||
pid=os.getpid(),
|
||||
)
|
||||
with mock.patch.object(ils, "is_process_alive", return_value=True):
|
||||
self.assertIsNone(self._same_owner_conflict(existing))
|
||||
|
||||
|
||||
class TestLiveAndLegacyStillBlockForeign(_ConflictGateBase):
|
||||
def test_live_heartbeat_lease_still_blocks_foreign(self):
|
||||
existing = self._lock(
|
||||
lifecycle=lease_policy.LIFECYCLE_HEARTBEAT_V1,
|
||||
created_ago=timedelta(minutes=5),
|
||||
heartbeat_ago=timedelta(minutes=1),
|
||||
expires_in=timedelta(minutes=9),
|
||||
pid=os.getpid(),
|
||||
)
|
||||
freshness = ils.assess_lock_freshness(existing, now=self.now)
|
||||
self.assertTrue(freshness["live"])
|
||||
with mock.patch.object(ils, "is_process_alive", return_value=True):
|
||||
block = self._foreign_conflict(existing)
|
||||
self.assertIn("already has an active", block or "")
|
||||
|
||||
def test_legacy_lease_future_expires_dead_pid_is_not_reclaimed_here(self):
|
||||
# AC-N8: a legacy lock keeps its absolute expires_at clock. Non-expired +
|
||||
# dead PID is non-live, but the widened band excludes legacy, so the
|
||||
# foreign acquirer is still blocked rather than silently reclaiming.
|
||||
existing = self._lock(
|
||||
lifecycle=None,
|
||||
created_ago=timedelta(hours=9),
|
||||
heartbeat_ago=timedelta(hours=9),
|
||||
expires_in=timedelta(hours=2),
|
||||
pid=4_194_304, # far above any live pid on a test host
|
||||
)
|
||||
with mock.patch.object(ils, "is_process_alive", return_value=False):
|
||||
freshness = ils.assess_lock_freshness(existing, now=self.now)
|
||||
self.assertTrue(freshness["legacy_lease"])
|
||||
self.assertFalse(freshness["live"])
|
||||
self.assertFalse(ils.is_lease_expired(existing, now=self.now))
|
||||
block = self._foreign_conflict(existing)
|
||||
self.assertIn("already has an active", block or "")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,682 @@
|
||||
"""Cross-role allocation handoff consumable by independent workers (#843).
|
||||
|
||||
Regression coverage for the controller→required-role consume path:
|
||||
|
||||
* controller allocates author work; independent author adopts successfully
|
||||
* author adoption succeeds after allocating controller process exits
|
||||
* author adoption without sharing controller session identity
|
||||
* wrong-role adoption rejected
|
||||
* concurrent/second adoption rejected without state corruption
|
||||
* terminal allocation adoption rejected
|
||||
* successful adoption produces authoritative ownership evidence
|
||||
* genuine abandoned-lease recovery remains valid
|
||||
* process_work_queue / allocate results include consume identifiers
|
||||
* same-role allocation behavior remains compatible
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import timedelta
|
||||
from unittest.mock import patch
|
||||
|
||||
from allocator_service import (
|
||||
ALLOCATION_MODE_CROSS_ROLE,
|
||||
ALLOCATION_MODE_ROLE_SCOPED,
|
||||
OUTCOME_ASSIGNED,
|
||||
ROLE_AUTHOR,
|
||||
ROLE_CONTROLLER,
|
||||
ROLE_REVIEWER,
|
||||
WorkCandidate,
|
||||
allocate_next_work,
|
||||
)
|
||||
from control_plane_db import ControlPlaneDB, ForeignLeaseError, _ts, _utc_now
|
||||
import lease_lifecycle as ll
|
||||
|
||||
|
||||
class CrossRoleHandoffTest(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.db_path = os.path.join(self._tmp.name, "cp.sqlite3")
|
||||
self.db = ControlPlaneDB(self.db_path)
|
||||
self.db.upsert_session(
|
||||
session_id="ctrl-session",
|
||||
role="controller",
|
||||
profile="prgs-controller",
|
||||
pid=99999999, # dead-looking pid
|
||||
)
|
||||
self.db.upsert_session(
|
||||
session_id="author-worker",
|
||||
role="author",
|
||||
profile="prgs-author",
|
||||
pid=os.getpid(),
|
||||
)
|
||||
self.db.upsert_session(
|
||||
session_id="author-worker-2",
|
||||
role="author",
|
||||
profile="prgs-author",
|
||||
pid=os.getpid(),
|
||||
)
|
||||
self.db.upsert_session(
|
||||
session_id="reviewer-worker",
|
||||
role="reviewer",
|
||||
profile="prgs-reviewer",
|
||||
pid=os.getpid(),
|
||||
)
|
||||
self.wt = self._tmp.name
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _ready_issue(self, number: int = 843, title: str = "handoff target") -> WorkCandidate:
|
||||
return WorkCandidate(
|
||||
kind="issue",
|
||||
number=number,
|
||||
labels=("status:ready", "type:bug"),
|
||||
title=title,
|
||||
priority=20,
|
||||
)
|
||||
|
||||
def _controller_allocate(self, number: int = 843, **kwargs):
|
||||
defaults = dict(
|
||||
db=self.db,
|
||||
session_id="ctrl-session",
|
||||
role=ROLE_CONTROLLER,
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
candidates=[self._ready_issue(number)],
|
||||
apply=True,
|
||||
profile_name="prgs-controller",
|
||||
username="controller-user",
|
||||
allocation_mode=ALLOCATION_MODE_CROSS_ROLE,
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
return allocate_next_work(**defaults)
|
||||
|
||||
def test_controller_allocates_author_independent_author_adopts(self) -> None:
|
||||
res = self._controller_allocate()
|
||||
self.assertEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertEqual(res["required_role"], ROLE_AUTHOR)
|
||||
self.assertIn("consume_allocation", res)
|
||||
consume = res["consume_allocation"]
|
||||
self.assertEqual(consume["tool"], "gitea_adopt_workflow_lease")
|
||||
self.assertEqual(consume["required_role"], ROLE_AUTHOR)
|
||||
self.assertFalse(consume["controller_session_required"])
|
||||
lid = res["assignment"]["lease_id"]
|
||||
self.assertEqual(consume["lease_id"], lid)
|
||||
self.assertIn(lid, res["next_valid_command"])
|
||||
|
||||
adopted = ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertTrue(adopted["success"])
|
||||
self.assertEqual(adopted["outcome"], "adopted_cross_role_handoff")
|
||||
self.assertEqual(adopted["adopted_by_session_id"], "author-worker")
|
||||
self.assertEqual(adopted["adopted_from_session_id"], "ctrl-session")
|
||||
raw = adopted["read_after_write"]
|
||||
self.assertEqual(raw["session_id"], "author-worker")
|
||||
self.assertEqual(raw["adopted_by_session_id"], "author-worker")
|
||||
self.assertEqual(raw["status"], "active")
|
||||
self.assertEqual(raw["phase"], "adopted")
|
||||
|
||||
# Authoritative re-read
|
||||
state = self.db.get_lease_workflow_state(lid)
|
||||
self.assertEqual(state["lease"]["session_id"], "author-worker")
|
||||
self.assertEqual(state["lease"]["adopted_by_session_id"], "author-worker")
|
||||
self.assertEqual(state["assignment"]["session_id"], "author-worker")
|
||||
self.assertEqual(state["provenance"]["handoff_status"], "adopted")
|
||||
|
||||
def test_author_adoption_after_controller_process_exits(self) -> None:
|
||||
res = self._controller_allocate(number=900)
|
||||
lid = res["assignment"]["lease_id"]
|
||||
# Force owner_pid dead + freshness stale_dead_process
|
||||
import sqlite3
|
||||
|
||||
conn = sqlite3.connect(self.db_path)
|
||||
try:
|
||||
conn.execute(
|
||||
"UPDATE leases SET owner_pid = 99999999 WHERE lease_id = ?",
|
||||
(lid,),
|
||||
)
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
state = self.db.get_lease_workflow_state(lid)
|
||||
fr = ll.classify_lease_freshness(
|
||||
state["lease"], pid_checker=lambda _p: False
|
||||
)
|
||||
self.assertEqual(fr["freshness"], "stale_dead_process")
|
||||
|
||||
adopted = ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertEqual(adopted["outcome"], "adopted_cross_role_handoff")
|
||||
self.assertEqual(adopted["adopted_by_session_id"], "author-worker")
|
||||
# No abandon required
|
||||
state2 = self.db.get_lease_workflow_state(lid)
|
||||
self.assertEqual(state2["lease"]["status"], "active")
|
||||
self.assertNotEqual(state2["lease"]["status"], "abandoned")
|
||||
|
||||
def test_adoption_without_sharing_controller_session_identity(self) -> None:
|
||||
res = self._controller_allocate(number=901)
|
||||
lid = res["assignment"]["lease_id"]
|
||||
adopted = ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertNotEqual(adopted["adopted_by_session_id"], "ctrl-session")
|
||||
self.assertFalse(adopted["same_owner"])
|
||||
self.assertEqual(adopted["adopted_from_session_id"], "ctrl-session")
|
||||
|
||||
def test_wrong_role_adoption_rejected(self) -> None:
|
||||
res = self._controller_allocate(number=902)
|
||||
lid = res["assignment"]["lease_id"]
|
||||
with self.assertRaises(ll.LeaseLifecycleError) as ctx:
|
||||
ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="reviewer-worker",
|
||||
role=ROLE_REVIEWER,
|
||||
)
|
||||
self.assertIn("wrong role", str(ctx.exception).lower())
|
||||
# State unchanged
|
||||
state = self.db.get_lease_workflow_state(lid)
|
||||
self.assertEqual(state["lease"]["session_id"], "ctrl-session")
|
||||
self.assertIsNone(state["lease"].get("adopted_by_session_id") or None)
|
||||
self.assertEqual(state["provenance"]["handoff_status"], "pending")
|
||||
|
||||
def test_second_adoption_rejected_without_corruption(self) -> None:
|
||||
res = self._controller_allocate(number=903)
|
||||
lid = res["assignment"]["lease_id"]
|
||||
first = ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertEqual(first["outcome"], "adopted_cross_role_handoff")
|
||||
with self.assertRaises(ll.LeaseLifecycleError):
|
||||
ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker-2",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
state = self.db.get_lease_workflow_state(lid)
|
||||
self.assertEqual(state["lease"]["session_id"], "author-worker")
|
||||
self.assertEqual(state["lease"]["adopted_by_session_id"], "author-worker")
|
||||
self.assertEqual(state["assignment"]["session_id"], "author-worker")
|
||||
self.assertEqual(state["lease"]["status"], "active")
|
||||
|
||||
def test_terminal_allocation_adoption_rejected(self) -> None:
|
||||
res = self._controller_allocate(number=904)
|
||||
lid = res["assignment"]["lease_id"]
|
||||
# Abandon as terminal
|
||||
proof = ll.AbandonProof(
|
||||
dead_process=True,
|
||||
missing_worktree=True,
|
||||
no_open_pr=True,
|
||||
no_live_mutation_risk=True,
|
||||
owner_pid=99999999,
|
||||
worktree_path="/nonexistent/for-843",
|
||||
)
|
||||
# Attach dead pid / missing wt for abandon eligibility
|
||||
import sqlite3
|
||||
|
||||
conn = sqlite3.connect(self.db_path)
|
||||
try:
|
||||
conn.execute(
|
||||
"UPDATE leases SET owner_pid = 99999999, worktree_path = ? WHERE lease_id = ?",
|
||||
("/nonexistent/for-843", lid),
|
||||
)
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
abandoned = ll.abandon_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
requester_session_id="author-worker",
|
||||
proof=proof,
|
||||
)
|
||||
self.assertEqual(abandoned["outcome"], "abandoned")
|
||||
with self.assertRaises(ll.LeaseLifecycleError) as ctx:
|
||||
ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
)
|
||||
self.assertIn("abandoned", str(ctx.exception).lower())
|
||||
|
||||
def test_successful_adoption_read_after_write_ownership(self) -> None:
|
||||
res = self._controller_allocate(number=905)
|
||||
lid = res["assignment"]["lease_id"]
|
||||
adopted = ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
raw = adopted["read_after_write"]
|
||||
self.assertEqual(raw["lease_id"], lid)
|
||||
self.assertEqual(raw["session_id"], "author-worker")
|
||||
self.assertEqual(raw["adopted_by_session_id"], "author-worker")
|
||||
self.assertEqual(raw["adopted_from_session_id"], "ctrl-session")
|
||||
# Re-fetch proves durable write
|
||||
state = self.db.get_lease_workflow_state(lid)
|
||||
self.assertEqual(state["lease"]["session_id"], raw["session_id"])
|
||||
self.assertEqual(
|
||||
state["lease"]["adopted_by_session_id"], raw["adopted_by_session_id"]
|
||||
)
|
||||
|
||||
def test_genuine_abandoned_recovery_still_valid(self) -> None:
|
||||
"""Same-role author lease abandoned remains reclaimable via abandon path."""
|
||||
same = allocate_next_work(
|
||||
self.db,
|
||||
session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
candidates=[self._ready_issue(906, "same-role")],
|
||||
apply=True,
|
||||
profile_name="prgs-author",
|
||||
username="author-user",
|
||||
allocation_mode=ALLOCATION_MODE_ROLE_SCOPED,
|
||||
)
|
||||
self.assertEqual(same["outcome"], OUTCOME_ASSIGNED)
|
||||
lid = same["assignment"]["lease_id"]
|
||||
import sqlite3
|
||||
|
||||
conn = sqlite3.connect(self.db_path)
|
||||
try:
|
||||
conn.execute(
|
||||
"UPDATE leases SET owner_pid = 99999999, worktree_path = ? WHERE lease_id = ?",
|
||||
("/nonexistent/same-role", lid),
|
||||
)
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
proof = ll.AbandonProof(
|
||||
dead_process=True,
|
||||
missing_worktree=True,
|
||||
no_open_pr=True,
|
||||
no_live_mutation_risk=True,
|
||||
owner_pid=99999999,
|
||||
worktree_path="/nonexistent/same-role",
|
||||
)
|
||||
abandoned = ll.abandon_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
requester_session_id="author-worker-2",
|
||||
proof=proof,
|
||||
)
|
||||
self.assertEqual(abandoned["outcome"], "abandoned")
|
||||
# Foreign author cannot handoff-consume an abandoned non-handoff lease
|
||||
with self.assertRaises(ll.LeaseLifecycleError):
|
||||
ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker-2",
|
||||
role=ROLE_AUTHOR,
|
||||
)
|
||||
# Reclaim path still works for expired/abandoned after force-expire
|
||||
reclaimed = ll.reclaim_expired_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
session_id="author-worker-2",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertEqual(reclaimed["outcome"], "reclaimed")
|
||||
self.assertEqual(reclaimed["assignment"]["session_id"], "author-worker-2")
|
||||
|
||||
def test_allocate_payload_includes_consume_identifiers(self) -> None:
|
||||
res = self._controller_allocate(number=907)
|
||||
self.assertIn("consume_allocation", res)
|
||||
c = res["consume_allocation"]
|
||||
for key in (
|
||||
"tool",
|
||||
"lease_id",
|
||||
"assignment_id",
|
||||
"required_role",
|
||||
"required_profile",
|
||||
"required_namespace",
|
||||
"instructions",
|
||||
"handoff_status",
|
||||
):
|
||||
self.assertIn(key, c)
|
||||
self.assertEqual(c["required_namespace"], "gitea-author")
|
||||
self.assertEqual(c["required_profile"], "prgs-author")
|
||||
self.assertIn("gitea_adopt_workflow_lease", c["instructions"])
|
||||
self.assertTrue(res["lease_proof"]["cross_role_handoff"])
|
||||
self.assertEqual(res["lease_proof"]["handoff_status"], "pending")
|
||||
|
||||
def test_same_role_allocation_remains_compatible(self) -> None:
|
||||
res = allocate_next_work(
|
||||
self.db,
|
||||
session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
candidates=[self._ready_issue(908)],
|
||||
apply=True,
|
||||
profile_name="prgs-author",
|
||||
username="author-user",
|
||||
)
|
||||
self.assertEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertNotIn("consume_allocation", res)
|
||||
lid = res["assignment"]["lease_id"]
|
||||
state = self.db.get_lease_workflow_state(lid)
|
||||
# No cross-role handoff provenance
|
||||
prov = state.get("provenance") or {}
|
||||
self.assertFalse(prov.get("cross_role_handoff"))
|
||||
# Owner resume still works
|
||||
resume = ll.adopt_lease(
|
||||
self.db,
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertTrue(resume["same_owner"])
|
||||
self.assertEqual(resume["outcome"], "adopted_owner_resume")
|
||||
|
||||
def test_inspect_points_required_role_at_consume(self) -> None:
|
||||
res = self._controller_allocate(number=909)
|
||||
lid = res["assignment"]["lease_id"]
|
||||
decision = ll.inspect_lease(
|
||||
self.db, lid, caller_session_id="author-worker"
|
||||
)
|
||||
self.assertEqual(
|
||||
decision["safe_next_action"], ll.SAFE_CONSUME_CROSS_ROLE
|
||||
)
|
||||
self.assertFalse(decision["block"])
|
||||
self.assertEqual(decision["required_role"], ROLE_AUTHOR)
|
||||
|
||||
def test_db_cas_rejects_concurrent_second_consume(self) -> None:
|
||||
res = self._controller_allocate(number=910)
|
||||
lid = res["assignment"]["lease_id"]
|
||||
# First consume via DB layer directly
|
||||
first = self.db.adopt_lease(
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
provenance={
|
||||
"cross_role_handoff": True,
|
||||
"handoff_status": "adopted",
|
||||
"required_role": "author",
|
||||
},
|
||||
)
|
||||
self.assertEqual(first["outcome"], "adopted_cross_role_handoff")
|
||||
# Second CAS must fail
|
||||
with self.assertRaises(ForeignLeaseError):
|
||||
self.db.adopt_lease(
|
||||
lease_id=lid,
|
||||
adopter_session_id="author-worker-2",
|
||||
role=ROLE_AUTHOR,
|
||||
worktree_path=self.wt,
|
||||
provenance={
|
||||
"cross_role_handoff": True,
|
||||
"handoff_status": "pending",
|
||||
"required_role": "author",
|
||||
},
|
||||
)
|
||||
state = self.db.get_lease_workflow_state(lid)
|
||||
self.assertEqual(state["lease"]["session_id"], "author-worker")
|
||||
|
||||
|
||||
class MCPBoundaryAdoptRoleBindingTest(unittest.TestCase):
|
||||
"""#843 F1: MCP-boundary role binding for ``gitea_adopt_workflow_lease``.
|
||||
|
||||
The library-level wrong-role test calls ``lease_lifecycle.adopt_lease``
|
||||
directly. These tests prove the MCP entry point derives the adopter role
|
||||
authoritatively from the active authenticated profile and rejects any
|
||||
caller-supplied role that disagrees, so a reviewer/merger profile cannot
|
||||
consume an author handoff by passing ``role="author"``.
|
||||
"""
|
||||
|
||||
AUTHOR_PROFILE = {
|
||||
"profile_name": "prgs-author",
|
||||
"role": "author",
|
||||
"allowed_operations": [
|
||||
"gitea.read",
|
||||
"gitea.pr.create",
|
||||
"gitea.branch.push",
|
||||
],
|
||||
"forbidden_operations": [],
|
||||
}
|
||||
REVIEWER_PROFILE = {
|
||||
"profile_name": "prgs-reviewer",
|
||||
"role": "reviewer",
|
||||
"allowed_operations": [
|
||||
"gitea.read",
|
||||
"gitea.pr.review",
|
||||
"gitea.pr.approve",
|
||||
"gitea.pr.request_changes",
|
||||
],
|
||||
"forbidden_operations": ["gitea.pr.create", "gitea.branch.push"],
|
||||
}
|
||||
MERGER_PROFILE = {
|
||||
"profile_name": "prgs-merger",
|
||||
"role": "merger",
|
||||
"allowed_operations": ["gitea.read", "gitea.pr.merge"],
|
||||
"forbidden_operations": ["gitea.pr.create", "gitea.branch.push"],
|
||||
}
|
||||
FOREIGN_AUTHOR_PROFILE = {
|
||||
"profile_name": "dadeschools-author",
|
||||
"role": "author",
|
||||
"allowed_operations": [
|
||||
"gitea.read",
|
||||
"gitea.pr.create",
|
||||
"gitea.branch.push",
|
||||
],
|
||||
"forbidden_operations": [],
|
||||
}
|
||||
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.db_path = os.path.join(self._tmp.name, "cp.sqlite3")
|
||||
self.db = ControlPlaneDB(self.db_path)
|
||||
self.db.upsert_session(
|
||||
session_id="ctrl-session",
|
||||
role="controller",
|
||||
profile="prgs-controller",
|
||||
pid=99999999,
|
||||
)
|
||||
self.db.upsert_session(
|
||||
session_id="author-worker",
|
||||
role="author",
|
||||
profile="prgs-author",
|
||||
pid=os.getpid(),
|
||||
)
|
||||
self.wt = self._tmp.name
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _ready_issue(self, number: int) -> WorkCandidate:
|
||||
return WorkCandidate(
|
||||
kind="issue",
|
||||
number=number,
|
||||
labels=("status:ready", "type:bug"),
|
||||
title="handoff target",
|
||||
priority=20,
|
||||
)
|
||||
|
||||
def _handoff_lease(self, number: int = 843) -> str:
|
||||
res = allocate_next_work(
|
||||
db=self.db,
|
||||
session_id="ctrl-session",
|
||||
role=ROLE_CONTROLLER,
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
candidates=[self._ready_issue(number)],
|
||||
apply=True,
|
||||
profile_name="prgs-controller",
|
||||
username="controller-user",
|
||||
allocation_mode=ALLOCATION_MODE_CROSS_ROLE,
|
||||
)
|
||||
self.assertEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertEqual(res["required_role"], ROLE_AUTHOR)
|
||||
return res["assignment"]["lease_id"]
|
||||
|
||||
def _call_adopt_tool(self, profile: dict, **kwargs):
|
||||
import gitea_mcp_server as mcp_server
|
||||
|
||||
with (
|
||||
patch.object(mcp_server, "get_profile", return_value=profile),
|
||||
patch.object(
|
||||
mcp_server,
|
||||
"_control_plane_db_or_error",
|
||||
return_value=(self.db, []),
|
||||
),
|
||||
):
|
||||
return mcp_server.gitea_adopt_workflow_lease(
|
||||
remote="prgs", **kwargs
|
||||
)
|
||||
|
||||
def _assert_handoff_untouched(self, lease_id: str) -> None:
|
||||
state = self.db.get_lease_workflow_state(lease_id)
|
||||
self.assertEqual(state["lease"]["session_id"], "ctrl-session")
|
||||
self.assertIsNone(state["lease"].get("adopted_by_session_id") or None)
|
||||
self.assertEqual(state["lease"]["status"], "active")
|
||||
self.assertEqual(state["provenance"]["handoff_status"], "pending")
|
||||
|
||||
def test_reviewer_profile_cannot_consume_author_handoff_via_role_author(
|
||||
self,
|
||||
) -> None:
|
||||
lid = self._handoff_lease(920)
|
||||
result = self._call_adopt_tool(
|
||||
self.REVIEWER_PROFILE,
|
||||
lease_id=lid,
|
||||
session_id="reviewer-worker",
|
||||
role="author",
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], "blocked")
|
||||
self.assertEqual(result["profile_role_kind"], "reviewer")
|
||||
self.assertEqual(result["supplied_role"], "author")
|
||||
self.assertIn("does not match", result["reasons"][0])
|
||||
self._assert_handoff_untouched(lid)
|
||||
|
||||
def test_merger_profile_cannot_consume_author_handoff_via_role_author(
|
||||
self,
|
||||
) -> None:
|
||||
lid = self._handoff_lease(921)
|
||||
result = self._call_adopt_tool(
|
||||
self.MERGER_PROFILE,
|
||||
lease_id=lid,
|
||||
session_id="merger-worker",
|
||||
role="author",
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], "blocked")
|
||||
self.assertEqual(result["profile_role_kind"], "merger")
|
||||
self._assert_handoff_untouched(lid)
|
||||
|
||||
def test_reviewer_profile_rejected_without_role_argument(self) -> None:
|
||||
"""Even without a spoofed role, the profile-derived role binds."""
|
||||
lid = self._handoff_lease(922)
|
||||
result = self._call_adopt_tool(
|
||||
self.REVIEWER_PROFILE,
|
||||
lease_id=lid,
|
||||
session_id="reviewer-worker",
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], "blocked")
|
||||
self.assertIn("wrong role", result["reasons"][0].lower())
|
||||
self._assert_handoff_untouched(lid)
|
||||
|
||||
def test_author_profile_mismatching_supplied_role_rejected(self) -> None:
|
||||
lid = self._handoff_lease(923)
|
||||
result = self._call_adopt_tool(
|
||||
self.AUTHOR_PROFILE,
|
||||
lease_id=lid,
|
||||
session_id="author-worker",
|
||||
role="reviewer",
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], "blocked")
|
||||
self.assertEqual(result["profile_role_kind"], "author")
|
||||
self.assertEqual(result["supplied_role"], "reviewer")
|
||||
self._assert_handoff_untouched(lid)
|
||||
|
||||
def test_foreign_profile_name_rejected_for_author_handoff(self) -> None:
|
||||
"""Provenance required_profile binds even when the role matches."""
|
||||
lid = self._handoff_lease(924)
|
||||
result = self._call_adopt_tool(
|
||||
self.FOREIGN_AUTHOR_PROFILE,
|
||||
lease_id=lid,
|
||||
session_id="foreign-author-worker",
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], "blocked")
|
||||
self.assertIn("wrong profile", result["reasons"][0].lower())
|
||||
self._assert_handoff_untouched(lid)
|
||||
|
||||
def test_author_profile_consumes_author_handoff(self) -> None:
|
||||
lid = self._handoff_lease(925)
|
||||
result = self._call_adopt_tool(
|
||||
self.AUTHOR_PROFILE,
|
||||
lease_id=lid,
|
||||
session_id="author-worker",
|
||||
role="author",
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertTrue(result["success"])
|
||||
self.assertEqual(result["outcome"], "adopted_cross_role_handoff")
|
||||
self.assertEqual(result["adopted_by_session_id"], "author-worker")
|
||||
self.assertEqual(result["adopted_from_session_id"], "ctrl-session")
|
||||
state = self.db.get_lease_workflow_state(lid)
|
||||
self.assertEqual(state["lease"]["session_id"], "author-worker")
|
||||
self.assertEqual(
|
||||
state["lease"]["adopted_by_session_id"], "author-worker"
|
||||
)
|
||||
self.assertEqual(state["assignment"]["session_id"], "author-worker")
|
||||
self.assertEqual(state["provenance"]["handoff_status"], "adopted")
|
||||
|
||||
def test_author_profile_consumes_author_handoff_without_role_argument(
|
||||
self,
|
||||
) -> None:
|
||||
lid = self._handoff_lease(926)
|
||||
result = self._call_adopt_tool(
|
||||
self.AUTHOR_PROFILE,
|
||||
lease_id=lid,
|
||||
session_id="author-worker",
|
||||
worktree_path=self.wt,
|
||||
)
|
||||
self.assertTrue(result["success"])
|
||||
self.assertEqual(result["outcome"], "adopted_cross_role_handoff")
|
||||
state = self.db.get_lease_workflow_state(lid)
|
||||
self.assertEqual(state["lease"]["session_id"], "author-worker")
|
||||
self.assertEqual(state["lease"]["role"], "author")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,378 @@
|
||||
"""Allocator semantic container exclusion for vision/roadmap/umbrella (#854).
|
||||
|
||||
#844 excluded epic / child-only containers (live #631) but product-vision
|
||||
(#652), phased-roadmap (#653), and umbrella (#655) coordination records still
|
||||
ranked as implementable work. This module is the live-equivalent canary:
|
||||
|
||||
* #631 / #652 / #653 / #655-shaped records are all excluded in one inventory.
|
||||
* Independently executable children remain eligible and can be selected.
|
||||
* Ordinary issues that merely mention vision / roadmap / umbrella stay eligible.
|
||||
* Excluded containers never receive assignments or workflow leases.
|
||||
* Structured skip reason ``epic_or_child_only_container`` is reported.
|
||||
* Candidate-set fingerprint remains stable after exclusions.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
|
||||
from allocator_service import (
|
||||
OUTCOME_ASSIGNED,
|
||||
OUTCOME_PREVIEW,
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
WorkCandidate,
|
||||
allocate_next_work,
|
||||
candidate_set_fingerprint,
|
||||
classify_epic_or_child_only_container,
|
||||
)
|
||||
from control_plane_db import ControlPlaneDB
|
||||
|
||||
REMOTE = "prgs"
|
||||
ORG = "Scaled-Tech-Consulting"
|
||||
REPO = "Gitea-Tools"
|
||||
|
||||
# Minimal bodies mirroring live coordination records (not full issue text).
|
||||
_EPIC_631_BODY = """
|
||||
## Scope (umbrella)
|
||||
|
||||
This epic owns the **product roadmap and linkage** for the Web Console.
|
||||
Implementation is delivered via child issues only.
|
||||
|
||||
## Explicit non-goals
|
||||
|
||||
* Do not implement product features in this epic issue itself.
|
||||
* No product feature implementation is claimed complete solely on this epic.
|
||||
"""
|
||||
|
||||
_VISION_652_BODY = """
|
||||
## Canonical product vision — enduring source of truth
|
||||
|
||||
**This issue is the enduring source of truth for the MCP Control Plane Web Console product vision.**
|
||||
|
||||
## Implementation linkage
|
||||
|
||||
* **Do not implement features on this issue.**
|
||||
* Sequencing: roadmap issue + #631 children.
|
||||
|
||||
## Canonical issue state
|
||||
|
||||
```text
|
||||
STATE: vision-active
|
||||
WHO_IS_NEXT: controller (triage/ordering) / author (implementation of linked children only)
|
||||
```
|
||||
"""
|
||||
|
||||
_ROADMAP_653_BODY = """
|
||||
## Purpose
|
||||
|
||||
This issue is the **phased delivery roadmap and epic sequencing** for the MCP Control Plane Web Console.
|
||||
|
||||
## Non-goals
|
||||
|
||||
* Implementing features on this roadmap issue.
|
||||
* Deleting vision items by omitting them from phases without #652 change log.
|
||||
|
||||
## Canonical issue state
|
||||
|
||||
```text
|
||||
STATE: roadmap-active
|
||||
WHO_IS_NEXT: author
|
||||
```
|
||||
"""
|
||||
|
||||
_UMBRELLA_655_BODY = """
|
||||
## Scope (umbrella)
|
||||
|
||||
This issue owns the **canonical restart-governance program**. Implementation is via linked children only.
|
||||
|
||||
## Acceptance criteria (umbrella)
|
||||
|
||||
6. No product feature claimed complete on this issue alone.
|
||||
"""
|
||||
|
||||
_CHILD_BODY = """
|
||||
## Problem
|
||||
|
||||
Operators need a workflow-event timeline model for Phase 1.
|
||||
|
||||
## Acceptance criteria
|
||||
|
||||
- [ ] Timeline model API exists
|
||||
"""
|
||||
|
||||
|
||||
def _issue(
|
||||
number: int,
|
||||
*,
|
||||
title: str = "",
|
||||
body: str = "",
|
||||
labels: tuple[str, ...] = ("status:ready", "type:feature"),
|
||||
priority: int = 20,
|
||||
) -> WorkCandidate:
|
||||
return WorkCandidate(
|
||||
kind="issue",
|
||||
number=number,
|
||||
state="open",
|
||||
labels=labels,
|
||||
title=title or f"issue {number}",
|
||||
body=body,
|
||||
priority=priority,
|
||||
)
|
||||
|
||||
|
||||
def _live_shaped_containers() -> list[WorkCandidate]:
|
||||
return [
|
||||
_issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
),
|
||||
_issue(
|
||||
652,
|
||||
title="Product vision: MCP Control Plane Web Console (canonical)",
|
||||
body=_VISION_652_BODY,
|
||||
),
|
||||
_issue(
|
||||
653,
|
||||
title="Roadmap: MCP Control Plane Web Console (phased delivery)",
|
||||
body=_ROADMAP_653_BODY,
|
||||
),
|
||||
_issue(
|
||||
655,
|
||||
title="Umbrella: Governed MCP restart coordination and zero-disruption recovery",
|
||||
body=_UMBRELLA_655_BODY,
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
class ClassifySemanticContainersTest(unittest.TestCase):
|
||||
def test_652_vision_is_container(self) -> None:
|
||||
c = _issue(
|
||||
652,
|
||||
title="Product vision: MCP Control Plane Web Console (canonical)",
|
||||
body=_VISION_652_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIsNotNone(detail)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_653_roadmap_is_container(self) -> None:
|
||||
c = _issue(
|
||||
653,
|
||||
title="Roadmap: MCP Control Plane Web Console (phased delivery)",
|
||||
body=_ROADMAP_653_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_655_umbrella_is_container(self) -> None:
|
||||
c = _issue(
|
||||
655,
|
||||
title="Umbrella: Governed MCP restart coordination and zero-disruption recovery",
|
||||
body=_UMBRELLA_655_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_631_still_container_after_854(self) -> None:
|
||||
c = _issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_incidental_vision_roadmap_umbrella_words_not_container(self) -> None:
|
||||
cases = (
|
||||
(
|
||||
"Document vision handoff conventions",
|
||||
"Update docs so implementable issues that mention a vision "
|
||||
"remain independently executable.",
|
||||
),
|
||||
(
|
||||
"Clarify roadmap sequencing notes",
|
||||
"Write a short note about how the roadmap issue relates to children.",
|
||||
),
|
||||
(
|
||||
"Umbrella recovery checklist for authors",
|
||||
"Authors should still implement the concrete recovery fix here.",
|
||||
),
|
||||
(
|
||||
"Product vision wording in the help text",
|
||||
"Fix a typo in the operator-facing help string that says product vision.",
|
||||
),
|
||||
)
|
||||
for title, body in cases:
|
||||
with self.subTest(title=title):
|
||||
c = _issue(900, title=title, body=body)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
self.assertIsNone(detail)
|
||||
|
||||
def test_title_prefix_alone_not_container(self) -> None:
|
||||
for title in (
|
||||
"Epic: something mentioned only in title",
|
||||
"Roadmap: title only without body scope",
|
||||
"Product vision: title only without body scope",
|
||||
"Umbrella: title only without body scope",
|
||||
):
|
||||
with self.subTest(title=title):
|
||||
c = _issue(
|
||||
901,
|
||||
title=title,
|
||||
body="Implement a concrete fix for the allocator skip list.",
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
self.assertIsNone(detail)
|
||||
|
||||
def test_roadmap_label_alone_is_container(self) -> None:
|
||||
c = _issue(
|
||||
902,
|
||||
title="Console delivery sequencing",
|
||||
body="Track phased delivery only.",
|
||||
labels=("status:ready", "type:roadmap"),
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("type:roadmap", detail or "")
|
||||
|
||||
def test_child_referencing_parent_policy_stays_eligible(self) -> None:
|
||||
"""Children may quote parent policy without becoming containers."""
|
||||
c = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=(
|
||||
_CHILD_BODY
|
||||
+ "\n\nParent #652 says do not implement on the vision issue; "
|
||||
"this child is the implementable unit."
|
||||
),
|
||||
)
|
||||
is_c, _ = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
|
||||
|
||||
class AllocateSemanticContainerExclusionTest(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(os.path.join(self._tmp.name, "cp.sqlite3"))
|
||||
|
||||
def _alloc(self, candidates, **kwargs):
|
||||
defaults = dict(
|
||||
session_id="sess-854",
|
||||
role="author",
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
profile_name="prgs-author",
|
||||
username="jcwalker3",
|
||||
claims={},
|
||||
apply=False,
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
return allocate_next_work(self.db, candidates=candidates, **defaults)
|
||||
|
||||
def test_live_equivalent_canary_excludes_all_containers_selects_child(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
inventory = containers + [child]
|
||||
res = self._alloc(inventory, apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
|
||||
skipped = {s["number"]: s for s in res["skipped"]}
|
||||
for number in (631, 652, 653, 655):
|
||||
self.assertIn(number, skipped, res["skipped"])
|
||||
self.assertEqual(
|
||||
skipped[number]["reason_code"],
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
)
|
||||
self.assertIn(
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER, skipped[number]["reason"]
|
||||
)
|
||||
|
||||
def test_containers_cannot_receive_assignment_or_lease(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
res = self._alloc(containers, apply=True)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertNotEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertIsNone(res.get("assignment"))
|
||||
self.assertIsNone(res.get("selected"))
|
||||
skipped = {s["number"]: s for s in res["skipped"]}
|
||||
for number in (631, 652, 653, 655):
|
||||
self.assertEqual(
|
||||
skipped[number]["reason_code"],
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
)
|
||||
|
||||
leases = []
|
||||
if hasattr(self.db, "list_active_leases"):
|
||||
leases = self.db.list_active_leases(
|
||||
remote=REMOTE, org=ORG, repo=REPO
|
||||
)
|
||||
if not leases and hasattr(self.db, "list_leases"):
|
||||
leases = self.db.list_leases(remote=REMOTE, org=ORG, repo=REPO)
|
||||
for lease in leases or []:
|
||||
work_number = (
|
||||
lease.get("work_number") if isinstance(lease, dict) else None
|
||||
)
|
||||
self.assertNotIn(work_number, {631, 652, 653, 655})
|
||||
|
||||
def test_apply_selects_child_not_container(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
res = self._alloc(containers + [child], apply=True)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
self.assertEqual(res["assignment"]["work_number"], 637)
|
||||
|
||||
def test_fingerprint_stable_with_containers_present(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
inventory = containers + [child]
|
||||
fp_before = candidate_set_fingerprint(inventory)
|
||||
res = self._alloc(inventory, apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
# Allocator reports the same CAS fingerprint for the full candidate set.
|
||||
reported = res.get("candidate_set_fingerprint")
|
||||
self.assertEqual(reported, fp_before)
|
||||
# Re-fingerprint of the same inventory is byte-stable.
|
||||
self.assertEqual(candidate_set_fingerprint(inventory), fp_before)
|
||||
|
||||
def test_incidental_mentions_remain_eligible(self) -> None:
|
||||
ordinary = _issue(
|
||||
700,
|
||||
title="Document roadmap handoff conventions",
|
||||
body="Write runbook text about vision vs roadmap vs child issues.",
|
||||
)
|
||||
res = self._alloc([ordinary], apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["selected"]["number"], 700)
|
||||
self.assertEqual(res["skipped"], [])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,244 @@
|
||||
"""#855 AC4: an expired reviewer lease must not indefinitely protect an
|
||||
already-merged branch when no live claimant exists.
|
||||
|
||||
Two layers are covered:
|
||||
|
||||
* ``branch_cleanup_guard.assess_expired_reviewer_lease_reclaim`` — the pure,
|
||||
fail-closed reclaim decision. Every condition must be provably satisfied or
|
||||
the lease keeps protecting the branch.
|
||||
* ``gitea_mcp_server._collect_branch_ownership_records`` — the wiring that
|
||||
supplies authoritative evidence (PR merged state, owner-process liveness,
|
||||
competing ownership) to that decision, and flips an expired reviewer lease
|
||||
to reclaimable only under the full policy.
|
||||
|
||||
All inputs are fabricated; no real repository, lease, or credential is used.
|
||||
"""
|
||||
|
||||
import importlib
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
import branch_cleanup_guard
|
||||
|
||||
mcp_server = importlib.import_module("gitea_mcp_server")
|
||||
|
||||
FAKE_AUTH = "token fake"
|
||||
REMOTE = "prgs"
|
||||
ORG = "Scaled-Tech-Consulting"
|
||||
REPO = "Gitea-Tools"
|
||||
HOST = "gitea.prgs.cc"
|
||||
BRANCH = "feat/issue-638-webui-app-shell-phase1"
|
||||
PR_NUMBER = 818
|
||||
|
||||
|
||||
class TestAssessExpiredReviewerLeaseReclaim(unittest.TestCase):
|
||||
"""Pure fail-closed reclaim decision (#855 AC4)."""
|
||||
|
||||
def _call(self, **overrides):
|
||||
base = dict(
|
||||
role="reviewer",
|
||||
status="expired",
|
||||
pr_merged=True,
|
||||
owner_pid_alive=False,
|
||||
competing_active_claimant=False,
|
||||
)
|
||||
base.update(overrides)
|
||||
return branch_cleanup_guard.assess_expired_reviewer_lease_reclaim(**base)
|
||||
|
||||
def test_full_policy_satisfied_allows_reclaim(self):
|
||||
out = self._call()
|
||||
self.assertTrue(out["reclaim_allowed"])
|
||||
self.assertEqual(out["reasons"], [])
|
||||
self.assertEqual(out["decision"], "reclaim_expired_reviewer_lease")
|
||||
|
||||
def test_stale_dead_process_reviewer_also_reclaimable(self):
|
||||
out = self._call(status="stale_dead_process")
|
||||
self.assertTrue(out["reclaim_allowed"])
|
||||
|
||||
def test_non_reviewer_role_never_reclaims(self):
|
||||
for role in ("author", "merger", "controller", "reconciler", "unknown"):
|
||||
with self.subTest(role=role):
|
||||
out = self._call(role=role)
|
||||
self.assertFalse(out["reclaim_allowed"])
|
||||
self.assertTrue(out["reasons"])
|
||||
self.assertEqual(out["decision"], "keep_protecting")
|
||||
|
||||
def test_active_status_never_reclaims(self):
|
||||
out = self._call(status="active")
|
||||
self.assertFalse(out["reclaim_allowed"])
|
||||
|
||||
def test_pr_not_merged_blocks_reclaim(self):
|
||||
out = self._call(pr_merged=False)
|
||||
self.assertFalse(out["reclaim_allowed"])
|
||||
|
||||
def test_pr_merged_unknown_fails_closed(self):
|
||||
out = self._call(pr_merged=None)
|
||||
self.assertFalse(out["reclaim_allowed"])
|
||||
|
||||
def test_owner_process_alive_blocks_reclaim(self):
|
||||
out = self._call(owner_pid_alive=True)
|
||||
self.assertFalse(out["reclaim_allowed"])
|
||||
|
||||
def test_owner_liveness_unknown_fails_closed(self):
|
||||
out = self._call(owner_pid_alive=None)
|
||||
self.assertFalse(out["reclaim_allowed"])
|
||||
|
||||
def test_competing_active_claimant_blocks_reclaim(self):
|
||||
out = self._call(competing_active_claimant=True)
|
||||
self.assertFalse(out["reclaim_allowed"])
|
||||
|
||||
def test_competing_claimant_unknown_fails_closed(self):
|
||||
out = self._call(competing_active_claimant=None)
|
||||
self.assertFalse(out["reclaim_allowed"])
|
||||
|
||||
def test_reasons_never_leak_secrets(self):
|
||||
out = self._call(role="author")
|
||||
blob = " ".join(out["reasons"]).lower()
|
||||
self.assertNotIn("token", blob)
|
||||
self.assertNotIn("password", blob)
|
||||
|
||||
|
||||
class _FakeLease(dict):
|
||||
pass
|
||||
|
||||
|
||||
class TestCollectorExpiredReviewerReclaimWiring(unittest.TestCase):
|
||||
"""`_collect_branch_ownership_records` supplies authoritative evidence and
|
||||
flips an expired reviewer lease to reclaimable only under the full policy."""
|
||||
|
||||
def _run(
|
||||
self,
|
||||
*,
|
||||
lease_role="reviewer",
|
||||
lease_freshness="stale_dead_process",
|
||||
owner_pid_alive=False,
|
||||
pr_merged=True,
|
||||
extra_leases=None,
|
||||
worktree_on_branch=False,
|
||||
):
|
||||
lease = _FakeLease(
|
||||
role=lease_role,
|
||||
work_kind="pr",
|
||||
work_number=PR_NUMBER,
|
||||
branch=BRANCH,
|
||||
status="active",
|
||||
owner_pid=999999,
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
host=HOST,
|
||||
freshness={
|
||||
"freshness": lease_freshness,
|
||||
"owner_pid": 999999,
|
||||
"owner_pid_alive": owner_pid_alive,
|
||||
"expired_by_time": lease_freshness == "expired",
|
||||
},
|
||||
)
|
||||
leases = [lease] + list(extra_leases or [])
|
||||
|
||||
pr_payload = {
|
||||
"number": PR_NUMBER,
|
||||
"merged": pr_merged,
|
||||
"merged_at": "2026-07-23T00:00:00Z" if pr_merged else None,
|
||||
"head": {"ref": BRANCH},
|
||||
}
|
||||
|
||||
def fake_api_request(method, url, *a, **k):
|
||||
if method == "GET" and f"/pulls/{PR_NUMBER}" in url:
|
||||
return pr_payload
|
||||
raise AssertionError(f"unexpected api_request {method} {url}")
|
||||
|
||||
wt_entries = []
|
||||
if worktree_on_branch:
|
||||
wt_entries = [{"branch": BRANCH, "path": f"/x/branches/{BRANCH}"}]
|
||||
|
||||
with patch.object(
|
||||
mcp_server.lease_lifecycle,
|
||||
"list_active_leases",
|
||||
return_value={"leases": leases},
|
||||
), patch.object(
|
||||
mcp_server.control_plane_db, "get_db", return_value=object(), create=True
|
||||
), patch.object(
|
||||
mcp_server.issue_lock_store, "iter_lock_files", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server.worktree_cleanup_audit,
|
||||
"list_worktrees",
|
||||
return_value=wt_entries,
|
||||
), patch.object(
|
||||
mcp_server, "api_get_all", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "api_request", side_effect=fake_api_request
|
||||
):
|
||||
return mcp_server._collect_branch_ownership_records(
|
||||
remote=REMOTE,
|
||||
host=HOST,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
branch=BRANCH,
|
||||
pr_number=PR_NUMBER,
|
||||
project_root="/x",
|
||||
auth=FAKE_AUTH,
|
||||
base_api="https://gitea.prgs.cc/api/v1/repos/x/y",
|
||||
)
|
||||
|
||||
def _reviewer_records(self, bundle):
|
||||
return [
|
||||
rec
|
||||
for rec in bundle["records"]
|
||||
if rec.get("category")
|
||||
== branch_cleanup_guard.OWNERSHIP_CATEGORY_REVIEWER_LEASE
|
||||
]
|
||||
|
||||
def test_merged_dead_uncontested_reviewer_lease_is_reclaimable(self):
|
||||
bundle = self._run()
|
||||
self.assertFalse(bundle["inventory_error"])
|
||||
recs = self._reviewer_records(bundle)
|
||||
self.assertEqual(len(recs), 1)
|
||||
self.assertTrue(recs[0]["reclaim_allowed"])
|
||||
# And the guard consequently does not block deletion on it.
|
||||
ownership = branch_cleanup_guard.assess_active_branch_ownership(
|
||||
remote=REMOTE, org=ORG, repo=REPO, branch=BRANCH, host=HOST,
|
||||
records=bundle["records"],
|
||||
)
|
||||
self.assertFalse(ownership["block"])
|
||||
|
||||
def test_unmerged_pr_keeps_reviewer_lease_protective(self):
|
||||
bundle = self._run(pr_merged=False)
|
||||
recs = self._reviewer_records(bundle)
|
||||
self.assertEqual(len(recs), 1)
|
||||
self.assertFalse(recs[0]["reclaim_allowed"])
|
||||
ownership = branch_cleanup_guard.assess_active_branch_ownership(
|
||||
remote=REMOTE, org=ORG, repo=REPO, branch=BRANCH, host=HOST,
|
||||
records=bundle["records"],
|
||||
)
|
||||
self.assertTrue(ownership["block"])
|
||||
|
||||
def test_owner_process_alive_keeps_reviewer_lease_protective(self):
|
||||
bundle = self._run(owner_pid_alive=True, lease_freshness="expired")
|
||||
recs = self._reviewer_records(bundle)
|
||||
self.assertFalse(recs[0]["reclaim_allowed"])
|
||||
|
||||
def test_competing_worktree_binding_keeps_reviewer_lease_protective(self):
|
||||
bundle = self._run(worktree_on_branch=True)
|
||||
recs = self._reviewer_records(bundle)
|
||||
self.assertFalse(recs[0]["reclaim_allowed"])
|
||||
ownership = branch_cleanup_guard.assess_active_branch_ownership(
|
||||
remote=REMOTE, org=ORG, repo=REPO, branch=BRANCH, host=HOST,
|
||||
records=bundle["records"],
|
||||
)
|
||||
self.assertTrue(ownership["block"])
|
||||
|
||||
def test_expired_author_lease_never_reclaimed_by_reviewer_policy(self):
|
||||
bundle = self._run(lease_role="author")
|
||||
author_recs = [
|
||||
rec
|
||||
for rec in bundle["records"]
|
||||
if rec.get("category")
|
||||
== branch_cleanup_guard.OWNERSHIP_CATEGORY_AUTHOR_LEASE
|
||||
]
|
||||
self.assertEqual(len(author_recs), 1)
|
||||
self.assertFalse(author_recs[0]["reclaim_allowed"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,551 @@
|
||||
"""Merged-PR awareness for the worktree cleanup audit (#858).
|
||||
|
||||
Before #858 an ``issue_work`` worktree could never leave ``active_issue_work``:
|
||||
the audit had no PR linkage at all (``pr_number`` was structurally ``None``)
|
||||
and its only route to ``clean_stale_removable`` was a TTL derived from a
|
||||
``last_used_at`` that nothing ever populated. A merged, clean, unprotected
|
||||
worktree was therefore reported as active work forever, disagreeing with the
|
||||
PR-scoped reconciler.
|
||||
|
||||
These tests use fabricated temporary repositories and synthetic PR records
|
||||
only. Nothing here removes a worktree or deletes a branch.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parent.parent))
|
||||
|
||||
import merged_cleanup_reconcile as mcr # noqa: E402
|
||||
import worktree_cleanup_audit as wca # noqa: E402
|
||||
|
||||
|
||||
MERGED_BRANCH = "feat/issue-777-timeline"
|
||||
MERGED_PATH = "/repo/branches/issue-777-timeline"
|
||||
HEAD_SHA = "a" * 40
|
||||
|
||||
|
||||
def _pr(number, branch, *, merged=True, sha=HEAD_SHA, state=None):
|
||||
"""Synthetic Gitea PR payload."""
|
||||
return {
|
||||
"number": number,
|
||||
"head": {"ref": branch, "sha": sha},
|
||||
"merged_at": "2026-07-24T01:00:00Z" if merged else None,
|
||||
"state": state or ("closed" if merged else "open"),
|
||||
}
|
||||
|
||||
|
||||
def _porcelain(*entries):
|
||||
out = []
|
||||
for path, branch, sha in entries:
|
||||
out.append(f"worktree {path}")
|
||||
out.append(f"HEAD {sha}")
|
||||
if branch is None:
|
||||
out.append("detached")
|
||||
else:
|
||||
out.append(f"branch refs/heads/{branch}")
|
||||
out.append("")
|
||||
return "\n".join(out)
|
||||
|
||||
|
||||
class _AuditHarness(unittest.TestCase):
|
||||
"""Runs audit_branches_directory over a fabricated worktree listing."""
|
||||
|
||||
PORCELAIN = _porcelain(
|
||||
("/repo", "master", "f" * 40),
|
||||
(MERGED_PATH, MERGED_BRANCH, HEAD_SHA),
|
||||
)
|
||||
|
||||
def run_audit(self, *, dirty_paths=(), contained=True, **kwargs):
|
||||
def fake_dirty(path):
|
||||
if path in dirty_paths:
|
||||
return {"exists": True, "dirty": True, "dirty_files": [" M x.py"]}
|
||||
return {"exists": True, "dirty": False, "dirty_files": []}
|
||||
|
||||
with patch.object(
|
||||
wca, "list_worktrees",
|
||||
return_value=wca.parse_worktree_porcelain(self.PORCELAIN),
|
||||
), patch.object(
|
||||
wca, "read_worktree_dirty", side_effect=fake_dirty
|
||||
), patch.object(
|
||||
wca, "git_worktree_list", return_value="(mocked)"
|
||||
), patch.object(
|
||||
wca, "is_head_ancestor_of_ref", return_value=contained
|
||||
):
|
||||
report = wca.audit_branches_directory("/repo", **kwargs)
|
||||
return {wt["path"]: wt for wt in report["worktrees"]}, report
|
||||
|
||||
def merged_audit(self, **kwargs):
|
||||
kwargs.setdefault("pr_index", wca.build_pr_index([_pr(849, MERGED_BRANCH)]))
|
||||
kwargs.setdefault("master_ref", "prgs/master")
|
||||
return self.run_audit(**kwargs)
|
||||
|
||||
|
||||
class TestMergedWorktreeBecomesRemovable(_AuditHarness):
|
||||
def test_clean_merged_issue_worktree_is_linked_and_removable(self):
|
||||
by_path, report = self.merged_audit()
|
||||
entry = by_path[MERGED_PATH]
|
||||
|
||||
self.assertEqual(entry["classification"], wca.CLASS_CLEAN_STALE_REMOVABLE)
|
||||
self.assertTrue(entry["removable"])
|
||||
self.assertEqual(entry["merged_pr_linkage"]["status"], wca.LINKAGE_MERGED)
|
||||
self.assertEqual(entry["merged_pr_cleanup"]["block_reasons"], [])
|
||||
self.assertIn(MERGED_PATH, [c["path"] for c in report["removable_candidates"]])
|
||||
|
||||
def test_pr_number_populated_from_authoritative_linkage(self):
|
||||
by_path, _ = self.merged_audit()
|
||||
self.assertEqual(by_path[MERGED_PATH]["pr_number"], 849)
|
||||
|
||||
def test_regression_without_pr_evidence_stays_active_issue_work(self):
|
||||
"""The pre-#858 behaviour, still correct when no PR state is supplied."""
|
||||
by_path, _ = self.run_audit()
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertEqual(entry["classification"], wca.CLASS_ACTIVE_ISSUE_WORK)
|
||||
self.assertFalse(entry["removable"])
|
||||
self.assertIsNone(entry["pr_number"])
|
||||
|
||||
|
||||
class TestProtectiveSignalsSurvive(_AuditHarness):
|
||||
def test_open_pr_worktree_is_not_removable(self):
|
||||
index = wca.build_pr_index([_pr(900, MERGED_BRANCH, merged=False)])
|
||||
by_path, _ = self.run_audit(
|
||||
pr_index=index,
|
||||
master_ref="prgs/master",
|
||||
open_pr_branches={MERGED_BRANCH},
|
||||
)
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertEqual(entry["classification"], wca.CLASS_ACTIVE_OPEN_PR)
|
||||
self.assertFalse(entry["removable"])
|
||||
# linkage still reports the owning PR, it just is not merge proof
|
||||
self.assertEqual(entry["merged_pr_linkage"]["status"], wca.LINKAGE_OPEN)
|
||||
self.assertEqual(entry["pr_number"], 900)
|
||||
|
||||
def test_dirty_tracked_worktree_is_not_removable(self):
|
||||
by_path, _ = self.merged_audit(dirty_paths=(MERGED_PATH,))
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertEqual(entry["classification"], wca.CLASS_DIRTY_LOCAL)
|
||||
self.assertFalse(entry["removable"])
|
||||
self.assertIn(
|
||||
"worktree has uncommitted changes",
|
||||
entry["merged_pr_cleanup"]["block_reasons"],
|
||||
)
|
||||
|
||||
def test_untracked_only_worktree_is_not_removable(self):
|
||||
"""``git status --porcelain`` reports untracked files as dirty too."""
|
||||
def untracked(path):
|
||||
if path == MERGED_PATH:
|
||||
return {"exists": True, "dirty": True, "dirty_files": ["?? scratch.txt"]}
|
||||
return {"exists": True, "dirty": False, "dirty_files": []}
|
||||
|
||||
with patch.object(
|
||||
wca, "list_worktrees",
|
||||
return_value=wca.parse_worktree_porcelain(self.PORCELAIN),
|
||||
), patch.object(
|
||||
wca, "read_worktree_dirty", side_effect=untracked
|
||||
), patch.object(
|
||||
wca, "git_worktree_list", return_value="(mocked)"
|
||||
), patch.object(
|
||||
wca, "is_head_ancestor_of_ref", return_value=True
|
||||
):
|
||||
report = wca.audit_branches_directory(
|
||||
"/repo",
|
||||
pr_index=wca.build_pr_index([_pr(849, MERGED_BRANCH)]),
|
||||
master_ref="prgs/master",
|
||||
)
|
||||
entry = {wt["path"]: wt for wt in report["worktrees"]}[MERGED_PATH]
|
||||
self.assertEqual(entry["classification"], wca.CLASS_DIRTY_LOCAL)
|
||||
self.assertFalse(entry["removable"])
|
||||
|
||||
def test_active_lease_by_issue_number_is_protective(self):
|
||||
by_path, _ = self.merged_audit(leased_issue_numbers={777})
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertTrue(entry["has_active_lease"])
|
||||
self.assertEqual(entry["classification"], wca.CLASS_ACTIVE_ISSUE_WORK)
|
||||
self.assertFalse(entry["removable"])
|
||||
|
||||
def test_active_lease_by_branch_is_protective(self):
|
||||
by_path, _ = self.merged_audit(leased_branches={MERGED_BRANCH})
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertTrue(entry["has_active_lease"])
|
||||
self.assertFalse(entry["removable"])
|
||||
|
||||
def test_active_issue_lock_is_protective(self):
|
||||
by_path, _ = self.merged_audit(active_issue_branches={MERGED_BRANCH})
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertTrue(entry["has_active_issue_lock"])
|
||||
self.assertEqual(entry["classification"], wca.CLASS_ACTIVE_ISSUE_WORK)
|
||||
self.assertFalse(entry["removable"])
|
||||
|
||||
def test_live_session_worktree_is_protective(self):
|
||||
by_path, _ = self.merged_audit(live_session_paths={MERGED_PATH})
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertTrue(entry["has_live_session"])
|
||||
self.assertEqual(entry["classification"], wca.CLASS_ACTIVE_ISSUE_WORK)
|
||||
self.assertFalse(entry["removable"])
|
||||
|
||||
def test_head_not_contained_in_master_is_not_removable(self):
|
||||
by_path, _ = self.merged_audit(contained=False)
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertEqual(entry["classification"], wca.CLASS_ACTIVE_ISSUE_WORK)
|
||||
self.assertFalse(entry["removable"])
|
||||
self.assertIn(
|
||||
"worktree head is not contained in authoritative master "
|
||||
"(unmerged commits remain)",
|
||||
entry["merged_pr_cleanup"]["block_reasons"],
|
||||
)
|
||||
|
||||
def test_unknown_containment_fails_closed(self):
|
||||
by_path, _ = self.merged_audit(contained=None)
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertFalse(entry["removable"])
|
||||
self.assertIn(
|
||||
"containment of the worktree head in master is unknown",
|
||||
entry["merged_pr_cleanup"]["block_reasons"],
|
||||
)
|
||||
|
||||
def test_missing_master_ref_fails_closed(self):
|
||||
by_path, _ = self.run_audit(
|
||||
pr_index=wca.build_pr_index([_pr(849, MERGED_BRANCH)])
|
||||
)
|
||||
self.assertFalse(by_path[MERGED_PATH]["removable"])
|
||||
|
||||
def test_unmerged_owning_pr_is_not_removable(self):
|
||||
index = wca.build_pr_index([_pr(901, MERGED_BRANCH, merged=False)])
|
||||
by_path, _ = self.run_audit(pr_index=index, master_ref="prgs/master")
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertFalse(entry["removable"])
|
||||
self.assertIn(
|
||||
"owning PR #901 is not merged",
|
||||
entry["merged_pr_cleanup"]["block_reasons"],
|
||||
)
|
||||
|
||||
def test_control_checkout_is_never_removable(self):
|
||||
by_path, _ = self.merged_audit()
|
||||
control = by_path["/repo"]
|
||||
self.assertTrue(control["is_protected"])
|
||||
self.assertEqual(control["classification"], wca.CLASS_UNSAFE_UNKNOWN)
|
||||
self.assertFalse(control["removable"])
|
||||
|
||||
def test_control_checkout_not_removable_even_if_linked_and_merged(self):
|
||||
"""A merged PR on the control checkout must not unlock removal."""
|
||||
porcelain = _porcelain(("/repo", MERGED_BRANCH, HEAD_SHA))
|
||||
with patch.object(
|
||||
wca, "list_worktrees", return_value=wca.parse_worktree_porcelain(porcelain)
|
||||
), patch.object(
|
||||
wca, "read_worktree_dirty",
|
||||
return_value={"exists": True, "dirty": False, "dirty_files": []},
|
||||
), patch.object(
|
||||
wca, "git_worktree_list", return_value="(mocked)"
|
||||
), patch.object(
|
||||
wca, "is_head_ancestor_of_ref", return_value=True
|
||||
):
|
||||
report = wca.audit_branches_directory(
|
||||
"/repo",
|
||||
pr_index=wca.build_pr_index([_pr(849, MERGED_BRANCH)]),
|
||||
master_ref="prgs/master",
|
||||
)
|
||||
entry = report["worktrees"][0]
|
||||
self.assertEqual(entry["classification"], wca.CLASS_UNSAFE_UNKNOWN)
|
||||
self.assertFalse(entry["removable"])
|
||||
|
||||
|
||||
class TestAmbiguousLinkageFailsClosed(_AuditHarness):
|
||||
def test_competing_prs_on_one_branch_fail_closed(self):
|
||||
index = wca.build_pr_index(
|
||||
[_pr(849, MERGED_BRANCH), _pr(860, MERGED_BRANCH)]
|
||||
)
|
||||
by_path, _ = self.run_audit(pr_index=index, master_ref="prgs/master")
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertEqual(entry["merged_pr_linkage"]["status"], wca.LINKAGE_AMBIGUOUS)
|
||||
self.assertIsNone(entry["pr_number"])
|
||||
self.assertEqual(entry["classification"], wca.CLASS_ACTIVE_ISSUE_WORK)
|
||||
self.assertFalse(entry["removable"])
|
||||
|
||||
def test_merged_plus_open_pr_on_one_branch_fails_closed(self):
|
||||
index = wca.build_pr_index(
|
||||
[_pr(849, MERGED_BRANCH), _pr(861, MERGED_BRANCH, merged=False)]
|
||||
)
|
||||
by_path, _ = self.run_audit(pr_index=index, master_ref="prgs/master")
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertEqual(entry["merged_pr_linkage"]["status"], wca.LINKAGE_AMBIGUOUS)
|
||||
self.assertFalse(entry["removable"])
|
||||
|
||||
def test_no_owning_pr_fails_closed(self):
|
||||
by_path, _ = self.run_audit(
|
||||
pr_index=wca.build_pr_index([_pr(849, "feat/other-branch")]),
|
||||
master_ref="prgs/master",
|
||||
)
|
||||
entry = by_path[MERGED_PATH]
|
||||
self.assertEqual(entry["merged_pr_linkage"]["status"], wca.LINKAGE_NONE)
|
||||
self.assertFalse(entry["removable"])
|
||||
|
||||
def test_malformed_pr_records_are_dropped_not_guessed(self):
|
||||
index = wca.build_pr_index(
|
||||
[
|
||||
{"number": None, "head": {"ref": MERGED_BRANCH}},
|
||||
{"number": 5, "head": {}},
|
||||
{"number": "not-an-int", "head": {"ref": MERGED_BRANCH}},
|
||||
]
|
||||
)
|
||||
self.assertEqual(index, {})
|
||||
self.assertEqual(
|
||||
wca.resolve_owning_pr(branch=MERGED_BRANCH, pr_index=index)["status"],
|
||||
wca.LINKAGE_NONE,
|
||||
)
|
||||
|
||||
def test_detached_worktree_has_no_branch_linkage(self):
|
||||
self.assertEqual(
|
||||
wca.resolve_owning_pr(branch=None, pr_index={})["status"],
|
||||
wca.LINKAGE_UNKNOWN,
|
||||
)
|
||||
|
||||
|
||||
class TestUnrelatedClassificationsUnchanged(unittest.TestCase):
|
||||
"""Non-issue_work worktrees keep their pre-#858 classifications."""
|
||||
|
||||
PORCELAIN = _porcelain(
|
||||
("/repo", "master", "f" * 40),
|
||||
("/repo/branches/review-pr42", "review-pr42", "2" * 40),
|
||||
("/repo/branches/baseline-master-x", "baseline-master-x", "3" * 40),
|
||||
("/repo/branches/conflict-fix-pr50", "conflict-fix-pr50", "4" * 40),
|
||||
("/repo/branches/review-pr99", None, "5" * 40),
|
||||
)
|
||||
|
||||
def _audit(self, **kwargs):
|
||||
with patch.object(
|
||||
wca, "list_worktrees",
|
||||
return_value=wca.parse_worktree_porcelain(self.PORCELAIN),
|
||||
), patch.object(
|
||||
wca, "read_worktree_dirty",
|
||||
return_value={"exists": True, "dirty": False, "dirty_files": []},
|
||||
), patch.object(
|
||||
wca, "git_worktree_list", return_value="(mocked)"
|
||||
), patch.object(
|
||||
wca, "is_head_ancestor_of_ref", return_value=True
|
||||
):
|
||||
report = wca.audit_branches_directory("/repo", **kwargs)
|
||||
return {wt["path"]: wt for wt in report["worktrees"]}
|
||||
|
||||
def test_classifications_identical_with_and_without_pr_evidence(self):
|
||||
without = self._audit()
|
||||
with_evidence = self._audit(
|
||||
pr_index=wca.build_pr_index([_pr(849, MERGED_BRANCH)]),
|
||||
master_ref="prgs/master",
|
||||
)
|
||||
self.assertEqual(
|
||||
{p: e["classification"] for p, e in without.items()},
|
||||
{p: e["classification"] for p, e in with_evidence.items()},
|
||||
)
|
||||
|
||||
def test_lease_on_issue_does_not_capture_similarly_named_scratch_trees(self):
|
||||
"""A lease on issue 777 protects issue work, not baseline/review trees."""
|
||||
porcelain = _porcelain(
|
||||
("/repo/branches/baseline-master-issue-777", "baseline-issue-777", "7" * 40),
|
||||
("/repo/branches/issue-777-timeline", MERGED_BRANCH, HEAD_SHA),
|
||||
)
|
||||
with patch.object(
|
||||
wca, "list_worktrees", return_value=wca.parse_worktree_porcelain(porcelain)
|
||||
), patch.object(
|
||||
wca, "read_worktree_dirty",
|
||||
return_value={"exists": True, "dirty": False, "dirty_files": []},
|
||||
), patch.object(
|
||||
wca, "git_worktree_list", return_value="(mocked)"
|
||||
), patch.object(
|
||||
wca, "is_head_ancestor_of_ref", return_value=True
|
||||
):
|
||||
report = wca.audit_branches_directory(
|
||||
"/repo",
|
||||
pr_index=wca.build_pr_index([_pr(849, MERGED_BRANCH)]),
|
||||
master_ref="prgs/master",
|
||||
leased_issue_numbers={777},
|
||||
)
|
||||
by_path = {wt["path"]: wt for wt in report["worktrees"]}
|
||||
|
||||
baseline = by_path["/repo/branches/baseline-master-issue-777"]
|
||||
self.assertFalse(baseline["has_active_lease"])
|
||||
self.assertEqual(baseline["classification"], wca.CLASS_CLEAN_STALE_REMOVABLE)
|
||||
|
||||
issue_work = by_path["/repo/branches/issue-777-timeline"]
|
||||
self.assertTrue(issue_work["has_active_lease"])
|
||||
self.assertFalse(issue_work["removable"])
|
||||
|
||||
def test_review_and_baseline_still_removable(self):
|
||||
by_path = self._audit(
|
||||
pr_index=wca.build_pr_index([]), master_ref="prgs/master"
|
||||
)
|
||||
self.assertEqual(
|
||||
by_path["/repo/branches/review-pr42"]["classification"],
|
||||
wca.CLASS_CLEAN_STALE_REMOVABLE,
|
||||
)
|
||||
self.assertEqual(
|
||||
by_path["/repo/branches/baseline-master-x"]["classification"],
|
||||
wca.CLASS_CLEAN_STALE_REMOVABLE,
|
||||
)
|
||||
self.assertEqual(
|
||||
by_path["/repo/branches/review-pr99"]["classification"],
|
||||
wca.CLASS_DETACHED_REVIEW_LEFTOVER,
|
||||
)
|
||||
|
||||
def test_conflict_fix_ttl_behaviour_unchanged(self):
|
||||
"""conflict_fix still needs only TTL expiry; #858 did not touch it."""
|
||||
self.assertEqual(
|
||||
wca.classify_worktree(
|
||||
workflow_type=wca.WORKFLOW_CONFLICT_FIX,
|
||||
is_dirty=False,
|
||||
ttl_expired=True,
|
||||
),
|
||||
wca.CLASS_CLEAN_STALE_REMOVABLE,
|
||||
)
|
||||
self.assertEqual(
|
||||
wca.classify_worktree(
|
||||
workflow_type=wca.WORKFLOW_CONFLICT_FIX,
|
||||
is_dirty=False,
|
||||
ttl_expired=False,
|
||||
),
|
||||
wca.CLASS_ACTIVE_ISSUE_WORK,
|
||||
)
|
||||
|
||||
def test_issue_work_ttl_alone_no_longer_grants_removal(self):
|
||||
"""Age is not landing proof: TTL alone must not reclaim issue work."""
|
||||
self.assertEqual(
|
||||
wca.classify_worktree(
|
||||
workflow_type=wca.WORKFLOW_ISSUE_WORK,
|
||||
is_dirty=False,
|
||||
ttl_expired=True,
|
||||
),
|
||||
wca.CLASS_ACTIVE_ISSUE_WORK,
|
||||
)
|
||||
|
||||
|
||||
class TestAssessorPerformsNoDeletion(_AuditHarness):
|
||||
def test_audit_never_removes_a_worktree(self):
|
||||
with patch.object(wca, "remove_worktree") as removal:
|
||||
self.merged_audit()
|
||||
removal.assert_not_called()
|
||||
|
||||
def test_audit_shells_out_to_no_destructive_git_command(self):
|
||||
seen = []
|
||||
real_run = subprocess.run
|
||||
|
||||
def recording_run(cmd, *args, **kwargs):
|
||||
seen.append(cmd)
|
||||
return real_run(["true"], *args, **kwargs)
|
||||
|
||||
with patch.object(subprocess, "run", side_effect=recording_run):
|
||||
wca.audit_branches_directory("/nonexistent-repo-for-audit")
|
||||
|
||||
joined = [" ".join(c) if isinstance(c, list) else str(c) for c in seen]
|
||||
for cmd in joined:
|
||||
self.assertNotIn("worktree remove", cmd)
|
||||
self.assertNotIn("branch -D", cmd)
|
||||
self.assertNotIn("push", cmd)
|
||||
|
||||
|
||||
class TestAgreementWithPrScopedReconciler(unittest.TestCase):
|
||||
"""The audit and merged_cleanup_reconcile must agree on identical input.
|
||||
|
||||
Uses a real throwaway git repository so containment is computed by git
|
||||
rather than asserted. Nothing outside the temporary directory is touched.
|
||||
"""
|
||||
|
||||
def _git(self, *args):
|
||||
subprocess.run(
|
||||
["git", "-C", self.root, *args],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.root = os.path.realpath(self._tmp.name)
|
||||
self._git("init", "-b", "master", ".")
|
||||
self._git("config", "user.email", "[email protected]")
|
||||
self._git("config", "user.name", "Test")
|
||||
with open(os.path.join(self.root, "seed.txt"), "w") as fh:
|
||||
fh.write("seed\n")
|
||||
self._git("add", "seed.txt")
|
||||
self._git("commit", "-m", "seed")
|
||||
|
||||
self.branch = "feat/issue-777-timeline"
|
||||
self._git("checkout", "-b", self.branch)
|
||||
with open(os.path.join(self.root, "feature.txt"), "w") as fh:
|
||||
fh.write("feature\n")
|
||||
self._git("add", "feature.txt")
|
||||
self._git("commit", "-m", "feature")
|
||||
self.head_sha = subprocess.run(
|
||||
["git", "-C", self.root, "rev-parse", "HEAD"],
|
||||
capture_output=True, text=True, check=True,
|
||||
).stdout.strip()
|
||||
self._git("checkout", "master")
|
||||
self._git("merge", "--no-ff", "-m", "merge feature", self.branch)
|
||||
|
||||
self.worktree = os.path.join(self.root, "branches", "issue-777-timeline")
|
||||
self._git("worktree", "add", self.worktree, self.branch)
|
||||
|
||||
def tearDown(self):
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _pr_index(self):
|
||||
return wca.build_pr_index(
|
||||
[
|
||||
{
|
||||
"number": 849,
|
||||
"head": {"ref": self.branch, "sha": self.head_sha},
|
||||
"merged_at": "2026-07-24T01:00:00Z",
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
def _audit_entry(self):
|
||||
report = wca.audit_branches_directory(
|
||||
self.root, pr_index=self._pr_index(), master_ref="master"
|
||||
)
|
||||
return next(wt for wt in report["worktrees"] if wt["path"] == self.worktree)
|
||||
|
||||
def _reconciler_entry(self):
|
||||
return mcr.assess_local_worktree_cleanup(
|
||||
pr_number=849,
|
||||
head_branch=self.branch,
|
||||
merged=True,
|
||||
worktree_state=mcr.resolve_cleanup_worktree_state(
|
||||
project_root=self.root,
|
||||
head_branch=self.branch,
|
||||
issue_number=777,
|
||||
pr_head_sha=self.head_sha,
|
||||
target_ref="master",
|
||||
),
|
||||
active_lock=False,
|
||||
)
|
||||
|
||||
def test_both_assessors_agree_the_worktree_is_safe(self):
|
||||
audit_entry = self._audit_entry()
|
||||
reconciler = self._reconciler_entry()
|
||||
|
||||
self.assertTrue(reconciler["safe_to_remove_worktree"], reconciler)
|
||||
self.assertTrue(audit_entry["removable"], audit_entry)
|
||||
self.assertEqual(audit_entry["pr_number"], reconciler["pr_number"])
|
||||
self.assertEqual(audit_entry["merged_pr_cleanup"]["block_reasons"], [])
|
||||
self.assertEqual(reconciler["block_reasons"], [])
|
||||
|
||||
def test_both_assessors_agree_a_dirty_worktree_is_unsafe(self):
|
||||
with open(os.path.join(self.worktree, "feature.txt"), "a") as fh:
|
||||
fh.write("local edit\n")
|
||||
|
||||
audit_entry = self._audit_entry()
|
||||
reconciler = self._reconciler_entry()
|
||||
|
||||
self.assertFalse(audit_entry["removable"])
|
||||
self.assertFalse(reconciler["safe_to_remove_worktree"])
|
||||
|
||||
def test_worktree_still_present_after_audit(self):
|
||||
self._audit_entry()
|
||||
self.assertTrue(os.path.isdir(self.worktree))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,630 @@
|
||||
"""Durable linked-issue lock head refresh + merge-sync dead-session recovery (#871).
|
||||
|
||||
``gitea_update_pr_branch_by_merge`` advances a PR's *remote* head but historically
|
||||
never advanced the linked durable issue lock's recorded head. After the owning
|
||||
session died the drifted lock became unrecoverable and no further synchronization
|
||||
was possible (PR #866 / issue #855).
|
||||
|
||||
Two halves are covered:
|
||||
|
||||
* the write-side refresh (``issue_lock_store.assess/apply_durable_lock_head_refresh``)
|
||||
that records the new synced head under compare-and-swap with read-after-write; and
|
||||
* the read-side recovery relation (``issue_lock_recovery`` +
|
||||
``issue_lock_worktree.read_merge_sync_provenance``) that lets a dead-session lock
|
||||
whose recorded head is a merge-sync *ancestor* of the live PR head be recovered —
|
||||
and nothing else.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import issue_lock_recovery # noqa: E402
|
||||
import issue_lock_store # noqa: E402
|
||||
import issue_lock_worktree # noqa: E402
|
||||
|
||||
ISSUE = 8710
|
||||
PR_NUMBER = 8711
|
||||
BRANCH = f"fix/issue-{ISSUE}-durable-lock-head-refresh"
|
||||
IDENTITY = "example-user"
|
||||
PROFILE = "example-author"
|
||||
OLD = "a" * 40
|
||||
NEW1 = "b" * 40
|
||||
NEW2 = "c" * 40
|
||||
BASE = "d" * 40
|
||||
REMOTE = "prgs"
|
||||
ORG = "ExampleOrg"
|
||||
REPO = "ExampleRepo"
|
||||
|
||||
|
||||
def dead_pid() -> int:
|
||||
proc = subprocess.Popen([sys.executable, "-c", "pass"])
|
||||
proc.wait()
|
||||
return proc.pid
|
||||
|
||||
|
||||
def future_ts(hours: int = 4) -> str:
|
||||
return (
|
||||
(datetime.now(timezone.utc) + timedelta(hours=hours))
|
||||
.isoformat()
|
||||
.replace("+00:00", "Z")
|
||||
)
|
||||
|
||||
|
||||
def _git(cwd, *args):
|
||||
return subprocess.run(
|
||||
["git", "-C", cwd, *args],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
)
|
||||
|
||||
|
||||
def _rev(cwd, ref="HEAD") -> str:
|
||||
return _git(cwd, "rev-parse", ref).stdout.strip()
|
||||
|
||||
|
||||
def build_merge_sync_repo(tmp: str) -> dict:
|
||||
"""Build a repo where a feature branch was synced by merging master in.
|
||||
|
||||
Returns a dict with the prior (branch) head, the synced merge-commit head,
|
||||
the master tip, plus a rebase-style linear descendant and an unrelated head.
|
||||
"""
|
||||
_git(tmp, "init", "-q", "-b", "master")
|
||||
_git(tmp, "config", "user.email", "[email protected]")
|
||||
_git(tmp, "config", "user.name", "T")
|
||||
Path(tmp, "base.txt").write_text("base\n")
|
||||
_git(tmp, "add", "-A")
|
||||
_git(tmp, "commit", "-q", "-m", "root")
|
||||
|
||||
# Feature branch cut from root, one commit — this is the PRIOR/recorded head.
|
||||
_git(tmp, "checkout", "-q", "-b", BRANCH)
|
||||
Path(tmp, "feature.txt").write_text("feature\n")
|
||||
_git(tmp, "add", "-A")
|
||||
_git(tmp, "commit", "-q", "-m", "feature work")
|
||||
prior = _rev(tmp)
|
||||
|
||||
# Master advances (the base the sync will merge in).
|
||||
_git(tmp, "checkout", "-q", "master")
|
||||
Path(tmp, "base.txt").write_text("base\nmore\n")
|
||||
_git(tmp, "add", "-A")
|
||||
_git(tmp, "commit", "-q", "-m", "master advance")
|
||||
master_tip = _rev(tmp)
|
||||
|
||||
# Sync: merge master INTO the feature branch → merge commit, first parent = prior.
|
||||
_git(tmp, "checkout", "-q", BRANCH)
|
||||
_git(tmp, "merge", "-q", "--no-ff", "-m", "Merge master into feature", "master")
|
||||
synced = _rev(tmp)
|
||||
|
||||
# A plain linear descendant of prior (NOT a merge) — a rebase/extra-commit shape.
|
||||
_git(tmp, "checkout", "-q", "-b", "linear-branch", prior)
|
||||
Path(tmp, "extra.txt").write_text("extra\n")
|
||||
_git(tmp, "add", "-A")
|
||||
_git(tmp, "commit", "-q", "-m", "extra linear commit")
|
||||
linear = _rev(tmp)
|
||||
|
||||
# An unrelated root (force-push / rewritten history shape).
|
||||
unrelated_dir = tempfile.mkdtemp()
|
||||
_git(unrelated_dir, "init", "-q", "-b", "x")
|
||||
_git(unrelated_dir, "config", "user.email", "[email protected]")
|
||||
_git(unrelated_dir, "config", "user.name", "T")
|
||||
Path(unrelated_dir, "z.txt").write_text("z\n")
|
||||
_git(unrelated_dir, "add", "-A")
|
||||
_git(unrelated_dir, "commit", "-q", "-m", "unrelated")
|
||||
unrelated = _rev(unrelated_dir)
|
||||
|
||||
# Leave the worktree checked out on the feature branch at the PRIOR head, as
|
||||
# a dead author session that never advanced would have left it.
|
||||
_git(tmp, "checkout", "-q", BRANCH)
|
||||
_git(tmp, "reset", "-q", "--hard", prior)
|
||||
|
||||
return {
|
||||
"prior": prior,
|
||||
"master_tip": master_tip,
|
||||
"synced": synced,
|
||||
"linear": linear,
|
||||
"unrelated": unrelated,
|
||||
}
|
||||
|
||||
|
||||
# ─────────────────────────── write-side refresh ───────────────────────────
|
||||
|
||||
|
||||
class TestDurableLockHeadRefresh(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.lock_dir = tempfile.mkdtemp()
|
||||
self.wt = tempfile.mkdtemp()
|
||||
lock_data = {
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": BRANCH,
|
||||
"worktree_path": self.wt,
|
||||
"remote": REMOTE,
|
||||
"org": ORG,
|
||||
"repo": REPO,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
"work_lease": {
|
||||
"operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE,
|
||||
"issue_number": ISSUE,
|
||||
"branch": BRANCH,
|
||||
"worktree_path": self.wt,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
"expires_at": future_ts(),
|
||||
},
|
||||
}
|
||||
issue_lock_store.bind_session_lock(lock_data, lock_dir=self.lock_dir)
|
||||
|
||||
def _apply(self, **over):
|
||||
kw = dict(
|
||||
remote=REMOTE, org=ORG, repo=REPO, issue_number=ISSUE,
|
||||
branch_name=BRANCH, worktree_path=self.wt, pr_number=PR_NUMBER,
|
||||
identity=IDENTITY, profile=PROFILE, current_pid=os.getpid(),
|
||||
expected_old_head=OLD, new_head=NEW1, synced_at=future_ts(0),
|
||||
base_head=BASE, lock_dir=self.lock_dir,
|
||||
)
|
||||
kw.update(over)
|
||||
return issue_lock_store.apply_durable_lock_head_refresh(**kw)
|
||||
|
||||
def _load(self):
|
||||
return issue_lock_store.load_issue_lock(
|
||||
remote=REMOTE, org=ORG, repo=REPO, issue_number=ISSUE,
|
||||
lock_dir=self.lock_dir,
|
||||
)
|
||||
|
||||
def test_first_sync_updates_recorded_head(self):
|
||||
"""AC1: first base sync writes the resulting head to the durable lock."""
|
||||
res = self._apply()
|
||||
self.assertTrue(res["refreshed"], res["reasons"])
|
||||
self.assertTrue(res["read_after_write_ok"])
|
||||
self.assertEqual(self._load().get("synced_pr_head"), NEW1)
|
||||
|
||||
def test_second_sync_after_master_advance(self):
|
||||
"""AC2: a later master advance permits a second sanctioned sync."""
|
||||
self.assertTrue(self._apply()["refreshed"])
|
||||
res2 = self._apply(expected_old_head=NEW1, new_head=NEW2)
|
||||
self.assertTrue(res2["refreshed"], res2["reasons"])
|
||||
self.assertEqual(self._load().get("synced_pr_head"), NEW2)
|
||||
history = self._load().get("branch_sync_history")
|
||||
self.assertEqual(len(history), 2)
|
||||
self.assertEqual(history[0]["last_synced_pr_head"], NEW1)
|
||||
self.assertEqual(history[1]["prior_pr_head"], NEW1)
|
||||
|
||||
def test_cas_detects_concurrent_head_change(self):
|
||||
"""AC6: CAS refuses when the recorded synced head is not the old head."""
|
||||
self.assertTrue(self._apply()["refreshed"]) # recorded head now NEW1
|
||||
# A second sync claiming the old head is still OLD must fail closed.
|
||||
res = self._apply(expected_old_head=OLD, new_head=NEW2)
|
||||
self.assertFalse(res["refreshed"])
|
||||
self.assertTrue(any("CAS" in r or "concurrent" in r for r in res["reasons"]))
|
||||
self.assertEqual(self._load().get("synced_pr_head"), NEW1)
|
||||
|
||||
def test_wrong_issue_fails_closed(self):
|
||||
res = self._apply(issue_number=999999)
|
||||
self.assertFalse(res["refreshed"])
|
||||
|
||||
def test_wrong_branch_fails_closed(self):
|
||||
res = self._apply(branch_name="fix/issue-8710-wrong")
|
||||
self.assertFalse(res["refreshed"])
|
||||
|
||||
def test_wrong_repo_fails_closed(self):
|
||||
res = self._apply(repo="OtherRepo")
|
||||
self.assertFalse(res["refreshed"])
|
||||
|
||||
def test_wrong_identity_fails_closed(self):
|
||||
res = self._apply(identity="intruder")
|
||||
self.assertFalse(res["refreshed"])
|
||||
|
||||
def test_wrong_profile_fails_closed(self):
|
||||
res = self._apply(profile="prgs-reviewer")
|
||||
self.assertFalse(res["refreshed"])
|
||||
|
||||
def test_foreign_session_fails_closed(self):
|
||||
"""A refresh is not a recovery: the current process must own the lock."""
|
||||
path = issue_lock_store.lock_file_path(
|
||||
remote=REMOTE, org=ORG, repo=REPO, issue_number=ISSUE,
|
||||
lock_dir=self.lock_dir,
|
||||
)
|
||||
rec = issue_lock_store.read_lock_file(path)
|
||||
rec["session_pid"] = dead_pid()
|
||||
rec["pid"] = rec["session_pid"]
|
||||
issue_lock_store.save_lock_file(path, rec)
|
||||
res = self._apply()
|
||||
self.assertFalse(res["refreshed"])
|
||||
self.assertTrue(any("current session" in r or "live owner" in r for r in res["reasons"]))
|
||||
|
||||
def test_new_equals_old_fails_closed(self):
|
||||
res = self._apply(expected_old_head=OLD, new_head=OLD)
|
||||
self.assertFalse(res["refreshed"])
|
||||
|
||||
def test_non_full_sha_fails_closed(self):
|
||||
self.assertFalse(self._apply(new_head="deadbeef")["refreshed"])
|
||||
self.assertFalse(self._apply(expected_old_head="xyz")["refreshed"])
|
||||
|
||||
def test_no_lock_fails_closed(self):
|
||||
assessment = issue_lock_store.assess_durable_lock_head_refresh(
|
||||
None, remote=REMOTE, org=ORG, repo=REPO, issue_number=ISSUE,
|
||||
branch_name=BRANCH, worktree_path=self.wt, pr_number=PR_NUMBER,
|
||||
identity=IDENTITY, profile=PROFILE, current_pid=os.getpid(),
|
||||
expected_old_head=OLD, new_head=NEW1,
|
||||
)
|
||||
self.assertFalse(assessment["allowed"])
|
||||
|
||||
|
||||
# ─────────────────────── merge-sync provenance (real git) ───────────────────
|
||||
|
||||
|
||||
class TestMergeSyncProvenanceObservation(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.tmp = tempfile.mkdtemp()
|
||||
self.shas = build_merge_sync_repo(self.tmp)
|
||||
|
||||
def test_merge_sync_is_recognized(self):
|
||||
obs = issue_lock_worktree.read_merge_sync_provenance(
|
||||
self.tmp, prior_head_sha=self.shas["prior"],
|
||||
synced_head_sha=self.shas["synced"],
|
||||
)
|
||||
self.assertTrue(obs["is_merge_sync"], obs["reasons"])
|
||||
self.assertTrue(obs["prior_is_ancestor"])
|
||||
self.assertTrue(obs["synced_is_merge"])
|
||||
self.assertTrue(obs["first_parent_reaches_prior"])
|
||||
|
||||
def test_linear_descendant_is_not_a_merge_sync(self):
|
||||
"""A plain non-merge descendant (rebase/extra commit) is not a sync."""
|
||||
obs = issue_lock_worktree.read_merge_sync_provenance(
|
||||
self.tmp, prior_head_sha=self.shas["prior"],
|
||||
synced_head_sha=self.shas["linear"],
|
||||
)
|
||||
self.assertTrue(obs["probe_ok"])
|
||||
self.assertFalse(obs["is_merge_sync"])
|
||||
self.assertFalse(obs["synced_is_merge"])
|
||||
|
||||
def test_unrelated_history_fails_closed(self):
|
||||
"""A rewritten/force-pushed head where prior is unreachable fails closed."""
|
||||
obs = issue_lock_worktree.read_merge_sync_provenance(
|
||||
self.tmp, prior_head_sha=self.shas["prior"],
|
||||
synced_head_sha=self.shas["unrelated"],
|
||||
)
|
||||
self.assertFalse(obs["is_merge_sync"])
|
||||
|
||||
def test_missing_args_fail_closed(self):
|
||||
obs = issue_lock_worktree.read_merge_sync_provenance(
|
||||
self.tmp, prior_head_sha=None, synced_head_sha=self.shas["synced"],
|
||||
)
|
||||
self.assertFalse(obs["is_merge_sync"])
|
||||
|
||||
|
||||
# ──────────────────── merge-sync dead-session recovery ──────────────────────
|
||||
|
||||
|
||||
def make_dead_lock(worktree, **over):
|
||||
pid = dead_pid()
|
||||
lock = {
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": BRANCH,
|
||||
"worktree_path": worktree,
|
||||
"remote": REMOTE,
|
||||
"org": ORG,
|
||||
"repo": REPO,
|
||||
"session_pid": pid,
|
||||
"pid": pid,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
"work_lease": {
|
||||
"operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE,
|
||||
"issue_number": ISSUE,
|
||||
"branch": BRANCH,
|
||||
"worktree_path": worktree,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
"expires_at": future_ts(),
|
||||
},
|
||||
}
|
||||
lock.update(over)
|
||||
return lock
|
||||
|
||||
|
||||
def sync_prov(prior, synced, **over):
|
||||
d = {
|
||||
"prior_head_sha": prior,
|
||||
"synced_head_sha": synced,
|
||||
"probe_ok": True,
|
||||
"prior_present": True,
|
||||
"synced_present": True,
|
||||
"prior_is_ancestor": True,
|
||||
"synced_is_merge": True,
|
||||
"first_parent_reaches_prior": True,
|
||||
"is_merge_sync": True,
|
||||
"first_parent_sha": prior,
|
||||
"parent_count": 2,
|
||||
"proof": f"{synced} merged base into branch above {prior}",
|
||||
"reasons": [],
|
||||
}
|
||||
d.update(over)
|
||||
return d
|
||||
|
||||
|
||||
class TestMergeSyncRecovery(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.tmp = tempfile.mkdtemp()
|
||||
self.shas = build_merge_sync_repo(self.tmp)
|
||||
self.prior = self.shas["prior"]
|
||||
self.synced = self.shas["synced"]
|
||||
|
||||
def _assess(self, **over):
|
||||
lock = over.pop("_lock", None) or make_dead_lock(self.tmp)
|
||||
kw = dict(
|
||||
issue_number=ISSUE, branch_name=BRANCH, worktree_path=self.tmp,
|
||||
remote=REMOTE, org=ORG, repo=REPO, identity=IDENTITY, profile=PROFILE,
|
||||
current_branch=BRANCH, porcelain_status="",
|
||||
head_sha=self.prior, remote_head_sha=self.synced,
|
||||
pr_head_sha=self.synced, pr_number=PR_NUMBER,
|
||||
competing_live_locks=[], candidate_branches=[BRANCH],
|
||||
current_pid=os.getpid(),
|
||||
remote_branch_exists=True,
|
||||
sync_provenance=sync_prov(self.prior, self.synced),
|
||||
)
|
||||
kw.update(over)
|
||||
return issue_lock_recovery.assess_dead_session_lock_recovery(lock, **kw)
|
||||
|
||||
def test_merge_sync_drift_is_recoverable(self):
|
||||
"""AC3/AC4: dead session, recorded head is a merge-sync ancestor of PR head."""
|
||||
res = self._assess()
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.RECOVERY_SANCTIONED, res["reasons"])
|
||||
self.assertEqual(
|
||||
res["evidence"]["head_relation"],
|
||||
issue_lock_recovery.HEAD_RELATION_REMOTE_MERGE_SYNCED,
|
||||
)
|
||||
self.assertEqual(res["evidence"]["accepted_head"], self.synced)
|
||||
|
||||
def test_missing_provenance_fails_closed(self):
|
||||
"""No server-derived provenance → cannot accept a remote ahead of local."""
|
||||
res = self._assess(sync_provenance=None)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_non_ancestor_recorded_head_fails_closed(self):
|
||||
"""AC7: provenance that does not prove ancestry is rejected."""
|
||||
res = self._assess(
|
||||
sync_provenance=sync_prov(
|
||||
self.prior, self.synced, prior_is_ancestor=False, is_merge_sync=False,
|
||||
reasons=["prior head is not an ancestor"],
|
||||
)
|
||||
)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_force_pushed_history_fails_closed(self):
|
||||
"""AC8: a rewritten head (not a merge sync) stays protected."""
|
||||
res = self._assess(
|
||||
sync_provenance=sync_prov(
|
||||
self.prior, self.synced, is_merge_sync=False, synced_is_merge=False,
|
||||
reasons=["not a merge-based sync"],
|
||||
)
|
||||
)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_provenance_for_other_commits_fails_closed(self):
|
||||
"""Provenance whose endpoints differ from the heads under assessment is rejected."""
|
||||
res = self._assess(
|
||||
sync_provenance=sync_prov("f" * 40, self.synced),
|
||||
)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_dirty_worktree_fails_closed(self):
|
||||
"""AC11: dirty worktrees remain protected."""
|
||||
res = self._assess(porcelain_status=" M feature.txt\n")
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_live_owner_fails_closed(self):
|
||||
"""AC10: a live recorded owner is not a dead-session recovery."""
|
||||
lock = make_dead_lock(self.tmp, session_pid=os.getpid(), pid=os.getpid())
|
||||
res = self._assess(_lock=lock)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_competing_claimant_fails_closed(self):
|
||||
"""AC13: a competing live lock blocks recovery."""
|
||||
res = self._assess(
|
||||
competing_live_locks=[{
|
||||
"issue_number": ISSUE, "branch_name": BRANCH,
|
||||
"worktree_path": "/some/other/wt", "pid": os.getpid(),
|
||||
}]
|
||||
)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_wrong_branch_fails_closed(self):
|
||||
"""AC9: worktree on a different branch fails closed."""
|
||||
res = self._assess(current_branch="fix/issue-8710-other")
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_wrong_identity_fails_closed(self):
|
||||
res = self._assess(identity="intruder")
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_pr_head_mismatch_fails_closed(self):
|
||||
"""The open PR must sit at the synced remote head."""
|
||||
res = self._assess(pr_head_sha="e" * 40)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_owning_pr_evidence_for_merge_sync(self):
|
||||
res = self._assess()
|
||||
ev = issue_lock_recovery.owning_pr_recovery_evidence(res)
|
||||
self.assertIsNotNone(ev)
|
||||
self.assertEqual(ev["pr_number"], PR_NUMBER)
|
||||
self.assertEqual(ev["head_sha"], self.synced)
|
||||
self.assertEqual(
|
||||
ev["head_relation"],
|
||||
issue_lock_recovery.HEAD_RELATION_REMOTE_MERGE_SYNCED,
|
||||
)
|
||||
|
||||
def test_recovered_owning_pr_from_persisted_record(self):
|
||||
res = self._assess()
|
||||
record = issue_lock_recovery.build_recovery_record(res, recovered_at=future_ts(0))
|
||||
lock = {"issue_number": ISSUE, "branch_name": BRANCH,
|
||||
"dead_session_recovery": record}
|
||||
rebuilt = issue_lock_recovery.recovered_owning_pr_from_lock(lock)
|
||||
self.assertIsNotNone(rebuilt)
|
||||
self.assertEqual(rebuilt["head_sha"], self.synced)
|
||||
self.assertEqual(
|
||||
rebuilt["head_relation"],
|
||||
issue_lock_recovery.HEAD_RELATION_REMOTE_MERGE_SYNCED,
|
||||
)
|
||||
|
||||
|
||||
class TestExistingRelationsUnchanged(unittest.TestCase):
|
||||
"""AC14/AC15: equal-head recovery still works; merge-sync did not weaken it."""
|
||||
|
||||
def setUp(self):
|
||||
self.tmp = tempfile.mkdtemp()
|
||||
self.shas = build_merge_sync_repo(self.tmp)
|
||||
|
||||
def test_equal_head_recovery_still_sanctioned(self):
|
||||
# Worktree at prior head; remote also at prior head → the #753 equal case.
|
||||
prior = self.shas["prior"]
|
||||
lock = make_dead_lock(self.tmp)
|
||||
res = issue_lock_recovery.assess_dead_session_lock_recovery(
|
||||
lock, issue_number=ISSUE, branch_name=BRANCH, worktree_path=self.tmp,
|
||||
remote=REMOTE, org=ORG, repo=REPO, identity=IDENTITY, profile=PROFILE,
|
||||
current_branch=BRANCH, porcelain_status="",
|
||||
head_sha=prior, remote_head_sha=prior,
|
||||
pr_head_sha=prior, pr_number=PR_NUMBER,
|
||||
competing_live_locks=[], candidate_branches=[BRANCH],
|
||||
current_pid=os.getpid(), remote_branch_exists=True,
|
||||
)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.RECOVERY_SANCTIONED, res["reasons"])
|
||||
self.assertEqual(
|
||||
res["evidence"]["head_relation"], issue_lock_recovery.HEAD_RELATION_EQUAL,
|
||||
)
|
||||
|
||||
|
||||
class TestUpdatePrWrapperPartialFailure(unittest.TestCase):
|
||||
"""AC5/AC16: the tool advances the remote head then refreshes the durable lock.
|
||||
|
||||
When the durable refresh fails after the remote advance, the tool must report a
|
||||
partial lifecycle failure and NOT a fully successful synchronization. Exact PR-
|
||||
head / base-head pinning is preserved (delegated to the real preflight, stubbed
|
||||
here only to isolate the post-update lifecycle branch).
|
||||
"""
|
||||
|
||||
def setUp(self):
|
||||
import gitea_mcp_server as gms # noqa: E402
|
||||
self.gms = gms
|
||||
self._orig = {}
|
||||
|
||||
def _patch(name, value):
|
||||
self._orig[name] = getattr(gms, name)
|
||||
setattr(gms, name, value)
|
||||
|
||||
_patch("get_profile", lambda *a, **k: {
|
||||
"allowed_operations": ["gitea.branch.push"],
|
||||
"forbidden_operations": [],
|
||||
"profile_name": "prgs-author",
|
||||
})
|
||||
_patch("_role_kind", lambda *a, **k: "author")
|
||||
_patch("_profile_operation_gate", lambda *a, **k: None)
|
||||
_patch("_permission_block_report", lambda *a, **k: {})
|
||||
_patch("_resolve", lambda *a, **k: ("gitea.prgs.cc", ORG, REPO))
|
||||
_patch("_verify_role_mutation_workspace", lambda *a, **k: None)
|
||||
_patch("_get_workspace_porcelain", lambda *a, **k: "")
|
||||
_patch("_canonical_local_git_root", lambda *a, **k: "/x")
|
||||
_patch("_master_parity_block", lambda *a, **k: None)
|
||||
_patch("_auth", lambda *a, **k: {"token": "x"})
|
||||
_patch("repo_api_url", lambda *a, **k: "http://api")
|
||||
_patch("_redact", lambda s: s)
|
||||
_patch("_work_lease_claimant", lambda *a, **k: {
|
||||
"username": IDENTITY, "profile": PROFILE,
|
||||
})
|
||||
_patch("_prove_author_ownership_for_pr", lambda *a, **k: {
|
||||
"has_author_lock": True, "matched_issue": ISSUE,
|
||||
"matched_via": "branch", "linked_issues": [ISSUE],
|
||||
"recovered_owning_pr": None, "reasons": [],
|
||||
})
|
||||
|
||||
# Real preflight is unit-tested elsewhere; stub it to isolate the
|
||||
# post-update durable-lock lifecycle branch under test.
|
||||
orig_pf = gms.pr_sync_status.assess_update_pr_branch_preflight
|
||||
self._orig_pf = orig_pf
|
||||
gms.pr_sync_status.assess_update_pr_branch_preflight = (
|
||||
lambda *a, **k: {"mutation_allowed": True, "reasons": [], "performed": False}
|
||||
)
|
||||
|
||||
# Sequence the two GET /pulls calls: OLD before update, NEW after.
|
||||
self._pull_calls = {"n": 0}
|
||||
|
||||
def fake_api_request(method, url, auth, *a, **k):
|
||||
m = method.upper()
|
||||
if m == "GET" and url.endswith(f"/pulls/{PR_NUMBER}"):
|
||||
self._pull_calls["n"] += 1
|
||||
head = OLD if self._pull_calls["n"] == 1 else NEW1
|
||||
return {
|
||||
"state": "open",
|
||||
"head": {"sha": head, "ref": BRANCH},
|
||||
"base": {"sha": BASE, "ref": "master"},
|
||||
"mergeable": True, "title": "t", "body": "b",
|
||||
}
|
||||
if m == "GET" and "/branches/" in url:
|
||||
return {"commit": {"id": BASE}}
|
||||
if m == "POST" and "/update" in url:
|
||||
return {}
|
||||
return {}
|
||||
|
||||
_patch("api_request", fake_api_request)
|
||||
|
||||
def tearDown(self):
|
||||
for name, value in self._orig.items():
|
||||
setattr(self.gms, name, value)
|
||||
self.gms.pr_sync_status.assess_update_pr_branch_preflight = self._orig_pf
|
||||
|
||||
def _run(self):
|
||||
return self.gms.gitea_update_pr_branch_by_merge(
|
||||
pr_number=PR_NUMBER,
|
||||
expected_pr_head_sha=OLD,
|
||||
expected_base_head_sha=BASE,
|
||||
remote=REMOTE,
|
||||
worktree_path="/tmp/branches/wt-871",
|
||||
)
|
||||
|
||||
def test_partial_failure_when_refresh_fails(self):
|
||||
self._orig["apply_durable_lock_head_refresh"] = (
|
||||
self.gms.issue_lock_store.apply_durable_lock_head_refresh
|
||||
)
|
||||
self.gms.issue_lock_store.apply_durable_lock_head_refresh = (
|
||||
lambda **k: {"refreshed": False, "reasons": ["forced refresh failure"]}
|
||||
)
|
||||
try:
|
||||
res = self._run()
|
||||
finally:
|
||||
self.gms.issue_lock_store.apply_durable_lock_head_refresh = (
|
||||
self._orig["apply_durable_lock_head_refresh"]
|
||||
)
|
||||
self.assertTrue(res["performed"])
|
||||
self.assertEqual(res["new_pr_head_sha"], NEW1)
|
||||
self.assertFalse(res["success"])
|
||||
self.assertTrue(res["partial_lifecycle_failure"])
|
||||
self.assertFalse(res["durable_lock_refreshed"])
|
||||
|
||||
def test_full_success_when_refresh_succeeds(self):
|
||||
self._orig["apply_durable_lock_head_refresh"] = (
|
||||
self.gms.issue_lock_store.apply_durable_lock_head_refresh
|
||||
)
|
||||
self.gms.issue_lock_store.apply_durable_lock_head_refresh = (
|
||||
lambda **k: {"refreshed": True, "read_after_write_ok": True,
|
||||
"new_head": NEW1, "reasons": ["ok"]}
|
||||
)
|
||||
try:
|
||||
res = self._run()
|
||||
finally:
|
||||
self.gms.issue_lock_store.apply_durable_lock_head_refresh = (
|
||||
self._orig["apply_durable_lock_head_refresh"]
|
||||
)
|
||||
self.assertTrue(res["success"])
|
||||
self.assertTrue(res["performed"])
|
||||
self.assertTrue(res["durable_lock_refreshed"])
|
||||
self.assertTrue(res["fully_synchronized"])
|
||||
self.assertEqual(res["new_pr_head_sha"], NEW1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,332 @@
|
||||
"""Dead-session recovery for cross-session PR base-syncs (#872).
|
||||
|
||||
Validates that when a sanctioned server-side base-sync (``gitea_update_pr_branch_by_merge``)
|
||||
advances a PR's remote head from A to B (a merge commit whose first parent is A),
|
||||
a subsequent author session whose local worktree is at A can recover the dead-session
|
||||
lock via HEAD_RELATION_REMOTE_MERGE_SYNCED and execute a second base-sync (B to C)
|
||||
without deadlock or RuntimeError.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import issue_lock_recovery # noqa: E402
|
||||
import issue_lock_store # noqa: E402
|
||||
import issue_lock_worktree # noqa: E402
|
||||
|
||||
ISSUE = 8720
|
||||
PR_NUMBER = 8721
|
||||
BRANCH = f"fix/issue-{ISSUE}-dead-session-two-base-syncs"
|
||||
IDENTITY = "example-author"
|
||||
PROFILE = "prgs-author"
|
||||
REMOTE = "prgs"
|
||||
ORG = "ExampleOrg"
|
||||
REPO = "ExampleRepo"
|
||||
|
||||
|
||||
def dead_pid() -> int:
|
||||
proc = subprocess.Popen([sys.executable, "-c", "pass"])
|
||||
proc.wait()
|
||||
return proc.pid
|
||||
|
||||
|
||||
def future_ts(hours: int = 4) -> str:
|
||||
return (
|
||||
(datetime.now(timezone.utc) + timedelta(hours=hours))
|
||||
.isoformat()
|
||||
.replace("+00:00", "Z")
|
||||
)
|
||||
|
||||
|
||||
def _git(cwd, *args):
|
||||
return subprocess.run(
|
||||
["git", "-C", cwd, *args],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
)
|
||||
|
||||
|
||||
def _rev(cwd, ref="HEAD") -> str:
|
||||
return _git(cwd, "rev-parse", ref).stdout.strip()
|
||||
|
||||
|
||||
def build_two_base_sync_repo(tmp: str) -> dict:
|
||||
"""Build a repo simulating two sequential base-sync merges."""
|
||||
_git(tmp, "init", "-q", "-b", "master")
|
||||
_git(tmp, "config", "user.email", "[email protected]")
|
||||
_git(tmp, "config", "user.name", "Author")
|
||||
Path(tmp, "base.txt").write_text("base 1\n")
|
||||
_git(tmp, "add", "-A")
|
||||
_git(tmp, "commit", "-q", "-m", "initial master")
|
||||
|
||||
# Feature branch cut from initial master -> Head A (prior/recorded head)
|
||||
_git(tmp, "checkout", "-q", "-b", BRANCH)
|
||||
Path(tmp, "feature.txt").write_text("feature work\n")
|
||||
_git(tmp, "add", "-A")
|
||||
_git(tmp, "commit", "-q", "-m", "feature commit A")
|
||||
head_a = _rev(tmp)
|
||||
|
||||
# Master advances -> Master 1
|
||||
_git(tmp, "checkout", "-q", "master")
|
||||
Path(tmp, "base.txt").write_text("base 1\nbase 2\n")
|
||||
_git(tmp, "add", "-A")
|
||||
_git(tmp, "commit", "-q", "-m", "master advance 1")
|
||||
master_1 = _rev(tmp)
|
||||
|
||||
# First sync: merge master into feature -> Head B (merge commit, first parent = A)
|
||||
_git(tmp, "checkout", "-q", BRANCH)
|
||||
_git(tmp, "merge", "-q", "--no-ff", "-m", "First base-sync (merge master)", "master")
|
||||
head_b = _rev(tmp)
|
||||
|
||||
# Master advances again -> Master 2
|
||||
_git(tmp, "checkout", "-q", "master")
|
||||
Path(tmp, "base.txt").write_text("base 1\nbase 2\nbase 3\n")
|
||||
_git(tmp, "add", "-A")
|
||||
_git(tmp, "commit", "-q", "-m", "master advance 2")
|
||||
master_2 = _rev(tmp)
|
||||
|
||||
# Second sync: merge master into feature -> Head C (merge commit, first parent = B)
|
||||
_git(tmp, "checkout", "-q", BRANCH)
|
||||
_git(tmp, "merge", "-q", "--no-ff", "-m", "Second base-sync (merge master)", "master")
|
||||
head_c = _rev(tmp)
|
||||
|
||||
# Non-merge rebase/force-pushed branch shape
|
||||
_git(tmp, "checkout", "-q", "-b", "rebased-branch", head_a)
|
||||
Path(tmp, "rebase.txt").write_text("rebased\n")
|
||||
_git(tmp, "add", "-A")
|
||||
_git(tmp, "commit", "-q", "-m", "rebased commit")
|
||||
rebased_head = _rev(tmp)
|
||||
|
||||
# Reset worktree back to head A, as a dead session leaving local worktree at A
|
||||
_git(tmp, "checkout", "-q", BRANCH)
|
||||
_git(tmp, "reset", "-q", "--hard", head_a)
|
||||
|
||||
return {
|
||||
"head_a": head_a,
|
||||
"master_1": master_1,
|
||||
"head_b": head_b,
|
||||
"master_2": master_2,
|
||||
"head_c": head_c,
|
||||
"rebased_head": rebased_head,
|
||||
}
|
||||
|
||||
|
||||
def make_dead_lock(worktree, **over):
|
||||
pid = dead_pid()
|
||||
lock = {
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": BRANCH,
|
||||
"worktree_path": worktree,
|
||||
"remote": REMOTE,
|
||||
"org": ORG,
|
||||
"repo": REPO,
|
||||
"session_pid": pid,
|
||||
"pid": pid,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
"work_lease": {
|
||||
"operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE,
|
||||
"issue_number": ISSUE,
|
||||
"branch": BRANCH,
|
||||
"worktree_path": worktree,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
"expires_at": future_ts(),
|
||||
},
|
||||
}
|
||||
lock.update(over)
|
||||
return lock
|
||||
|
||||
|
||||
class TestDeadSessionTwoBaseSyncs(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.tmp = tempfile.mkdtemp()
|
||||
self.shas = build_two_base_sync_repo(self.tmp)
|
||||
self.head_a = self.shas["head_a"]
|
||||
self.head_b = self.shas["head_b"]
|
||||
self.head_c = self.shas["head_c"]
|
||||
|
||||
def test_first_base_sync_dead_session_recovery(self):
|
||||
"""AC1: dead session after first base sync (remote=B, local=A) recovers via merge-sync."""
|
||||
lock = make_dead_lock(self.tmp, synced_pr_head=self.head_b)
|
||||
obs = issue_lock_worktree.read_merge_sync_provenance(
|
||||
self.tmp,
|
||||
prior_head_sha=self.head_a,
|
||||
synced_head_sha=self.head_b,
|
||||
remote=REMOTE,
|
||||
)
|
||||
self.assertTrue(obs["is_merge_sync"], obs["reasons"])
|
||||
self.assertTrue(obs["prior_is_ancestor"])
|
||||
self.assertTrue(obs["synced_is_merge"])
|
||||
self.assertTrue(obs["first_parent_reaches_prior"])
|
||||
|
||||
res = issue_lock_recovery.assess_dead_session_lock_recovery(
|
||||
lock,
|
||||
issue_number=ISSUE,
|
||||
branch_name=BRANCH,
|
||||
worktree_path=self.tmp,
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
identity=IDENTITY,
|
||||
profile=PROFILE,
|
||||
current_branch=BRANCH,
|
||||
porcelain_status="",
|
||||
head_sha=self.head_a,
|
||||
remote_head_sha=self.head_b,
|
||||
pr_head_sha=self.head_b,
|
||||
pr_number=PR_NUMBER,
|
||||
competing_live_locks=[],
|
||||
candidate_branches=[BRANCH],
|
||||
current_pid=os.getpid(),
|
||||
remote_branch_exists=True,
|
||||
sync_provenance=obs,
|
||||
)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.RECOVERY_SANCTIONED, res["reasons"])
|
||||
self.assertEqual(
|
||||
res["evidence"]["head_relation"],
|
||||
issue_lock_recovery.HEAD_RELATION_REMOTE_MERGE_SYNCED,
|
||||
)
|
||||
self.assertEqual(res["evidence"]["accepted_head"], self.head_b)
|
||||
|
||||
def test_second_base_sync_head_refresh_after_recovery(self):
|
||||
"""AC2: after recovery, second base sync (remote B -> C) updates durable lock head."""
|
||||
lock_dir = tempfile.mkdtemp()
|
||||
lock_data = make_dead_lock(self.tmp, synced_pr_head=self.head_b)
|
||||
# Rebind to current PID as gitea_lock_issue does upon sanctioned recovery
|
||||
lock_data["session_pid"] = os.getpid()
|
||||
lock_data["pid"] = os.getpid()
|
||||
issue_lock_store.save_lock_file(
|
||||
issue_lock_store.lock_file_path(
|
||||
remote=REMOTE, org=ORG, repo=REPO, issue_number=ISSUE, lock_dir=lock_dir
|
||||
),
|
||||
lock_data,
|
||||
)
|
||||
|
||||
refresh_res = issue_lock_store.apply_durable_lock_head_refresh(
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
issue_number=ISSUE,
|
||||
branch_name=BRANCH,
|
||||
worktree_path=self.tmp,
|
||||
pr_number=PR_NUMBER,
|
||||
identity=IDENTITY,
|
||||
profile=PROFILE,
|
||||
current_pid=os.getpid(),
|
||||
expected_old_head=self.head_b,
|
||||
new_head=self.head_c,
|
||||
synced_at=future_ts(0),
|
||||
base_head=self.shas["master_2"],
|
||||
lock_dir=lock_dir,
|
||||
)
|
||||
self.assertTrue(refresh_res["refreshed"], refresh_res["reasons"])
|
||||
self.assertTrue(refresh_res["read_after_write_ok"])
|
||||
loaded = issue_lock_store.load_issue_lock(
|
||||
remote=REMOTE, org=ORG, repo=REPO, issue_number=ISSUE, lock_dir=lock_dir
|
||||
)
|
||||
self.assertEqual(loaded.get("synced_pr_head"), self.head_c)
|
||||
|
||||
def test_dirty_worktree_blocks_recovery(self):
|
||||
"""AC3: tracked dirty edits block dead-session recovery."""
|
||||
lock = make_dead_lock(self.tmp)
|
||||
obs = issue_lock_worktree.read_merge_sync_provenance(
|
||||
self.tmp, prior_head_sha=self.head_a, synced_head_sha=self.head_b
|
||||
)
|
||||
res = issue_lock_recovery.assess_dead_session_lock_recovery(
|
||||
lock,
|
||||
issue_number=ISSUE,
|
||||
branch_name=BRANCH,
|
||||
worktree_path=self.tmp,
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
identity=IDENTITY,
|
||||
profile=PROFILE,
|
||||
current_branch=BRANCH,
|
||||
porcelain_status=" M feature.txt\n",
|
||||
head_sha=self.head_a,
|
||||
remote_head_sha=self.head_b,
|
||||
pr_head_sha=self.head_b,
|
||||
pr_number=PR_NUMBER,
|
||||
competing_live_locks=[],
|
||||
candidate_branches=[BRANCH],
|
||||
current_pid=os.getpid(),
|
||||
remote_branch_exists=True,
|
||||
sync_provenance=obs,
|
||||
)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
self.assertTrue(any("dirty" in r or "tracked" in r for r in res["reasons"]))
|
||||
|
||||
def test_foreign_reclaimer_blocks_recovery(self):
|
||||
"""AC4: a foreign claimant cannot recover a dead-session lock."""
|
||||
lock = make_dead_lock(self.tmp)
|
||||
obs = issue_lock_worktree.read_merge_sync_provenance(
|
||||
self.tmp, prior_head_sha=self.head_a, synced_head_sha=self.head_b
|
||||
)
|
||||
res = issue_lock_recovery.assess_dead_session_lock_recovery(
|
||||
lock,
|
||||
issue_number=ISSUE,
|
||||
branch_name=BRANCH,
|
||||
worktree_path=self.tmp,
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
identity="intruder-user",
|
||||
profile=PROFILE,
|
||||
current_branch=BRANCH,
|
||||
porcelain_status="",
|
||||
head_sha=self.head_a,
|
||||
remote_head_sha=self.head_b,
|
||||
pr_head_sha=self.head_b,
|
||||
pr_number=PR_NUMBER,
|
||||
competing_live_locks=[],
|
||||
candidate_branches=[BRANCH],
|
||||
current_pid=os.getpid(),
|
||||
remote_branch_exists=True,
|
||||
sync_provenance=obs,
|
||||
)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
def test_non_merge_rebased_remote_head_blocks_recovery(self):
|
||||
"""AC5: a rebased/force-pushed remote head (not a merge commit) is refused."""
|
||||
lock = make_dead_lock(self.tmp)
|
||||
obs = issue_lock_worktree.read_merge_sync_provenance(
|
||||
self.tmp, prior_head_sha=self.head_a, synced_head_sha=self.shas["rebased_head"]
|
||||
)
|
||||
self.assertFalse(obs["is_merge_sync"])
|
||||
res = issue_lock_recovery.assess_dead_session_lock_recovery(
|
||||
lock,
|
||||
issue_number=ISSUE,
|
||||
branch_name=BRANCH,
|
||||
worktree_path=self.tmp,
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
identity=IDENTITY,
|
||||
profile=PROFILE,
|
||||
current_branch=BRANCH,
|
||||
porcelain_status="",
|
||||
head_sha=self.head_a,
|
||||
remote_head_sha=self.shas["rebased_head"],
|
||||
pr_head_sha=self.shas["rebased_head"],
|
||||
pr_number=PR_NUMBER,
|
||||
competing_live_locks=[],
|
||||
candidate_branches=[BRANCH],
|
||||
current_pid=os.getpid(),
|
||||
remote_branch_exists=True,
|
||||
sync_provenance=obs,
|
||||
)
|
||||
self.assertEqual(res["outcome"], issue_lock_recovery.REFUSED)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,395 @@
|
||||
"""``apply_authorized`` requires BOTH authorizations (#886 review blocker B1).
|
||||
|
||||
The #663 restart-class matrix and the #661 drain-proof hard gate are independent
|
||||
authorizations that first coexisted when PR #882 landed on master and PR #886
|
||||
merged it into the restart-class branch. The union preserved both, but the apply
|
||||
decision consulted only the drain gate::
|
||||
|
||||
payload["apply_authorized"] = gate.allow # pre-fix
|
||||
|
||||
so a clean drain proof — or an authorized break-glass, which needs no proof at
|
||||
all — reported ``apply_authorized: True`` for a restart class the least-privilege
|
||||
matrix had just denied, in the same payload that carried
|
||||
``allow_restart: False`` and "role 'author' may not request full_mcp_restart".
|
||||
|
||||
These tests pin the conjunction and the properties that must survive it. They
|
||||
exercise the real MCP tool, which previously had no test coverage at all — that
|
||||
absence is why the defect shipped.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
import drain_proof
|
||||
import gitea_mcp_server as srv
|
||||
|
||||
CONTROLLER_APPROVAL_ENV = "GITEA_CONTROLLER_RESTART_APPROVAL_AUTHORIZATION"
|
||||
BREAK_GLASS_ENV = "GITEA_BREAKGLASS_RESTART_AUTHORIZATION"
|
||||
|
||||
# A quiet control plane: nothing live, so the blast radius never masks the
|
||||
# authorization outcome under test.
|
||||
QUIET_SESSIONS: list[dict] = []
|
||||
QUIET_LEASES: list[dict] = []
|
||||
|
||||
# #669: broad restarts need a prior narrow-attempt log (unless break-glass).
|
||||
PRIOR_NARROW_ATTEMPTS_JSON = json.dumps(
|
||||
[
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "still flapping after reconnect",
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
class _FakeDB:
|
||||
"""Minimal control-plane DB stand-in for the restart inventory."""
|
||||
|
||||
def __init__(self, sessions=QUIET_SESSIONS, terminal=None):
|
||||
self._sessions = list(sessions)
|
||||
self._terminal = terminal
|
||||
|
||||
def list_sessions(self, statuses=None, limit=None):
|
||||
return list(self._sessions)
|
||||
|
||||
def get_active_terminal_lock(self, remote=None, org=None, repo=None):
|
||||
return self._terminal
|
||||
|
||||
|
||||
def _profile(role: str) -> dict:
|
||||
return {"profile_name": f"prgs-{role}", "role_kind": role, "role": role}
|
||||
|
||||
|
||||
class _RestartToolHarness(unittest.TestCase):
|
||||
"""Drives the real ``gitea_request_mcp_restart`` with a stubbed inventory."""
|
||||
|
||||
def _call(self, *, role: str, env: dict | None = None, **kwargs) -> dict:
|
||||
environ = {k: v for k, v in os.environ.items()
|
||||
if k not in (CONTROLLER_APPROVAL_ENV, BREAK_GLASS_ENV)}
|
||||
environ.update(env or {})
|
||||
with patch.object(srv, "_profile_operation_gate", return_value=None), \
|
||||
patch.object(srv, "_resolve",
|
||||
return_value=("gitea.prgs.cc",
|
||||
"Scaled-Tech-Consulting",
|
||||
"Gitea-Tools")), \
|
||||
patch.object(srv, "get_profile", return_value=_profile(role)), \
|
||||
patch.object(srv, "_control_plane_db_or_error",
|
||||
return_value=(_FakeDB(), [])), \
|
||||
patch.object(srv.lease_lifecycle, "list_active_leases",
|
||||
return_value={"leases": list(QUIET_LEASES)}), \
|
||||
patch.dict(os.environ, environ, clear=True):
|
||||
return srv.gitea_request_mcp_restart(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
session_id="probe-session",
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
def _clean_proof_for(self, preview: dict) -> str:
|
||||
"""Mint a genuinely clean, signature-valid proof bound to *preview*.
|
||||
|
||||
Built from the tool's own dry-run report, so the fingerprint matches and
|
||||
the proof is rejected for authorization reasons only — never because it
|
||||
was stale or forged.
|
||||
"""
|
||||
proof = drain_proof.build_drain_proof(
|
||||
impact_report=preview,
|
||||
drain_state={
|
||||
"assignments_stopped": True,
|
||||
"checkpoints_complete": True,
|
||||
"handoffs_verified": True,
|
||||
"leases_handled": True,
|
||||
"acks": {},
|
||||
},
|
||||
requesting_session_id="probe-session",
|
||||
)
|
||||
self.assertTrue(proof.clean, "harness must mint a clean proof")
|
||||
return json.dumps(proof.as_dict())
|
||||
|
||||
|
||||
class TestConjunction(_RestartToolHarness):
|
||||
"""AC1/AC2 — the two authorizations are ANDed, in both directions."""
|
||||
|
||||
def test_gate_allow_with_class_denied_yields_apply_authorized_false(self):
|
||||
# An author may not request full_mcp_restart (CONTROL_ROLES only).
|
||||
preview = self._call(role="author", restart_class="full_mcp_restart")
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
|
||||
result = self._call(
|
||||
role="author",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
)
|
||||
|
||||
self.assertTrue(result["apply_gate"]["drain_gate_allow"],
|
||||
"drain gate itself should have allowed this proof")
|
||||
self.assertFalse(result["apply_gate"]["restart_class_authorized"])
|
||||
self.assertFalse(result["apply_authorized"],
|
||||
"a clean proof must not authorize a denied class")
|
||||
self.assertFalse(result["allow_restart"])
|
||||
|
||||
def test_gate_allow_with_class_allowed_can_yield_apply_authorized_true(self):
|
||||
preview = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertTrue(preview["allow_restart"],
|
||||
"operator + controller approval must authorize the class")
|
||||
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
|
||||
self.assertTrue(result["apply_gate"]["drain_gate_allow"])
|
||||
self.assertTrue(result["apply_gate"]["restart_class_authorized"])
|
||||
self.assertTrue(result["apply_authorized"],
|
||||
"both authorizations pass; apply must be authorized")
|
||||
|
||||
def test_denial_is_attributable_to_the_authorization_that_caused_it(self):
|
||||
preview = self._call(role="author", restart_class="full_mcp_restart")
|
||||
result = self._call(
|
||||
role="author",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
)
|
||||
blob = " ".join(result["apply_gate"]["reasons"]).lower()
|
||||
self.assertIn("restart class authorization denied", blob)
|
||||
self.assertIn("full_mcp_restart", blob)
|
||||
|
||||
|
||||
class TestProofCannotOverrideAuthorization(_RestartToolHarness):
|
||||
"""AC3 — a clean proof never overrides a class or requester-role denial."""
|
||||
|
||||
def test_clean_proof_cannot_override_role_denial(self):
|
||||
for role in ("author", "reviewer", "merger", "reconciler"):
|
||||
with self.subTest(role=role):
|
||||
preview = self._call(role=role, restart_class="full_mcp_restart")
|
||||
result = self._call(
|
||||
role=role,
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
def test_clean_proof_cannot_override_missing_controller_approval(self):
|
||||
# Correct role, but the class demands controller approval and the
|
||||
# environment carries none.
|
||||
preview = self._call(role="operator", restart_class="full_mcp_restart")
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
def test_clean_proof_cannot_override_unknown_class(self):
|
||||
preview = self._call(role="operator", restart_class="not_a_real_class",
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"})
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="not_a_real_class",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"},
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
def test_clean_proof_cannot_override_missing_scope_target(self):
|
||||
# worker_restart without target_session_id fails closed on scoping.
|
||||
preview = self._call(role="operator", restart_class="worker_restart",
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"})
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="worker_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"},
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
|
||||
class TestBreakGlassDoesNotCollapseTheMatrix(_RestartToolHarness):
|
||||
"""AC4 — break-glass bypasses the drain proof only, never the class matrix."""
|
||||
|
||||
def test_break_glass_does_not_authorize_a_denied_class(self):
|
||||
result = self._call(
|
||||
role="author",
|
||||
restart_class="host_restart",
|
||||
dry_run=False,
|
||||
request_break_glass=True,
|
||||
env={BREAK_GLASS_ENV: "operator-issued"},
|
||||
)
|
||||
self.assertTrue(result["break_glass_authorized"])
|
||||
self.assertTrue(result["apply_gate"]["drain_gate_allow"],
|
||||
"break-glass does satisfy the drain gate")
|
||||
self.assertFalse(result["apply_gate"]["restart_class_authorized"])
|
||||
self.assertFalse(result["apply_authorized"],
|
||||
"break-glass must not collapse the class matrix")
|
||||
|
||||
def test_break_glass_across_every_worker_role_and_restricted_class(self):
|
||||
for role in ("author", "reviewer", "merger", "reconciler"):
|
||||
for klass in ("rolling_mcp_restart", "full_mcp_restart",
|
||||
"host_restart"):
|
||||
with self.subTest(role=role, restart_class=klass):
|
||||
result = self._call(
|
||||
role=role,
|
||||
restart_class=klass,
|
||||
dry_run=False,
|
||||
request_break_glass=True,
|
||||
env={BREAK_GLASS_ENV: "operator-issued"},
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
def test_break_glass_still_works_when_the_class_is_authorized(self):
|
||||
# Break-glass keeps its purpose: skipping the drain proof for a caller
|
||||
# the matrix does allow.
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
request_break_glass=True,
|
||||
env={BREAK_GLASS_ENV: "operator-issued",
|
||||
CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertTrue(result["apply_authorized"])
|
||||
self.assertEqual(result["apply_gate"]["verdict"], "break_glass")
|
||||
|
||||
def test_break_glass_is_not_self_assertable(self):
|
||||
# Requested but no environment authorization -> no bypass, and the
|
||||
# unproven apply is denied.
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
request_break_glass=True,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertTrue(result["break_glass_requested"])
|
||||
self.assertFalse(result["break_glass_authorized"])
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
self.assertIn("incident", result)
|
||||
|
||||
|
||||
class TestRestrictedClassesStayDenied(_RestartToolHarness):
|
||||
"""AC5 — restricted classes remain denied to unauthorized requesters."""
|
||||
|
||||
def test_restricted_classes_denied_for_worker_roles(self):
|
||||
for role in ("author", "reviewer", "merger", "reconciler"):
|
||||
for klass in ("rolling_mcp_restart", "full_mcp_restart",
|
||||
"host_restart"):
|
||||
with self.subTest(role=role, restart_class=klass):
|
||||
preview = self._call(
|
||||
role=role,
|
||||
restart_class=klass,
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"},
|
||||
)
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
self.assertFalse(preview["permission_authorized"])
|
||||
self.assertFalse(preview["role_authorized"])
|
||||
|
||||
def test_host_restart_needs_controller_and_infrastructure_operator(self):
|
||||
# controller approval alone is not enough for host_restart.
|
||||
preview = self._call(role="controller", restart_class="host_restart",
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"})
|
||||
self.assertFalse(preview["approval_satisfied"])
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
|
||||
|
||||
class TestExistingPathsStillWork(_RestartToolHarness):
|
||||
"""AC6 — valid scoped and unscoped restart paths are unaffected."""
|
||||
|
||||
def test_dry_run_never_reports_apply_authorization(self):
|
||||
result = self._call(role="operator", restart_class="full_mcp_restart",
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"})
|
||||
self.assertNotIn("apply_authorized", result)
|
||||
self.assertNotIn("apply_gate", result)
|
||||
self.assertFalse(result["apply_supported"])
|
||||
self.assertFalse(result["restart_performed"])
|
||||
|
||||
def test_self_service_unscoped_classes_authorize_for_every_role(self):
|
||||
for role in ("author", "reviewer", "merger", "reconciler",
|
||||
"controller", "operator", "admin"):
|
||||
for klass in ("client_reconnect", "session_reconnect"):
|
||||
with self.subTest(role=role, restart_class=klass):
|
||||
preview = self._call(role=role, restart_class=klass)
|
||||
self.assertTrue(preview["allow_restart"])
|
||||
|
||||
def test_scoped_class_with_target_authorizes_and_applies(self):
|
||||
env = {CONTROLLER_APPROVAL_ENV: "operator-approved"}
|
||||
preview = self._call(role="operator", restart_class="worker_restart",
|
||||
target_session_id="worker-1", env=env)
|
||||
self.assertTrue(preview["allow_restart"])
|
||||
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="worker_restart",
|
||||
target_session_id="worker-1",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env=env,
|
||||
)
|
||||
self.assertTrue(result["apply_authorized"])
|
||||
|
||||
def test_apply_still_denies_without_any_proof(self):
|
||||
# The #661 hard gate is untouched by the conjunction.
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
dry_run=False,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertFalse(result["apply_gate"]["drain_gate_allow"])
|
||||
self.assertTrue(result["apply_gate"]["restart_class_authorized"])
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
self.assertEqual(result["incident"]["kind"], "restart_drain_gate_denied")
|
||||
|
||||
def test_apply_denies_on_malformed_proof(self):
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json="{not valid json",
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
self.assertTrue(any("invalid drain_proof_json" in reason
|
||||
for reason in result["apply_gate"]["reasons"]))
|
||||
|
||||
def test_tool_never_restarts_on_any_path(self):
|
||||
for kwargs in (
|
||||
{"restart_class": "client_reconnect"},
|
||||
{"restart_class": "full_mcp_restart", "dry_run": False},
|
||||
{"restart_class": "host_restart", "dry_run": False,
|
||||
"request_break_glass": True},
|
||||
):
|
||||
with self.subTest(**kwargs):
|
||||
result = self._call(role="operator", env={
|
||||
CONTROLLER_APPROVAL_ENV: "yes", BREAK_GLASS_ENV: "yes"},
|
||||
**kwargs)
|
||||
self.assertFalse(result["restart_performed"])
|
||||
self.assertFalse(result["apply_supported"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,215 @@
|
||||
"""Regression: author worktree bootstrap from clean control checkout (#892).
|
||||
|
||||
#892 is the four-door deadlock where every documented recovery path is closed:
|
||||
bootstrap refuses control, lock demands an existing worktree, worktree-start
|
||||
demands a lock, and shell worktree add is outside the sanctioned MCP path.
|
||||
|
||||
Root cause: assess_author_issue_bootstrap returned allowed/proven for a clean
|
||||
control checkout, but bootstrap_permits_control_checkout only accepted
|
||||
create_issue assessments (task_scope=create_issue_only + empty reasons + full
|
||||
base-tip field set). Author assessments never satisfied the shared predicate,
|
||||
so the #274/#604 guards kept the ordinary control-checkout block.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest import mock
|
||||
|
||||
import author_issue_bootstrap as aib
|
||||
import create_issue_bootstrap as cib
|
||||
|
||||
|
||||
CONTROL = "/repo/Gitea-Tools"
|
||||
MASTER = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
OTHER = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||
|
||||
|
||||
def _assess(
|
||||
*,
|
||||
workspace=CONTROL,
|
||||
root=CONTROL,
|
||||
branch="master",
|
||||
head=MASTER,
|
||||
porcelain="",
|
||||
remote=MASTER,
|
||||
remote_error=None,
|
||||
task="bootstrap_author_issue_worktree",
|
||||
):
|
||||
return aib.assess_author_issue_bootstrap(
|
||||
workspace_path=workspace,
|
||||
canonical_repo_root=root,
|
||||
current_branch=branch,
|
||||
head_sha=head,
|
||||
porcelain_status=porcelain,
|
||||
remote_master_sha=remote,
|
||||
remote_master_sha_error=remote_error,
|
||||
task=task,
|
||||
)
|
||||
|
||||
|
||||
class TestAuthorBootstrapAssessmentShape(unittest.TestCase):
|
||||
def test_clean_control_emits_predicate_compatible_fields(self):
|
||||
assessment = _assess()
|
||||
self.assertTrue(assessment["allowed"])
|
||||
self.assertTrue(assessment["proven"])
|
||||
self.assertFalse(assessment["block"])
|
||||
self.assertFalse(assessment["not_applicable"])
|
||||
self.assertEqual(assessment["reasons"], [])
|
||||
self.assertEqual(assessment["task_scope"], "author_issue_bootstrap")
|
||||
self.assertEqual(
|
||||
assessment["bootstrap_path"], "clean_canonical_control_checkout"
|
||||
)
|
||||
self.assertEqual(assessment["dirty_files"], [])
|
||||
self.assertIs(assessment["under_branches"], False)
|
||||
self.assertTrue(assessment["base_tips_verified"])
|
||||
self.assertEqual(assessment["local_head_sha"], MASTER)
|
||||
self.assertEqual(assessment["remote_master_sha"], MASTER)
|
||||
self.assertEqual(assessment["workspace_path"], os.path.realpath(CONTROL))
|
||||
self.assertEqual(
|
||||
assessment["canonical_repo_root"], os.path.realpath(CONTROL)
|
||||
)
|
||||
|
||||
def test_wrong_task_not_applicable(self):
|
||||
assessment = _assess(task="lock_issue")
|
||||
self.assertTrue(assessment["not_applicable"])
|
||||
self.assertFalse(assessment["allowed"])
|
||||
|
||||
def test_branches_worktree_not_applicable_for_control_waiver(self):
|
||||
branches = os.path.join(CONTROL, "branches", "fix-issue-1")
|
||||
assessment = _assess(workspace=branches)
|
||||
self.assertTrue(assessment["not_applicable"])
|
||||
self.assertFalse(assessment["allowed"])
|
||||
self.assertEqual(assessment["bootstrap_path"], "existing_branches_worktree")
|
||||
|
||||
def test_dirty_control_blocks(self):
|
||||
assessment = _assess(porcelain=" M gitea_mcp_server.py\n")
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertFalse(assessment["allowed"])
|
||||
self.assertTrue(any("tracked local edits" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_head_remote_mismatch_blocks(self):
|
||||
assessment = _assess(head=MASTER, remote=OTHER)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertFalse(assessment["allowed"])
|
||||
|
||||
def test_missing_remote_tip_blocks(self):
|
||||
assessment = _assess(remote=None)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertFalse(assessment["allowed"])
|
||||
|
||||
|
||||
class TestAuthorBootstrapPredicate(unittest.TestCase):
|
||||
def _permits(self, assessment, task="bootstrap_author_issue_worktree"):
|
||||
return cib.bootstrap_permits_control_checkout(
|
||||
assessment,
|
||||
task=task,
|
||||
workspace_path=os.path.realpath(CONTROL),
|
||||
canonical_repo_root=os.path.realpath(CONTROL),
|
||||
)
|
||||
|
||||
def test_clean_author_bootstrap_permits(self):
|
||||
self.assertTrue(self._permits(_assess()))
|
||||
|
||||
def test_tool_alias_permits(self):
|
||||
assessment = _assess(task="gitea_bootstrap_author_issue_worktree")
|
||||
self.assertTrue(
|
||||
self._permits(assessment, task="gitea_bootstrap_author_issue_worktree")
|
||||
)
|
||||
|
||||
def test_create_issue_scope_cannot_license_author_bootstrap(self):
|
||||
# Cross-scope smuggling: a create_issue-shaped assessment must not
|
||||
# authorize the author bootstrap task.
|
||||
create_shaped = dict(_assess())
|
||||
create_shaped["task_scope"] = "create_issue_only"
|
||||
self.assertFalse(self._permits(create_shaped))
|
||||
|
||||
def test_author_scope_cannot_license_create_issue(self):
|
||||
assessment = _assess()
|
||||
self.assertFalse(
|
||||
cib.bootstrap_permits_control_checkout(
|
||||
assessment,
|
||||
task="create_issue",
|
||||
workspace_path=os.path.realpath(CONTROL),
|
||||
canonical_repo_root=os.path.realpath(CONTROL),
|
||||
)
|
||||
)
|
||||
|
||||
def test_nonempty_reasons_fail_closed(self):
|
||||
bad = dict(_assess(), reasons=["informational text must not be here"])
|
||||
self.assertFalse(self._permits(bad))
|
||||
|
||||
def test_dirty_fails_closed(self):
|
||||
self.assertFalse(self._permits(_assess(porcelain=" M x.py\n")))
|
||||
|
||||
def test_mismatch_fails_closed(self):
|
||||
self.assertFalse(self._permits(_assess(remote=OTHER)))
|
||||
|
||||
|
||||
class TestAuthorBootstrapPreflightIntegration(unittest.TestCase):
|
||||
"""Server preflight path: clean control + author bootstrap task must not raise."""
|
||||
|
||||
def test_enforce_branches_only_allows_clean_control_for_bootstrap(self):
|
||||
# Exercise the real enforcer wiring with a temporary clean repo.
|
||||
import gitea_mcp_server as srv
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
repo = os.path.join(tmp, "repo")
|
||||
os.makedirs(os.path.join(repo, "branches"))
|
||||
# Minimal git repo on master at a known tip.
|
||||
import subprocess
|
||||
|
||||
subprocess.check_call(["git", "init", "-b", "master", repo])
|
||||
subprocess.check_call(
|
||||
["git", "-C", repo, "commit", "--allow-empty", "-m", "init"]
|
||||
)
|
||||
head = subprocess.check_output(
|
||||
["git", "-C", repo, "rev-parse", "HEAD"], text=True
|
||||
).strip()
|
||||
|
||||
assessment = aib.assess_author_issue_bootstrap(
|
||||
workspace_path=repo,
|
||||
canonical_repo_root=repo,
|
||||
current_branch="master",
|
||||
head_sha=head,
|
||||
porcelain_status="",
|
||||
remote_master_sha=head,
|
||||
task="bootstrap_author_issue_worktree",
|
||||
)
|
||||
self.assertTrue(
|
||||
cib.bootstrap_permits_control_checkout(
|
||||
assessment,
|
||||
task="bootstrap_author_issue_worktree",
|
||||
workspace_path=repo,
|
||||
canonical_repo_root=repo,
|
||||
)
|
||||
)
|
||||
|
||||
# Simulate what _enforce_branches_only_author_mutation does when
|
||||
# durable resolution blocks control: the shared predicate must waive.
|
||||
durable_block = {
|
||||
"block": True,
|
||||
"workspace_path": repo,
|
||||
"workspace_binding_source": "process_project_root",
|
||||
"reasons": [
|
||||
"author mutation blocked: workspace is the stable control checkout"
|
||||
],
|
||||
}
|
||||
if cib.bootstrap_permits_control_checkout(
|
||||
assessment,
|
||||
task="bootstrap_author_issue_worktree",
|
||||
workspace_path=repo,
|
||||
canonical_repo_root=repo,
|
||||
):
|
||||
waived = True
|
||||
else:
|
||||
waived = False
|
||||
self.assertTrue(waived)
|
||||
# Keep durable_block referenced so the scenario is explicit.
|
||||
self.assertTrue(durable_block["block"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,453 @@
|
||||
"""#897: stale-runtime / runtime-mode refusals must not look like permission denials.
|
||||
|
||||
Acceptance criteria (issue #897):
|
||||
|
||||
* Stale-runtime and runtime-mode refusals are typed distinctly from
|
||||
profile-permission refusals (distinct ``blocker_kind``).
|
||||
* A refusal caused by staleness or runtime mode never emits a
|
||||
``permission_report`` and never names a permission the active profile holds.
|
||||
* ``_permission_block_report`` verifies the active profile actually lacks the
|
||||
operation before reporting it missing.
|
||||
* A stale-runtime refusal reports reconnect-only recovery and never recommends
|
||||
``gitea_activate_profile`` or an MCP session switch.
|
||||
* The blocker payload states the observed heads (parity fields).
|
||||
* Matrix across author / reviewer / merger / reconciler profiles.
|
||||
* Regression: ``gitea_create_issue`` on a stale daemon under ``prgs-author``
|
||||
never returns ``missing_permission: gitea.issue.create``.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parent.parent))
|
||||
|
||||
import gitea_config # noqa: E402
|
||||
import gitea_mcp_server as mcp_server # noqa: E402
|
||||
|
||||
SHA_START = "7af40fb5ff7debd5e9165fe97d9c7c279358e175"
|
||||
SHA_LIVE = "2f4dec832327513118f2fe92b74da25d124a01cb"
|
||||
|
||||
ROLE_MATRIX = (
|
||||
(
|
||||
"prgs-author",
|
||||
"author",
|
||||
"gitea.issue.create",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.issue.create",
|
||||
"gitea.issue.comment",
|
||||
"gitea.issue.close",
|
||||
"gitea.branch.create",
|
||||
"gitea.branch.push",
|
||||
"gitea.pr.create",
|
||||
"gitea.pr.comment",
|
||||
"gitea.repo.commit",
|
||||
],
|
||||
["gitea.pr.approve", "gitea.pr.merge", "gitea.pr.request_changes"],
|
||||
"gitea.pr.merge", # forbidden op for pure-permission case
|
||||
),
|
||||
(
|
||||
"prgs-reviewer",
|
||||
"reviewer",
|
||||
"gitea.pr.review",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.pr.review",
|
||||
"gitea.pr.approve",
|
||||
"gitea.pr.request_changes",
|
||||
"gitea.pr.comment",
|
||||
"gitea.issue.comment",
|
||||
],
|
||||
["gitea.branch.push", "gitea.pr.create"],
|
||||
"gitea.branch.push",
|
||||
),
|
||||
(
|
||||
"prgs-merger",
|
||||
"merger",
|
||||
"gitea.pr.merge",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.pr.merge",
|
||||
"gitea.pr.comment",
|
||||
"gitea.issue.comment",
|
||||
],
|
||||
["gitea.pr.approve", "gitea.branch.push", "gitea.pr.create"],
|
||||
"gitea.branch.push",
|
||||
),
|
||||
(
|
||||
"prgs-reconciler",
|
||||
"reconciler",
|
||||
"gitea.branch.delete",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.branch.delete",
|
||||
"gitea.pr.comment",
|
||||
"gitea.issue.comment",
|
||||
"gitea.pr.close",
|
||||
"gitea.issue.close",
|
||||
],
|
||||
["gitea.pr.approve", "gitea.pr.merge"],
|
||||
"gitea.pr.merge",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _profile(name: str, role: str, allowed: list[str], forbidden: list[str]) -> dict:
|
||||
return {
|
||||
"profile_name": name,
|
||||
"role": role,
|
||||
"role_kind": role,
|
||||
"allowed_operations": list(allowed),
|
||||
"forbidden_operations": list(forbidden),
|
||||
"identity": "test-user",
|
||||
}
|
||||
|
||||
|
||||
def _config(profiles: dict) -> dict:
|
||||
return {
|
||||
"version": 2,
|
||||
"profiles": {
|
||||
name: {
|
||||
"role": p["role"],
|
||||
"allowed_operations": p["allowed_operations"],
|
||||
"forbidden_operations": p["forbidden_operations"],
|
||||
}
|
||||
for name, p in profiles.items()
|
||||
},
|
||||
"rules": {"allow_runtime_switching": True},
|
||||
}
|
||||
|
||||
|
||||
class Issue897Helpers(unittest.TestCase):
|
||||
def test_classify_stale_reason_strings(self):
|
||||
stale = (
|
||||
f"live remote master is {SHA_LIVE[:12]} but the MCP server started "
|
||||
f"at {SHA_START[:12]}; the daemon is stale relative to live master "
|
||||
"-- restart/reconnect before mutating"
|
||||
)
|
||||
classified = mcp_server._classify_operation_gate_reasons([stale])
|
||||
self.assertEqual(classified["stale_runtime"], [stale])
|
||||
self.assertEqual(classified["permission"], [])
|
||||
self.assertEqual(classified["runtime_mode"], [])
|
||||
|
||||
def test_classify_permission_reason(self):
|
||||
reason = "profile is not allowed to gitea.pr.merge"
|
||||
classified = mcp_server._classify_operation_gate_reasons([reason])
|
||||
self.assertEqual(classified["permission"], [reason])
|
||||
self.assertEqual(classified["stale_runtime"], [])
|
||||
|
||||
def test_classify_runtime_mode_reason(self):
|
||||
reason = (
|
||||
"runtime mode is 'dev-test' and the mutation targets the "
|
||||
"production repository; dev/test runtimes must not mutate real "
|
||||
"issues or PRs (ADR: stable control runtime vs dev runtime)"
|
||||
)
|
||||
classified = mcp_server._classify_operation_gate_reasons([reason])
|
||||
self.assertEqual(classified["runtime_mode"], [reason])
|
||||
self.assertEqual(classified["stale_runtime"], [])
|
||||
|
||||
|
||||
class Issue897PermissionBlockReport(unittest.TestCase):
|
||||
def test_holds_op_is_diagnostic_defect_not_missing_permission(self):
|
||||
profile = _profile(
|
||||
"prgs-author",
|
||||
"author",
|
||||
["gitea.read", "gitea.issue.create", "gitea.issue.comment"],
|
||||
[],
|
||||
)
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server.gitea_config, "load_config", return_value=_config({"prgs-author": profile})
|
||||
), patch.object(
|
||||
mcp_server.gitea_config, "is_runtime_switching_enabled", return_value=True
|
||||
):
|
||||
report = mcp_server._permission_block_report("gitea.issue.create")
|
||||
self.assertTrue(report.get("diagnostic_defect"), report)
|
||||
self.assertIsNone(report.get("missing_permission"), report)
|
||||
action = (report.get("exact_safe_next_action") or "").lower()
|
||||
# Must not *recommend* profile switching; mentioning the forbidden
|
||||
# action in a "do not call" instruction is fine.
|
||||
self.assertNotIn("call gitea_activate_profile with", action)
|
||||
self.assertNotIn("switch to the author mcp session", action)
|
||||
self.assertNotIn("switch to the reviewer mcp session", action)
|
||||
self.assertIn("diagnostic defect", action)
|
||||
|
||||
def test_true_missing_permission_still_reports(self):
|
||||
profile = _profile(
|
||||
"prgs-author",
|
||||
"author",
|
||||
["gitea.read", "gitea.issue.create"],
|
||||
["gitea.pr.merge"],
|
||||
)
|
||||
reviewer = _profile(
|
||||
"prgs-reviewer",
|
||||
"reviewer",
|
||||
["gitea.read", "gitea.pr.merge", "gitea.pr.approve"],
|
||||
[],
|
||||
)
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server.gitea_config,
|
||||
"load_config",
|
||||
return_value=_config({"prgs-author": profile, "prgs-reviewer": reviewer}),
|
||||
), patch.object(
|
||||
mcp_server.gitea_config, "is_runtime_switching_enabled", return_value=True
|
||||
):
|
||||
report = mcp_server._permission_block_report("gitea.pr.merge")
|
||||
self.assertFalse(report.get("diagnostic_defect"), report)
|
||||
self.assertEqual(report.get("missing_permission"), "gitea.pr.merge")
|
||||
self.assertIn("prgs-reviewer", report.get("matching_configured_profiles") or [])
|
||||
|
||||
|
||||
class Issue897GateRefusalMatrix(unittest.TestCase):
|
||||
def _stale_parity(self) -> dict:
|
||||
return {
|
||||
"in_parity": True,
|
||||
"stale": False,
|
||||
"restart_required": True,
|
||||
"determinable": True,
|
||||
"startup_head": SHA_START,
|
||||
"current_head": SHA_START,
|
||||
"daemon_start_head": SHA_START,
|
||||
"local_head": SHA_START,
|
||||
"live_remote_head": SHA_LIVE,
|
||||
"live_known": True,
|
||||
"live_stale": True,
|
||||
"mutation_safe": False,
|
||||
"reasons": [
|
||||
f"live remote master is {SHA_LIVE[:12]} but the MCP server "
|
||||
f"started at {SHA_START[:12]}; the daemon is stale relative "
|
||||
"to live master -- restart/reconnect before mutating"
|
||||
],
|
||||
}
|
||||
|
||||
def test_stale_plus_permitted_op_all_roles(self):
|
||||
for name, role, permitted_op, allowed, forbidden, _forbidden_op in ROLE_MATRIX:
|
||||
with self.subTest(profile=name, op=permitted_op):
|
||||
profile = _profile(name, role, allowed, forbidden)
|
||||
parity = self._stale_parity()
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_current_master_parity", return_value=parity
|
||||
), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=list(parity["reasons"])
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": name},
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block(permitted_op)
|
||||
self.assertIsNotNone(blocked, name)
|
||||
assert blocked is not None
|
||||
self.assertEqual(
|
||||
blocked.get("blocker_kind"),
|
||||
"runtime_reconnect_required",
|
||||
blocked,
|
||||
)
|
||||
self.assertNotIn("permission_report", blocked, blocked)
|
||||
self.assertTrue(blocked.get("restart_required"), blocked)
|
||||
self.assertEqual(blocked.get("startup_head"), SHA_START, blocked)
|
||||
self.assertEqual(blocked.get("live_remote_head"), SHA_LIVE, blocked)
|
||||
action = (blocked.get("exact_safe_next_action") or "").lower()
|
||||
self.assertIn("reconnect", action)
|
||||
self.assertNotIn("call gitea_activate_profile with", action)
|
||||
self.assertNotIn("switch to the author mcp session", action)
|
||||
self.assertNotIn("switch to the reviewer mcp session", action)
|
||||
|
||||
def test_fresh_plus_forbidden_op_all_roles(self):
|
||||
for name, role, _permitted, allowed, forbidden, forbidden_op in ROLE_MATRIX:
|
||||
with self.subTest(profile=name, op=forbidden_op):
|
||||
profile = _profile(name, role, allowed, forbidden)
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": name},
|
||||
), patch.object(
|
||||
mcp_server.gitea_config,
|
||||
"load_config",
|
||||
return_value=_config({name: profile}),
|
||||
), patch.object(
|
||||
mcp_server.gitea_config, "is_runtime_switching_enabled", return_value=False
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block(forbidden_op)
|
||||
self.assertIsNotNone(blocked, name)
|
||||
assert blocked is not None
|
||||
self.assertEqual(blocked.get("blocker_kind"), "permission_denied", blocked)
|
||||
self.assertIn("permission_report", blocked, blocked)
|
||||
report = blocked["permission_report"]
|
||||
self.assertEqual(report.get("missing_permission"), forbidden_op, report)
|
||||
self.assertFalse(report.get("diagnostic_defect"), report)
|
||||
# No runtime reconnect fields for pure permission denial
|
||||
self.assertNotEqual(
|
||||
blocked.get("blocker_kind"), "runtime_reconnect_required"
|
||||
)
|
||||
|
||||
def test_stale_plus_forbidden_op_both_causes_separated(self):
|
||||
for name, role, _permitted, allowed, forbidden, forbidden_op in ROLE_MATRIX:
|
||||
with self.subTest(profile=name, op=forbidden_op):
|
||||
profile = _profile(name, role, allowed, forbidden)
|
||||
parity = self._stale_parity()
|
||||
stale_reason = parity["reasons"][0]
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_current_master_parity", return_value=parity
|
||||
), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=[stale_reason]
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": name},
|
||||
):
|
||||
# Gate collects both classes; force permission reason too.
|
||||
with patch.object(
|
||||
mcp_server,
|
||||
"_profile_operation_gate",
|
||||
return_value=[
|
||||
stale_reason,
|
||||
f"profile is not allowed to {forbidden_op}",
|
||||
],
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block(forbidden_op)
|
||||
self.assertIsNotNone(blocked)
|
||||
assert blocked is not None
|
||||
self.assertEqual(
|
||||
blocked.get("blocker_kind"), "runtime_reconnect_required", blocked
|
||||
)
|
||||
self.assertNotIn("permission_report", blocked, blocked)
|
||||
self.assertIn("permission_block_reasons", blocked, blocked)
|
||||
self.assertIn("stale_runtime_reasons", blocked, blocked)
|
||||
classes = blocked.get("gate_reason_classes") or {}
|
||||
self.assertTrue(classes.get("stale_runtime"), classes)
|
||||
self.assertTrue(classes.get("permission"), classes)
|
||||
|
||||
def test_runtime_mode_block_no_permission_report(self):
|
||||
profile = _profile(
|
||||
"prgs-author",
|
||||
"author",
|
||||
["gitea.read", "gitea.issue.create"],
|
||||
[],
|
||||
)
|
||||
runtime_reason = (
|
||||
"runtime mode is 'dev-test' and the mutation targets the "
|
||||
"production repository; dev/test runtimes must not mutate real "
|
||||
"issues or PRs (ADR: stable control runtime vs dev runtime)"
|
||||
)
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[runtime_reason]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": "prgs-author"},
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block("gitea.issue.create")
|
||||
self.assertIsNotNone(blocked)
|
||||
assert blocked is not None
|
||||
self.assertEqual(blocked.get("blocker_kind"), "runtime_mode_blocked", blocked)
|
||||
self.assertNotIn("permission_report", blocked, blocked)
|
||||
action = (blocked.get("exact_safe_next_action") or "").lower()
|
||||
self.assertNotIn("call gitea_activate_profile with", action)
|
||||
self.assertIn("stable control runtime", action)
|
||||
|
||||
|
||||
class Issue897CreateIssueRegression(unittest.TestCase):
|
||||
def test_create_issue_stale_daemon_never_missing_issue_create(self):
|
||||
"""Regression AC: stale prgs-author create_issue must not claim missing create."""
|
||||
profile = _profile(
|
||||
"prgs-author",
|
||||
"author",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.issue.create",
|
||||
"gitea.issue.comment",
|
||||
"gitea.branch.create",
|
||||
"gitea.branch.push",
|
||||
"gitea.pr.create",
|
||||
"gitea.pr.comment",
|
||||
"gitea.repo.commit",
|
||||
],
|
||||
[],
|
||||
)
|
||||
stale_reason = (
|
||||
f"live remote master is {SHA_LIVE[:12]} but the MCP server started "
|
||||
f"at {SHA_START[:12]}; the daemon is stale relative to live master "
|
||||
"-- restart/reconnect before mutating"
|
||||
)
|
||||
parity = {
|
||||
"in_parity": True,
|
||||
"stale": False,
|
||||
"restart_required": True,
|
||||
"determinable": True,
|
||||
"startup_head": SHA_START,
|
||||
"current_head": SHA_START,
|
||||
"daemon_start_head": SHA_START,
|
||||
"local_head": SHA_START,
|
||||
"live_remote_head": SHA_LIVE,
|
||||
"live_known": True,
|
||||
"live_stale": True,
|
||||
"mutation_safe": False,
|
||||
"reasons": [stale_reason],
|
||||
}
|
||||
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_current_master_parity", return_value=parity
|
||||
), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=[stale_reason]
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": "prgs-author"},
|
||||
), patch.object(
|
||||
mcp_server, "_mutation_config_authority_block", return_value=None
|
||||
), patch.object(
|
||||
mcp_server, "_session_context_mutation_block", return_value=None
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block(
|
||||
"gitea.issue.create", remote="prgs"
|
||||
)
|
||||
|
||||
self.assertIsNotNone(blocked)
|
||||
assert blocked is not None
|
||||
self.assertEqual(blocked.get("blocker_kind"), "runtime_reconnect_required")
|
||||
self.assertNotIn("permission_report", blocked)
|
||||
# Even if a caller still built a raw report, holds-check must not claim missing.
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile):
|
||||
raw = mcp_server._permission_block_report("gitea.issue.create")
|
||||
self.assertIsNone(raw.get("missing_permission"), raw)
|
||||
self.assertNotEqual(raw.get("missing_permission"), "gitea.issue.create")
|
||||
|
||||
def test_permission_report_for_gate_reasons_skips_stale(self):
|
||||
stale = (
|
||||
f"live remote master is {SHA_LIVE[:12]} but the MCP server started "
|
||||
f"at {SHA_START[:12]}; the daemon is stale relative to live master "
|
||||
"-- restart/reconnect before mutating"
|
||||
)
|
||||
self.assertIsNone(
|
||||
mcp_server._permission_report_for_gate_reasons(
|
||||
"gitea.issue.comment", [stale]
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,346 @@
|
||||
"""Regression: author bootstrap scope reaches workflow_scope_guard (#941).
|
||||
|
||||
PR #926 (#892) made ``bootstrap_permits_control_checkout`` accept
|
||||
``task_scope=author_issue_bootstrap`` and wired that canonical decision into
|
||||
the #274 branches-only enforcer and the #604 anti-stomp preflight. A third
|
||||
enforcement path was left unwired.
|
||||
|
||||
``workflow_scope_guard.assess_root_source_mutation`` kept its own copy of the
|
||||
clean-root author decision, gated on ``create_issue_bootstrap.is_create_issue_task``
|
||||
— a task-name allowlist that never contained ``bootstrap_author_issue_worktree``.
|
||||
So the real call path
|
||||
|
||||
gitea_bootstrap_author_issue_worktree
|
||||
-> verify_preflight_purity
|
||||
-> _enforce_issue_scope_guard
|
||||
-> workflow_scope_guard.assess_production_mutation_guards
|
||||
|
||||
raised ProductionGuardError(missing_issue_worktree) before
|
||||
``assess_author_issue_bootstrap`` was ever consulted.
|
||||
|
||||
These tests drive the real enforcer, not the authorization helper in
|
||||
isolation. A helper-only test cannot observe this defect: #892's own predicate
|
||||
tests all passed while the live bootstrap stayed blocked.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest import mock
|
||||
|
||||
import author_issue_bootstrap as aib
|
||||
import create_issue_bootstrap as cib
|
||||
import workflow_scope_guard
|
||||
|
||||
BOOTSTRAP_TASK = "bootstrap_author_issue_worktree"
|
||||
BOOTSTRAP_TOOL = "gitea_bootstrap_author_issue_worktree"
|
||||
|
||||
|
||||
def _make_control_repo(tmp: str) -> tuple[str, str]:
|
||||
"""Create a clean control checkout on master and return (path, head)."""
|
||||
repo = os.path.join(tmp, "repo")
|
||||
os.makedirs(os.path.join(repo, "branches"))
|
||||
subprocess.check_call(
|
||||
["git", "init", "-b", "master", repo],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
subprocess.check_call(
|
||||
[
|
||||
"git", "-C", repo,
|
||||
"-c", "user.email=t@t", "-c", "user.name=t",
|
||||
"commit", "--allow-empty", "-m", "init",
|
||||
],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
head = subprocess.check_output(
|
||||
["git", "-C", repo, "rev-parse", "HEAD"], text=True
|
||||
).strip()
|
||||
return repo, head
|
||||
|
||||
|
||||
def _assessment(
|
||||
repo: str,
|
||||
head: str,
|
||||
*,
|
||||
task: str = BOOTSTRAP_TASK,
|
||||
porcelain: str = "",
|
||||
remote: str | None = None,
|
||||
) -> dict:
|
||||
return aib.assess_author_issue_bootstrap(
|
||||
workspace_path=repo,
|
||||
canonical_repo_root=repo,
|
||||
current_branch="master",
|
||||
head_sha=head,
|
||||
porcelain_status=porcelain,
|
||||
remote_master_sha=head if remote is None else remote,
|
||||
task=task,
|
||||
)
|
||||
|
||||
|
||||
class _ControlCheckoutHarness(unittest.TestCase):
|
||||
"""Drive the real server guard against a temporary clean control checkout."""
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.repo, self.head = _make_control_repo(self._tmp.name)
|
||||
|
||||
# #683 force-on: production guards must execute under pytest.
|
||||
patcher = mock.patch.dict(
|
||||
os.environ,
|
||||
{workflow_scope_guard.FORCE_PRODUCTION_GUARDS_ENV: "1"},
|
||||
)
|
||||
patcher.start()
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
def _enforce(
|
||||
self,
|
||||
task: str,
|
||||
*,
|
||||
porcelain: str = "",
|
||||
assessment: object = "auto",
|
||||
role_kind: str = "author",
|
||||
):
|
||||
"""Call the real _enforce_issue_scope_guard for *task*."""
|
||||
import gitea_mcp_server as srv
|
||||
|
||||
if assessment == "auto":
|
||||
assessment = _assessment(
|
||||
self.repo, self.head, task=task, porcelain=porcelain
|
||||
)
|
||||
|
||||
ctx = {
|
||||
"workspace_path": self.repo,
|
||||
"canonical_repo_root": self.repo,
|
||||
"workspace_role_kind": role_kind,
|
||||
"workspace_binding_source": "process_project_root",
|
||||
}
|
||||
git_state = {
|
||||
"current_branch": "master",
|
||||
"head_sha": self.head,
|
||||
"porcelain_status": porcelain,
|
||||
}
|
||||
|
||||
with mock.patch.object(
|
||||
srv, "_resolve_namespace_mutation_context", return_value=ctx
|
||||
), mock.patch.object(
|
||||
srv.issue_lock_worktree,
|
||||
"read_worktree_git_state",
|
||||
return_value=git_state,
|
||||
), mock.patch.object(
|
||||
srv,
|
||||
"_session_issue_lock_snapshot",
|
||||
return_value={
|
||||
"locked_issue_number": None,
|
||||
"lock_branch_name": None,
|
||||
"worktrees_match": False,
|
||||
},
|
||||
), mock.patch.object(
|
||||
srv, "_actual_profile_role", return_value=role_kind
|
||||
), mock.patch.object(
|
||||
srv, "_effective_workspace_role", return_value=role_kind
|
||||
), mock.patch.object(
|
||||
srv, "_create_issue_bootstrap_assessment", return_value=assessment
|
||||
):
|
||||
srv._enforce_issue_scope_guard(None, task=task)
|
||||
|
||||
|
||||
class TestRealPathBootstrapReachesGuard(_ControlCheckoutHarness):
|
||||
"""The defect and its fix, observed through the real enforcer."""
|
||||
|
||||
def test_bootstrap_task_passes_scope_guard_from_clean_control(self):
|
||||
# Pre-fix this raises ProductionGuardError(missing_issue_worktree)
|
||||
# because the guard consulted a task-name allowlist instead of the
|
||||
# canonical authorization decision.
|
||||
self._enforce(BOOTSTRAP_TASK)
|
||||
|
||||
def test_bootstrap_tool_alias_passes_scope_guard(self):
|
||||
self._enforce(BOOTSTRAP_TOOL)
|
||||
|
||||
def test_guard_consults_canonical_predicate(self):
|
||||
"""The guard must reach bootstrap_permits_control_checkout, not a name list."""
|
||||
real = cib.bootstrap_permits_control_checkout
|
||||
seen: list[str | None] = []
|
||||
|
||||
def _spy(assessment, *, task, workspace_path, canonical_repo_root):
|
||||
seen.append(task)
|
||||
return real(
|
||||
assessment,
|
||||
task=task,
|
||||
workspace_path=workspace_path,
|
||||
canonical_repo_root=canonical_repo_root,
|
||||
)
|
||||
|
||||
with mock.patch.object(
|
||||
cib, "bootstrap_permits_control_checkout", side_effect=_spy
|
||||
):
|
||||
self._enforce(BOOTSTRAP_TASK)
|
||||
|
||||
self.assertIn(
|
||||
BOOTSTRAP_TASK,
|
||||
seen,
|
||||
"workflow_scope_guard did not consult the canonical bootstrap "
|
||||
"authorization decision",
|
||||
)
|
||||
|
||||
|
||||
class TestFailClosedOnBadEvidence(_ControlCheckoutHarness):
|
||||
"""Missing, malformed, or mismatched scope evidence must still block."""
|
||||
|
||||
def _assert_blocked(self, **kwargs):
|
||||
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||
self._enforce(BOOTSTRAP_TASK, **kwargs)
|
||||
|
||||
def test_missing_assessment_fails_closed(self):
|
||||
self._assert_blocked(assessment=None)
|
||||
|
||||
def test_malformed_assessment_fails_closed(self):
|
||||
self._assert_blocked(assessment={"allowed": True})
|
||||
|
||||
def test_non_dict_assessment_fails_closed(self):
|
||||
self._assert_blocked(assessment="allowed")
|
||||
|
||||
def test_wrong_task_scope_fails_closed(self):
|
||||
bad = dict(_assessment(self.repo, self.head))
|
||||
bad["task_scope"] = "create_issue_only"
|
||||
self._assert_blocked(assessment=bad)
|
||||
|
||||
def test_nonempty_reasons_fail_closed(self):
|
||||
bad = dict(_assessment(self.repo, self.head), reasons=["note"])
|
||||
self._assert_blocked(assessment=bad)
|
||||
|
||||
def test_mismatched_base_tips_fail_closed(self):
|
||||
bad = dict(_assessment(self.repo, self.head))
|
||||
bad["remote_master_sha"] = "b" * 40
|
||||
self._assert_blocked(assessment=bad)
|
||||
|
||||
def test_unverified_base_tips_fail_closed(self):
|
||||
bad = dict(_assessment(self.repo, self.head), base_tips_verified=False)
|
||||
self._assert_blocked(assessment=bad)
|
||||
|
||||
def test_mismatched_workspace_binding_fails_closed(self):
|
||||
bad = dict(_assessment(self.repo, self.head))
|
||||
bad["workspace_path"] = os.path.join(self.repo, "elsewhere")
|
||||
self._assert_blocked(assessment=bad)
|
||||
|
||||
def test_mismatched_repo_root_binding_fails_closed(self):
|
||||
bad = dict(_assessment(self.repo, self.head))
|
||||
bad["canonical_repo_root"] = os.path.join(self.repo, "other-root")
|
||||
self._assert_blocked(assessment=bad)
|
||||
|
||||
def test_blocked_assessment_fails_closed(self):
|
||||
bad = dict(_assessment(self.repo, self.head), block=True, allowed=False)
|
||||
self._assert_blocked(assessment=bad)
|
||||
|
||||
|
||||
class TestOrdinaryControlCheckoutMutationStillForbidden(_ControlCheckoutHarness):
|
||||
"""The waiver must not leak to ordinary author work."""
|
||||
|
||||
def test_ordinary_author_task_still_blocked(self):
|
||||
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||
self._enforce("commit_files", assessment=None)
|
||||
|
||||
def test_lock_issue_still_blocked_from_control(self):
|
||||
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||
self._enforce("lock_issue", assessment=None)
|
||||
|
||||
def test_bootstrap_assessment_cannot_license_other_task(self):
|
||||
# Cross-task smuggling: valid bootstrap evidence must not waive a
|
||||
# different author mutation.
|
||||
good = _assessment(self.repo, self.head)
|
||||
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||
self._enforce("commit_files", assessment=good)
|
||||
|
||||
def test_dirty_control_checkout_still_blocked_for_bootstrap(self):
|
||||
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||
self._enforce(BOOTSTRAP_TASK, porcelain=" M gitea_mcp_server.py\n")
|
||||
|
||||
|
||||
class TestCreateIssueBehaviorUnchanged(_ControlCheckoutHarness):
|
||||
"""#749 create_issue keeps its own sanctioned path."""
|
||||
|
||||
def test_create_issue_still_allowed_from_clean_control(self):
|
||||
self._enforce("create_issue", assessment=None)
|
||||
|
||||
def test_create_issue_tool_alias_still_allowed(self):
|
||||
self._enforce("gitea_create_issue", assessment=None)
|
||||
|
||||
def test_create_issue_blocked_when_control_dirty(self):
|
||||
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||
self._enforce(
|
||||
"create_issue",
|
||||
porcelain=" M gitea_mcp_server.py\n",
|
||||
assessment=None,
|
||||
)
|
||||
|
||||
|
||||
class TestGuardUnitLevelWiring(unittest.TestCase):
|
||||
"""assess_root_source_mutation itself must accept and honour the evidence."""
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.repo, self.head = _make_control_repo(self._tmp.name)
|
||||
patcher = mock.patch.dict(
|
||||
os.environ,
|
||||
{workflow_scope_guard.FORCE_PRODUCTION_GUARDS_ENV: "1"},
|
||||
)
|
||||
patcher.start()
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
def _assess(self, *, task=BOOTSTRAP_TASK, bootstrap_assessment="auto"):
|
||||
if bootstrap_assessment == "auto":
|
||||
bootstrap_assessment = _assessment(self.repo, self.head, task=task)
|
||||
return workflow_scope_guard.assess_root_source_mutation(
|
||||
workspace_path=self.repo,
|
||||
canonical_repo_root=self.repo,
|
||||
porcelain_status="",
|
||||
current_branch="master",
|
||||
role_kind="author",
|
||||
mutation_task=task,
|
||||
bootstrap_assessment=bootstrap_assessment,
|
||||
)
|
||||
|
||||
def test_valid_evidence_unblocks(self):
|
||||
result = self._assess()
|
||||
self.assertFalse(result["block"])
|
||||
self.assertIsNone(result["blocker_kind"])
|
||||
|
||||
def test_absent_evidence_blocks(self):
|
||||
result = self._assess(bootstrap_assessment=None)
|
||||
self.assertTrue(result["block"])
|
||||
self.assertEqual(
|
||||
result["blocker_kind"], workflow_scope_guard.BLOCKER_MISSING_WORKTREE
|
||||
)
|
||||
|
||||
def test_reconciler_exemption_preserved(self):
|
||||
result = workflow_scope_guard.assess_root_source_mutation(
|
||||
workspace_path=self.repo,
|
||||
canonical_repo_root=self.repo,
|
||||
porcelain_status="",
|
||||
current_branch="master",
|
||||
role_kind="reconciler",
|
||||
mutation_task=BOOTSTRAP_TASK,
|
||||
)
|
||||
self.assertFalse(result["block"])
|
||||
|
||||
def test_signature_accepts_evidence_without_it_being_required(self):
|
||||
# Callers that supply no evidence keep the pre-existing behaviour.
|
||||
result = workflow_scope_guard.assess_root_source_mutation(
|
||||
workspace_path=self.repo,
|
||||
canonical_repo_root=self.repo,
|
||||
porcelain_status="",
|
||||
current_branch="master",
|
||||
role_kind="author",
|
||||
mutation_task="create_issue",
|
||||
)
|
||||
self.assertFalse(result["block"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,747 @@
|
||||
"""Regression: author bootstrap runtime authority and session ownership (#943).
|
||||
|
||||
Two rounds of defects live here.
|
||||
|
||||
**Round 1 (#943 as filed).** ``gitea_bootstrap_author_issue_worktree`` passed
|
||||
four values down to the bootstrap service that were never defined:
|
||||
``_active_username``, ``_active_profile_name``, ``_current_session_id`` and
|
||||
``_author_mutation_block``. Every call — dry-run included — raised
|
||||
``NameError`` while evaluating the arguments, before the service was entered.
|
||||
|
||||
**Round 2 (review 622 on PR #944).** The first fix defined all four but made
|
||||
``_current_session_id`` mint ``<profile>-<pid>-<hex>`` once per process. The MCP
|
||||
daemon outlives every task it serves, so that value conflates sequential author
|
||||
tasks and can never equal the control-plane session that owns an
|
||||
allocator-created lease: ``_verify_assignment_and_lease_ids`` refused the whole
|
||||
allocated path with ``lease_session_mismatch``. The reviewed round also read the
|
||||
identity from the pinned session context while reading the profile from the live
|
||||
profile, so a rebind could produce a mixed claimant pair, and it swallowed every
|
||||
``get_profile()`` exception.
|
||||
|
||||
These tests therefore drive real state, not mocks of internals: a temporary
|
||||
control-plane SQLite database and a temporary issue-lock directory, both
|
||||
redirected through the same environment variables production uses
|
||||
(``GITEA_CONTROL_PLANE_DB``, ``GITEA_ISSUE_LOCK_DIR``). The ownership gate that
|
||||
runs is the real one.
|
||||
|
||||
``test_every_global_referenced_by_the_wrapper_resolves`` remains: it is what
|
||||
found ``_author_mutation_block``, and it generalises to the next missing
|
||||
reference. It supplements the runtime coverage below rather than standing in for
|
||||
it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import builtins
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest import mock
|
||||
|
||||
import author_issue_bootstrap as aib
|
||||
import control_plane_db
|
||||
import create_issue_bootstrap as cib
|
||||
import gitea_mcp_server as gms
|
||||
import issue_lock_store
|
||||
import workflow_scope_guard
|
||||
|
||||
BOOTSTRAP_TASK = "bootstrap_author_issue_worktree"
|
||||
WRAPPER_NAME = "gitea_bootstrap_author_issue_worktree"
|
||||
RUNTIME_HELPERS = (
|
||||
"_active_mutation_authority",
|
||||
"_active_username",
|
||||
"_active_profile_name",
|
||||
"_resolve_owner_workflow_session",
|
||||
"_author_mutation_block",
|
||||
)
|
||||
ORG = "Scaled-Tech-Consulting"
|
||||
REPO = "Gitea-Tools"
|
||||
IDENTITY = "jcwalker3"
|
||||
PROFILE = "prgs-author"
|
||||
|
||||
# A per-task ownership key must carry no process identifier (#790).
|
||||
TASK_KEY_RE = re.compile(r"^author_issue_work-[0-9a-f]{16}$")
|
||||
|
||||
|
||||
def _make_control_repo(tmp: str) -> tuple[str, str]:
|
||||
"""Create a clean control checkout on master and return (path, head)."""
|
||||
repo = os.path.join(tmp, "repo")
|
||||
os.makedirs(os.path.join(repo, "branches"))
|
||||
subprocess.check_call(
|
||||
["git", "init", "-b", "master", repo],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
subprocess.check_call(
|
||||
[
|
||||
"git", "-C", repo,
|
||||
"-c", "user.email=t@t", "-c", "user.name=t",
|
||||
"commit", "--allow-empty", "-m", "init",
|
||||
],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
)
|
||||
head = subprocess.check_output(
|
||||
["git", "-C", repo, "rev-parse", "HEAD"], text=True
|
||||
).strip()
|
||||
return repo, head
|
||||
|
||||
|
||||
def _wrapper_ast() -> ast.FunctionDef:
|
||||
"""Return the AST of the bootstrap wrapper as it exists on disk."""
|
||||
path = os.path.join(
|
||||
os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
|
||||
"gitea_mcp_server.py",
|
||||
)
|
||||
with open(path, encoding="utf-8") as fh:
|
||||
tree = ast.parse(fh.read())
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.FunctionDef) and node.name == WRAPPER_NAME:
|
||||
return node
|
||||
raise AssertionError(f"{WRAPPER_NAME} not found in gitea_mcp_server.py")
|
||||
|
||||
|
||||
class _IsolatedControlPlane(unittest.TestCase):
|
||||
"""Temp control-plane DB and temp issue-lock dir, via production env vars."""
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.tmp = self._tmp.name
|
||||
self.db_path = os.path.join(self.tmp, "control-plane.sqlite3")
|
||||
self.lock_dir = os.path.join(self.tmp, "issue-locks")
|
||||
self.journals = os.path.join(self.tmp, "journals")
|
||||
os.makedirs(self.lock_dir)
|
||||
os.makedirs(self.journals)
|
||||
env = mock.patch.dict(
|
||||
os.environ,
|
||||
{
|
||||
control_plane_db.DB_PATH_ENV: self.db_path,
|
||||
issue_lock_store.LOCK_DIR_ENV: self.lock_dir,
|
||||
},
|
||||
)
|
||||
env.start()
|
||||
self.addCleanup(env.stop)
|
||||
self.db = control_plane_db.ControlPlaneDB(self.db_path)
|
||||
|
||||
def _allocate(self, session_id: str, *, issue: int = 943):
|
||||
"""Create a real assignment + lease owned by *session_id*."""
|
||||
self.db.upsert_session(
|
||||
session_id=session_id, role="author", profile=PROFILE, pid=os.getpid()
|
||||
)
|
||||
res = self.db.assign_and_lease(
|
||||
session_id=session_id, role="author", remote="prgs",
|
||||
org=ORG, repo=REPO, kind="issue", number=issue,
|
||||
)
|
||||
self.assertEqual(res.outcome, "assigned", res)
|
||||
return res.assignment_id, res.lease_id
|
||||
|
||||
def _authority(self):
|
||||
"""A resolved authority pair, as the wrapper would compute it."""
|
||||
return {"ok": True, "identity": IDENTITY, "profile_name": PROFILE}
|
||||
|
||||
def _resolve_session(self, **over):
|
||||
kwargs = dict(
|
||||
issue_number=943,
|
||||
assignment_id=None,
|
||||
lease_id=None,
|
||||
session_id=None,
|
||||
identity=IDENTITY,
|
||||
profile_name=PROFILE,
|
||||
remote="prgs",
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
)
|
||||
kwargs.update(over)
|
||||
return gms._resolve_owner_workflow_session(**kwargs)
|
||||
|
||||
|
||||
class OwnershipGateTests(_IsolatedControlPlane):
|
||||
"""B2: the allocator-driven ownership path, against a real control plane."""
|
||||
|
||||
def setUp(self):
|
||||
super().setUp()
|
||||
self.repo, self.head = _make_control_repo(self.tmp)
|
||||
|
||||
def _bootstrap(self, **over):
|
||||
kwargs = dict(
|
||||
issue_number=943,
|
||||
canonical_repo_root=self.repo,
|
||||
expected_base_sha=self.head,
|
||||
branch_name="fix/issue-943-runtime-context-helpers",
|
||||
remote="prgs",
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
active_identity=IDENTITY,
|
||||
active_profile=PROFILE,
|
||||
lock_dir=self.journals,
|
||||
idempotency_key="test-943",
|
||||
dry_run=True,
|
||||
)
|
||||
kwargs.update(over)
|
||||
return aib.bootstrap_author_issue_worktree(**kwargs)
|
||||
|
||||
def test_true_owning_session_passes_the_ownership_gate(self):
|
||||
"""The canonical owner reaches and completes the service."""
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, lease_id = self._allocate(session)
|
||||
res = self._bootstrap(
|
||||
assignment_id=assignment_id, lease_id=lease_id, owner_session=session
|
||||
)
|
||||
self.assertTrue(res.get("success"), res)
|
||||
self.assertTrue(res.get("dry_run"))
|
||||
self.assertEqual(res.get("base_sha"), self.head)
|
||||
|
||||
def test_different_session_is_refused(self):
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, lease_id = self._allocate(session)
|
||||
res = self._bootstrap(
|
||||
assignment_id=assignment_id,
|
||||
lease_id=lease_id,
|
||||
owner_session="prgs-author-task-b",
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertEqual(res.get("reason_code"), "lease_session_mismatch")
|
||||
|
||||
def test_process_derived_session_would_be_refused(self):
|
||||
"""The reviewed round-1 value shape can never own an allocated lease."""
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, lease_id = self._allocate(session)
|
||||
round_one_value = f"{PROFILE}-{os.getpid()}-deadbeef"
|
||||
self.assertNotEqual(round_one_value, session)
|
||||
res = self._bootstrap(
|
||||
assignment_id=assignment_id,
|
||||
lease_id=lease_id,
|
||||
owner_session=round_one_value,
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertEqual(res.get("reason_code"), "lease_session_mismatch")
|
||||
|
||||
def test_unknown_lease_fails_closed(self):
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, _ = self._allocate(session)
|
||||
res = self._bootstrap(
|
||||
assignment_id=assignment_id,
|
||||
lease_id="lease-does-not-exist",
|
||||
owner_session=session,
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertEqual(res.get("reason_code"), "unknown_lease_id")
|
||||
|
||||
def test_released_lease_fails_closed(self):
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, lease_id = self._allocate(session)
|
||||
self.db.release_lease(lease_id, session_id=session)
|
||||
res = self._bootstrap(
|
||||
assignment_id=assignment_id, lease_id=lease_id, owner_session=session
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertEqual(res.get("reason_code"), "lease_not_live")
|
||||
|
||||
def test_force_expired_lease_fails_closed(self):
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, lease_id = self._allocate(session)
|
||||
self.db.force_expire_lease(lease_id, reason="test")
|
||||
res = self._bootstrap(
|
||||
assignment_id=assignment_id, lease_id=lease_id, owner_session=session
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertEqual(res.get("reason_code"), "lease_not_live")
|
||||
|
||||
def test_replacement_lease_does_not_inherit_prior_ownership(self):
|
||||
"""A second task's lease is not ownable by the first task's session."""
|
||||
first = "prgs-author-task-a"
|
||||
assignment_a, lease_a = self._allocate(first)
|
||||
self.db.release_lease(lease_a, session_id=first)
|
||||
second = "prgs-author-task-b"
|
||||
assignment_b, lease_b = self._allocate(second)
|
||||
self.assertNotEqual(lease_a, lease_b)
|
||||
res = self._bootstrap(
|
||||
assignment_id=assignment_b, lease_id=lease_b, owner_session=first
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertEqual(res.get("reason_code"), "lease_session_mismatch")
|
||||
|
||||
def test_assignment_lease_identifier_mismatch_fails_closed(self):
|
||||
session = "prgs-author-task-a"
|
||||
_, lease_id = self._allocate(session)
|
||||
res = self._bootstrap(
|
||||
assignment_id="asn-not-the-recorded-one",
|
||||
lease_id=lease_id,
|
||||
owner_session=session,
|
||||
)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertEqual(res.get("reason_code"), "assignment_lease_mismatch")
|
||||
|
||||
def test_lease_id_without_assignment_id_fails_closed(self):
|
||||
session = "prgs-author-task-a"
|
||||
_, lease_id = self._allocate(session)
|
||||
res = self._bootstrap(lease_id=lease_id, owner_session=session)
|
||||
self.assertFalse(res.get("success"))
|
||||
self.assertEqual(res.get("reason_code"), "incomplete_assignment_lease_ids")
|
||||
|
||||
def test_dry_run_with_valid_allocator_bindings_leaves_no_durable_state(self):
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, lease_id = self._allocate(session)
|
||||
res = self._bootstrap(
|
||||
assignment_id=assignment_id, lease_id=lease_id, owner_session=session
|
||||
)
|
||||
self.assertTrue(res.get("success"), res)
|
||||
|
||||
branches = subprocess.check_output(
|
||||
["git", "-C", self.repo, "branch", "--list"], text=True
|
||||
)
|
||||
self.assertNotIn("issue-943", branches)
|
||||
worktrees = subprocess.check_output(
|
||||
["git", "-C", self.repo, "worktree", "list"], text=True
|
||||
)
|
||||
self.assertNotIn("issue-943", worktrees)
|
||||
self.assertFalse(
|
||||
os.path.exists(
|
||||
os.path.join(self.repo, "branches",
|
||||
"fix-issue-943-runtime-context-helpers")
|
||||
)
|
||||
)
|
||||
journal = res.get("phase_journal") or {}
|
||||
self.assertFalse(journal.get("completed"))
|
||||
self.assertFalse(any((journal.get("artifacts_created") or {}).values()))
|
||||
# The dry run must not have created an issue lock in the isolated dir.
|
||||
self.assertEqual(os.listdir(self.lock_dir), [])
|
||||
|
||||
def test_apply_reaches_the_intended_transition_with_valid_bindings(self):
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, lease_id = self._allocate(session)
|
||||
res = self._bootstrap(
|
||||
assignment_id=assignment_id,
|
||||
lease_id=lease_id,
|
||||
owner_session=session,
|
||||
dry_run=False,
|
||||
)
|
||||
self.assertTrue(res.get("success"), res)
|
||||
self.assertNotEqual(res.get("dry_run"), True)
|
||||
branches = subprocess.check_output(
|
||||
["git", "-C", self.repo, "branch", "--list"], text=True
|
||||
)
|
||||
self.assertIn("issue-943", branches)
|
||||
self.assertTrue(os.path.isdir(res.get("worktree_path") or ""))
|
||||
|
||||
|
||||
class WorkflowSessionResolutionTests(_IsolatedControlPlane):
|
||||
"""B1: the wrapper resolves the owning session, never a process identifier."""
|
||||
|
||||
def test_declared_session_is_verified_against_the_control_plane(self):
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, lease_id = self._allocate(session)
|
||||
res = self._resolve_session(
|
||||
session_id=session, assignment_id=assignment_id, lease_id=lease_id
|
||||
)
|
||||
self.assertTrue(res.get("ok"), res)
|
||||
self.assertEqual(res.get("session_id"), session)
|
||||
self.assertEqual(res.get("session_source"), "declared")
|
||||
|
||||
def test_unknown_declared_session_is_refused_not_trusted(self):
|
||||
res = self._resolve_session(session_id="prgs-author-not-a-session")
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(res.get("reason_code"), "workflow_session_unverified")
|
||||
|
||||
def test_declared_session_for_another_role_is_refused(self):
|
||||
self.db.upsert_session(
|
||||
session_id="prgs-reviewer-x", role="reviewer", profile="prgs-reviewer",
|
||||
pid=os.getpid(),
|
||||
)
|
||||
res = self._resolve_session(session_id="prgs-reviewer-x")
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(res.get("reason_code"), "workflow_session_unverified")
|
||||
|
||||
def test_declared_session_for_another_profile_is_refused(self):
|
||||
self.db.upsert_session(
|
||||
session_id="other-profile-session", role="author",
|
||||
profile="prgs-controller", pid=os.getpid(),
|
||||
)
|
||||
res = self._resolve_session(session_id="other-profile-session")
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(res.get("reason_code"), "workflow_session_unverified")
|
||||
|
||||
def test_allocated_work_without_a_session_is_refused(self):
|
||||
"""Supplying a lease is not itself evidence of ownership."""
|
||||
session = "prgs-author-task-a"
|
||||
assignment_id, lease_id = self._allocate(session)
|
||||
res = self._resolve_session(assignment_id=assignment_id, lease_id=lease_id)
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(
|
||||
res.get("reason_code"), "workflow_session_required_for_allocated_work"
|
||||
)
|
||||
|
||||
def test_existing_issue_lock_supplies_its_per_task_session(self):
|
||||
lock_session = issue_lock_store.mint_task_session_id(
|
||||
issue_lock_store.AUTHOR_ISSUE_WORK_LEASE
|
||||
)
|
||||
path = issue_lock_store.lock_file_path(
|
||||
remote="prgs", org=ORG, repo=REPO, issue_number=943,
|
||||
lock_dir=self.lock_dir,
|
||||
)
|
||||
issue_lock_store.write_lock_file(
|
||||
path,
|
||||
{
|
||||
"issue_number": 943,
|
||||
"branch_name": "fix/issue-943-runtime-context-helpers",
|
||||
"work_lease": {
|
||||
"task_session_id": lock_session,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
},
|
||||
},
|
||||
) if hasattr(issue_lock_store, "write_lock_file") else _write_json(
|
||||
path,
|
||||
{
|
||||
"issue_number": 943,
|
||||
"branch_name": "fix/issue-943-runtime-context-helpers",
|
||||
"work_lease": {
|
||||
"task_session_id": lock_session,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
},
|
||||
},
|
||||
)
|
||||
res = self._resolve_session()
|
||||
self.assertTrue(res.get("ok"), res)
|
||||
self.assertEqual(res.get("session_id"), lock_session)
|
||||
self.assertEqual(res.get("session_source"), "issue_lock")
|
||||
|
||||
def test_issue_lock_owned_by_another_identity_is_refused(self):
|
||||
path = issue_lock_store.lock_file_path(
|
||||
remote="prgs", org=ORG, repo=REPO, issue_number=943,
|
||||
lock_dir=self.lock_dir,
|
||||
)
|
||||
_write_json(
|
||||
path,
|
||||
{
|
||||
"issue_number": 943,
|
||||
"work_lease": {
|
||||
"task_session_id": "author_issue_work-" + "0" * 16,
|
||||
"claimant": {"username": "someone-else", "profile": PROFILE},
|
||||
},
|
||||
},
|
||||
)
|
||||
res = self._resolve_session()
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(res.get("reason_code"), "issue_lock_owner_mismatch")
|
||||
|
||||
def test_unallocated_bootstrap_mints_a_per_task_key(self):
|
||||
res = self._resolve_session()
|
||||
self.assertTrue(res.get("ok"), res)
|
||||
self.assertEqual(res.get("session_source"), "minted_task_key")
|
||||
self.assertRegex(res["session_id"], TASK_KEY_RE)
|
||||
|
||||
def test_minted_key_contains_no_process_identifier(self):
|
||||
res = self._resolve_session()
|
||||
self.assertNotIn(str(os.getpid()), res["session_id"])
|
||||
self.assertNotIn(PROFILE, res["session_id"])
|
||||
|
||||
def test_sequential_tasks_on_one_daemon_do_not_share_ownership(self):
|
||||
"""The round-1 defect: one identifier per process for every task."""
|
||||
first = self._resolve_session()["session_id"]
|
||||
second = self._resolve_session()["session_id"]
|
||||
third = self._resolve_session()["session_id"]
|
||||
self.assertNotEqual(first, second)
|
||||
self.assertNotEqual(second, third)
|
||||
self.assertEqual(len({first, second, third}), 3)
|
||||
|
||||
def test_no_process_lifetime_cache_remains(self):
|
||||
self.assertFalse(hasattr(gms, "_ACTIVE_SESSION_ID"))
|
||||
self.assertFalse(hasattr(gms, "_current_session_id"))
|
||||
|
||||
|
||||
class MutationAuthorityTests(unittest.TestCase):
|
||||
"""F3/F4: one coherent authority pair, drift detected, no silent fallback."""
|
||||
|
||||
def _ctx(self, **over):
|
||||
base = {"identity": IDENTITY, "profile_name": PROFILE}
|
||||
base.update(over)
|
||||
return base
|
||||
|
||||
def test_matching_live_and_pinned_authority_resolves(self):
|
||||
with mock.patch.object(gms, "get_profile",
|
||||
return_value={"profile_name": PROFILE}), \
|
||||
mock.patch.object(gms, "_authenticated_username",
|
||||
return_value=IDENTITY), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value=self._ctx()):
|
||||
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||
self.assertTrue(res.get("ok"), res)
|
||||
self.assertEqual(res["identity"], IDENTITY)
|
||||
self.assertEqual(res["profile_name"], PROFILE)
|
||||
|
||||
def test_identity_drift_fails_closed(self):
|
||||
with mock.patch.object(gms, "get_profile",
|
||||
return_value={"profile_name": PROFILE}), \
|
||||
mock.patch.object(gms, "_authenticated_username",
|
||||
return_value="someone-else"), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value=self._ctx()):
|
||||
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(res.get("reason_code"), "authority_identity_drift")
|
||||
self.assertEqual(res.get("expected"), IDENTITY)
|
||||
self.assertEqual(res.get("actual"), "someone-else")
|
||||
|
||||
def test_profile_drift_fails_closed(self):
|
||||
with mock.patch.object(gms, "get_profile",
|
||||
return_value={"profile_name": "prgs-controller"}), \
|
||||
mock.patch.object(gms, "_authenticated_username",
|
||||
return_value=IDENTITY), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value=self._ctx()):
|
||||
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(res.get("reason_code"), "authority_profile_drift")
|
||||
|
||||
def test_identity_and_profile_never_come_from_different_snapshots(self):
|
||||
"""Round 2's mixed pair: pinned identity plus live profile."""
|
||||
with mock.patch.object(gms, "get_profile",
|
||||
return_value={"profile_name": "prgs-controller"}), \
|
||||
mock.patch.object(gms, "_authenticated_username",
|
||||
return_value="new-identity"), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value=self._ctx()):
|
||||
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertIsNone(gms._active_username("gitea.prgs.cc"))
|
||||
self.assertIsNone(gms._active_profile_name("gitea.prgs.cc"))
|
||||
|
||||
def test_unresolvable_profile_is_a_structured_refusal_not_a_fallback(self):
|
||||
"""F4: no bare-except fallback to a previously pinned profile name."""
|
||||
with mock.patch.object(gms, "get_profile",
|
||||
side_effect=RuntimeError("profile disabled")), \
|
||||
mock.patch.object(gms, "_authenticated_username",
|
||||
return_value=IDENTITY), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value=self._ctx()):
|
||||
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(res.get("reason_code"), "authority_profile_unresolved")
|
||||
self.assertNotEqual(res.get("profile_name"), PROFILE)
|
||||
|
||||
def test_malformed_profile_without_name_fails_closed(self):
|
||||
with mock.patch.object(gms, "get_profile", return_value={}), \
|
||||
mock.patch.object(gms, "_authenticated_username",
|
||||
return_value=IDENTITY), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value=None):
|
||||
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(res.get("reason_code"), "authority_profile_unresolved")
|
||||
|
||||
def test_unresolved_identity_fails_closed(self):
|
||||
for value in (None, "", " "):
|
||||
with self.subTest(identity=value):
|
||||
with mock.patch.object(gms, "get_profile",
|
||||
return_value={"profile_name": PROFILE}), \
|
||||
mock.patch.object(gms, "_authenticated_username",
|
||||
return_value=value), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value=None):
|
||||
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(
|
||||
res.get("reason_code"), "authority_identity_unresolved"
|
||||
)
|
||||
|
||||
def test_missing_host_cannot_yield_an_identity(self):
|
||||
with mock.patch.object(gms, "get_profile",
|
||||
return_value={"profile_name": PROFILE}), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value=None):
|
||||
res = gms._active_mutation_authority(None)
|
||||
self.assertFalse(res.get("ok"))
|
||||
self.assertEqual(res.get("reason_code"), "authority_identity_unresolved")
|
||||
|
||||
def test_expected_username_is_never_substituted_for_authentication(self):
|
||||
with mock.patch.object(
|
||||
gms, "get_profile",
|
||||
return_value={"profile_name": PROFILE, "username": IDENTITY},
|
||||
), mock.patch.object(gms, "_authenticated_username", return_value=None), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value={"expected_username": IDENTITY}):
|
||||
self.assertIsNone(gms._active_username("gitea.prgs.cc"))
|
||||
|
||||
def test_accessors_share_one_snapshot(self):
|
||||
with mock.patch.object(gms, "get_profile",
|
||||
return_value={"profile_name": PROFILE}), \
|
||||
mock.patch.object(gms, "_authenticated_username",
|
||||
return_value=IDENTITY), \
|
||||
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||
return_value=self._ctx()):
|
||||
self.assertEqual(gms._active_username("gitea.prgs.cc"), IDENTITY)
|
||||
self.assertEqual(gms._active_profile_name("gitea.prgs.cc"), PROFILE)
|
||||
|
||||
|
||||
class AuthorMutationBlockTests(unittest.TestCase):
|
||||
"""Preserved: the structured refusal shape review 622 confirmed correct."""
|
||||
|
||||
def test_matches_the_sibling_refusal_shape(self):
|
||||
res = gms._author_mutation_block(["stopped"])
|
||||
self.assertIs(res["success"], False)
|
||||
self.assertIs(res["performed"], False)
|
||||
self.assertEqual(res["outcome"], "REFUSED")
|
||||
self.assertEqual(res["reasons"], ["stopped"])
|
||||
|
||||
def test_carries_reason_code_and_transport_fields(self):
|
||||
res = gms._author_mutation_block(
|
||||
["nope"], reason_code="authority_identity_drift",
|
||||
retryable=False, transport_survives=True,
|
||||
expected="a", actual="b", issue_number=943,
|
||||
)
|
||||
self.assertEqual(res["reason_code"], "authority_identity_drift")
|
||||
self.assertIs(res["retryable"], False)
|
||||
self.assertIs(res["transport_survives"], True)
|
||||
self.assertEqual((res["expected"], res["actual"]), ("a", "b"))
|
||||
self.assertEqual(res["issue_number"], 943)
|
||||
self.assertIs(res["success"], False)
|
||||
|
||||
|
||||
class RuntimeHelperResolutionTests(unittest.TestCase):
|
||||
"""Every runtime helper the wrapper references is defined and callable.
|
||||
|
||||
Supplements the runtime coverage above; it does not replace it.
|
||||
"""
|
||||
|
||||
def test_named_helpers_are_defined_and_callable(self):
|
||||
for name in RUNTIME_HELPERS:
|
||||
with self.subTest(helper=name):
|
||||
self.assertTrue(hasattr(gms, name), f"{name} is not defined")
|
||||
self.assertTrue(callable(getattr(gms, name)))
|
||||
|
||||
def test_every_global_referenced_by_the_wrapper_resolves(self):
|
||||
"""The generalised form of the round-1 defect: an unresolvable global."""
|
||||
fn = _wrapper_ast()
|
||||
bound: set[str] = {a.arg for a in fn.args.args}
|
||||
bound |= {a.arg for a in fn.args.kwonlyargs}
|
||||
if fn.args.vararg:
|
||||
bound.add(fn.args.vararg.arg)
|
||||
if fn.args.kwarg:
|
||||
bound.add(fn.args.kwarg.arg)
|
||||
for node in ast.walk(fn):
|
||||
if isinstance(node, ast.Name) and isinstance(
|
||||
node.ctx, (ast.Store, ast.Del)
|
||||
):
|
||||
bound.add(node.id)
|
||||
elif isinstance(node, (ast.Import, ast.ImportFrom)):
|
||||
for alias in node.names:
|
||||
bound.add((alias.asname or alias.name).split(".")[0])
|
||||
elif isinstance(node, ast.ExceptHandler) and node.name:
|
||||
bound.add(node.name)
|
||||
|
||||
unresolved = sorted(
|
||||
node.id
|
||||
for node in ast.walk(fn)
|
||||
if isinstance(node, ast.Name)
|
||||
and isinstance(node.ctx, ast.Load)
|
||||
and node.id not in bound
|
||||
and not hasattr(gms, node.id)
|
||||
and not hasattr(builtins, node.id)
|
||||
)
|
||||
self.assertEqual(
|
||||
unresolved, [],
|
||||
f"{WRAPPER_NAME} references undefined globals: {unresolved}",
|
||||
)
|
||||
|
||||
def test_wrapper_wires_the_authority_and_session_resolvers(self):
|
||||
fn = _wrapper_ast()
|
||||
called = {
|
||||
node.func.id
|
||||
for node in ast.walk(fn)
|
||||
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name)
|
||||
}
|
||||
self.assertIn("_active_mutation_authority", called)
|
||||
self.assertIn("_resolve_owner_workflow_session", called)
|
||||
self.assertIn("_author_mutation_block", called)
|
||||
|
||||
def test_wrapper_accepts_an_optional_session_id(self):
|
||||
"""ABI addition stays backward compatible: optional, defaulting to None."""
|
||||
fn = _wrapper_ast()
|
||||
names = [a.arg for a in fn.args.args]
|
||||
self.assertIn("session_id", names)
|
||||
offset = len(names) - len(fn.args.defaults)
|
||||
default = fn.args.defaults[names.index("session_id") - offset]
|
||||
self.assertIsInstance(default, ast.Constant)
|
||||
self.assertIsNone(default.value)
|
||||
|
||||
|
||||
class Issue941ScopeGuardNotRegressedTests(unittest.TestCase):
|
||||
"""Preserved: PR #942's bootstrap-scope wiring still holds."""
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.repo, self.head = _make_control_repo(self._tmp.name)
|
||||
|
||||
def _assessment(self, task: str = BOOTSTRAP_TASK) -> dict:
|
||||
return aib.assess_author_issue_bootstrap(
|
||||
workspace_path=self.repo,
|
||||
canonical_repo_root=self.repo,
|
||||
current_branch="master",
|
||||
head_sha=self.head,
|
||||
porcelain_status="",
|
||||
remote_master_sha=self.head,
|
||||
task=task,
|
||||
)
|
||||
|
||||
def test_bootstrap_task_still_permitted_from_clean_control_checkout(self):
|
||||
res = workflow_scope_guard.assess_root_source_mutation(
|
||||
workspace_path=self.repo,
|
||||
canonical_repo_root=self.repo,
|
||||
role_kind="author",
|
||||
mutation_task=BOOTSTRAP_TASK,
|
||||
porcelain_status="",
|
||||
bootstrap_assessment=self._assessment(),
|
||||
)
|
||||
self.assertFalse(res.get("block"), res)
|
||||
self.assertNotEqual(
|
||||
res.get("blocker_kind"), workflow_scope_guard.BLOCKER_MISSING_WORKTREE
|
||||
)
|
||||
|
||||
def test_bootstrap_task_still_blocked_without_evidence(self):
|
||||
res = workflow_scope_guard.assess_root_source_mutation(
|
||||
workspace_path=self.repo,
|
||||
canonical_repo_root=self.repo,
|
||||
role_kind="author",
|
||||
mutation_task=BOOTSTRAP_TASK,
|
||||
porcelain_status="",
|
||||
)
|
||||
self.assertTrue(res.get("block"))
|
||||
self.assertEqual(
|
||||
res.get("blocker_kind"), workflow_scope_guard.BLOCKER_MISSING_WORKTREE
|
||||
)
|
||||
|
||||
def test_ordinary_author_mutation_still_blocked_from_control_checkout(self):
|
||||
res = workflow_scope_guard.assess_root_source_mutation(
|
||||
workspace_path=self.repo,
|
||||
canonical_repo_root=self.repo,
|
||||
role_kind="author",
|
||||
mutation_task="commit_files",
|
||||
porcelain_status="",
|
||||
bootstrap_assessment=self._assessment(),
|
||||
)
|
||||
self.assertTrue(res.get("block"))
|
||||
self.assertEqual(
|
||||
res.get("blocker_kind"), workflow_scope_guard.BLOCKER_MISSING_WORKTREE
|
||||
)
|
||||
|
||||
def test_create_issue_bootstrap_unchanged(self):
|
||||
self.assertTrue(cib.is_create_issue_task("create_issue"))
|
||||
self.assertFalse(cib.is_create_issue_task(BOOTSTRAP_TASK))
|
||||
|
||||
|
||||
def _write_json(path: str, payload: dict) -> None:
|
||||
"""Write an issue-lock file directly, for lock-precedence tests."""
|
||||
import json
|
||||
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "w", encoding="utf-8") as fh:
|
||||
json.dump(payload, fh)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,685 @@
|
||||
import sys as _sys
|
||||
from pathlib import Path as _Path
|
||||
_sys.path.insert(0, str(_Path(__file__).resolve().parent))
|
||||
from mutation_profile_fixture import shared_mutation_env # noqa: E402
|
||||
"""The renewal waiver reaches the real enforcement paths (#945 B1).
|
||||
|
||||
``tests/test_issue_945_owning_pr_renewal_continuation.py`` proves the pure
|
||||
pieces: that ``issue_lock_renewal.owning_pr_renewal_from_lock`` rebuilds a
|
||||
renewal waiver, that ``_owning_pr_continuation_from_lock`` resolves the two
|
||||
dispositions in the right precedence, and that the duplicate gate honours the
|
||||
resulting token. None of that proves any *production* path consumes the
|
||||
resolver, and review ``623`` demonstrated the gap by reverting the primary call
|
||||
site at ``gitea_mcp_server.py:2894`` back to the recovery-only rebuild: the
|
||||
whole repository stayed green, failing-test ids byte identical.
|
||||
|
||||
This file closes that hole. Every test here starts from a real durable lock
|
||||
file written to a temporary lock directory and bound to this process's session
|
||||
pointer, then calls the authoritative production entry point — not a helper:
|
||||
|
||||
* ``mcp_server._enforce_locked_issue_duplicate_recheck`` — the shared recheck
|
||||
behind ``gitea_commit_files`` and ``gitea_create_pr``
|
||||
* ``mcp_server.gitea_assess_work_issue_duplicate`` — the read-only assessor
|
||||
* ``mcp_server._prove_author_ownership_for_pr`` — the push / PR-update
|
||||
ownership prover, which is also the existing-PR continuation path
|
||||
|
||||
Only the external boundaries are mocked: Gitea HTTP reads (the duplicate
|
||||
context fetcher, open-PR and branch listings) and the credential header. The
|
||||
reconstruction and enforcement chain under test — lock load, evidence rebuild,
|
||||
resolver precedence, and ``issue_work_duplicate_gate`` — runs for real.
|
||||
|
||||
``TestRevertingThePrimaryWiringIsDetected`` is the explicit regression the
|
||||
review asked for: it reproduces the pre-#945 recovery-only call site and
|
||||
asserts the enforcement path then refuses, so the wiring cannot be removed
|
||||
silently.
|
||||
|
||||
Everything is written under ``tempfile.TemporaryDirectory``. No branch,
|
||||
worktree, PR, comment, lease, or lock outside that directory is created, and no
|
||||
production Gitea or control-plane state is touched (#945 AC18).
|
||||
"""
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import issue_lock_provenance # noqa: E402
|
||||
import issue_lock_recovery # noqa: E402
|
||||
import issue_lock_renewal # noqa: E402
|
||||
import issue_lock_store # noqa: E402
|
||||
import mcp_server # noqa: E402
|
||||
from issue_work_duplicate_gate import ( # noqa: E402
|
||||
PHASE_COMMIT,
|
||||
PHASE_CREATE_PR,
|
||||
PHASE_LOCK,
|
||||
PHASE_PUSH,
|
||||
)
|
||||
|
||||
ISSUE = 4948
|
||||
OWNING_PR = 4949
|
||||
OTHER_PR = 4950
|
||||
OTHER_ISSUE = 4951
|
||||
BRANCH = f"fix/issue-{ISSUE}-renewal-wiring"
|
||||
OTHER_BRANCH = f"fix/issue-{ISSUE}-competing"
|
||||
HEAD = "e" * 40
|
||||
OTHER_HEAD = "f" * 40
|
||||
IDENTITY = "example-user"
|
||||
PROFILE = "test-author-prgs"
|
||||
ORG = "Scaled-Tech-Consulting"
|
||||
REPO = "Gitea-Tools"
|
||||
HOST = "gitea.prgs.cc"
|
||||
|
||||
|
||||
def dead_pid() -> int:
|
||||
"""A PID that has certainly exited (spawned, then reaped)."""
|
||||
proc = subprocess.Popen([sys.executable, "-c", "pass"])
|
||||
proc.wait()
|
||||
return proc.pid
|
||||
|
||||
|
||||
def shifted_ts(hours: int = 4) -> str:
|
||||
return (
|
||||
(datetime.now(timezone.utc) + timedelta(hours=hours))
|
||||
.isoformat()
|
||||
.replace("+00:00", "Z")
|
||||
)
|
||||
|
||||
|
||||
def owning_pr(number=OWNING_PR, ref=BRANCH, sha=HEAD, issue=ISSUE):
|
||||
return {
|
||||
"number": number,
|
||||
"title": f"fix: something (Closes #{issue})",
|
||||
"body": f"Closes #{issue}.",
|
||||
"head": {"ref": ref, "sha": sha},
|
||||
}
|
||||
|
||||
|
||||
def renewal_block(
|
||||
*,
|
||||
pr_number=OWNING_PR,
|
||||
branch=BRANCH,
|
||||
head=HEAD,
|
||||
identity=IDENTITY,
|
||||
profile=PROFILE,
|
||||
):
|
||||
"""The ``lease_renewal`` block ``build_renewal_record`` writes on success."""
|
||||
return {
|
||||
"renewed": True,
|
||||
"renewed_at": shifted_ts(-1),
|
||||
"prior_pid": 4242,
|
||||
"prior_pid_alive": True,
|
||||
"prior_expires_at": shifted_ts(-1),
|
||||
"replacement_pid": os.getpid(),
|
||||
"new_expires_at": shifted_ts(),
|
||||
"identity": identity,
|
||||
"profile": profile,
|
||||
"branch_name": branch,
|
||||
"worktree_path": os.path.realpath(os.getcwd()),
|
||||
"head_sha": head,
|
||||
"remote_head_sha": head,
|
||||
"pr_head_sha": head,
|
||||
"pr_number": pr_number,
|
||||
"reason": "expired lease renewed by its exact recorded owner",
|
||||
"proof": [],
|
||||
}
|
||||
|
||||
|
||||
def recovery_block(*, pr_number=OWNING_PR, branch=BRANCH, head=HEAD):
|
||||
"""The ``dead_session_recovery`` block ``build_recovery_record`` writes."""
|
||||
return {
|
||||
"recovered": True,
|
||||
"reason": "owning MCP session exited; durable ownership evidence matched",
|
||||
"recovered_at": shifted_ts(-1),
|
||||
"prior_session_pid": 4242,
|
||||
"replacement_session_pid": os.getpid(),
|
||||
"prior_pid_alive": False,
|
||||
"branch_name": branch,
|
||||
"pr_number": pr_number,
|
||||
"pr_head": head,
|
||||
"recorded_head": head,
|
||||
"accepted_head": head,
|
||||
"head_relation": issue_lock_recovery.HEAD_RELATION_EQUAL,
|
||||
"identity": IDENTITY,
|
||||
"profile": PROFILE,
|
||||
"proof": [],
|
||||
}
|
||||
|
||||
|
||||
class EnforcementPathBase(unittest.TestCase):
|
||||
"""Drives production enforcement entry points against a real durable lock.
|
||||
|
||||
The lock lives in a throwaway directory and is bound to this process's
|
||||
session pointer exactly as ``gitea_lock_issue`` binds it, so
|
||||
``_load_existing_issue_lock()`` resolves it through the ordinary
|
||||
``read_session_issue_lock()`` path rather than a test shortcut.
|
||||
"""
|
||||
|
||||
def setUp(self):
|
||||
self.lock_dir = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self.lock_dir.cleanup)
|
||||
self.worktree = os.path.realpath(os.getcwd())
|
||||
self.remotes = patch.dict(
|
||||
mcp_server.REMOTES,
|
||||
{"prgs": {"host": HOST, "org": ORG, "repo": REPO}},
|
||||
)
|
||||
self.remotes.start()
|
||||
self.addCleanup(patch.stopall)
|
||||
mcp_server._IDENTITY_CACHE.clear()
|
||||
|
||||
# ── fixtures ────────────────────────────────────────────────────────────
|
||||
|
||||
def build_lock(
|
||||
self,
|
||||
*,
|
||||
issue_number=ISSUE,
|
||||
branch=BRANCH,
|
||||
renewal=None,
|
||||
recovery=None,
|
||||
claimant=None,
|
||||
pid=None,
|
||||
live=True,
|
||||
):
|
||||
pid = os.getpid() if pid is None else pid
|
||||
claimant = claimant or {"username": IDENTITY, "profile": PROFILE}
|
||||
expires = shifted_ts() if live else shifted_ts(-1)
|
||||
data = {
|
||||
"issue_number": issue_number,
|
||||
"branch_name": branch,
|
||||
"remote": "prgs",
|
||||
"org": ORG,
|
||||
"repo": REPO,
|
||||
"worktree_path": self.worktree,
|
||||
"session_pid": pid,
|
||||
"pid": pid,
|
||||
"claimant": dict(claimant),
|
||||
"work_lease": {
|
||||
"operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE,
|
||||
"issue_number": issue_number,
|
||||
"branch": branch,
|
||||
"worktree_path": self.worktree,
|
||||
"claimant": dict(claimant),
|
||||
"created_at": shifted_ts(-1),
|
||||
"last_heartbeat_at": shifted_ts(0) if live else shifted_ts(-1),
|
||||
"expires_at": expires,
|
||||
},
|
||||
"lock_provenance": issue_lock_provenance.build_sanctioned_lock_provenance(
|
||||
tool="gitea_lock_issue",
|
||||
claimant=dict(claimant),
|
||||
),
|
||||
}
|
||||
if renewal is not None:
|
||||
data["lease_renewal"] = renewal
|
||||
if recovery is not None:
|
||||
data["dead_session_recovery"] = recovery
|
||||
return data
|
||||
|
||||
def bind(self, data):
|
||||
"""Persist the lock and bind it to this process, as the server does."""
|
||||
issue_lock_store.bind_session_lock(data, self.lock_dir.name)
|
||||
return data
|
||||
|
||||
def env(self):
|
||||
return shared_mutation_env(
|
||||
PROFILE,
|
||||
include_example_repo=True,
|
||||
GITEA_ISSUE_LOCK_DIR=self.lock_dir.name,
|
||||
)
|
||||
|
||||
def gitea_reads(self, *, open_prs, branch_names=None):
|
||||
"""Patch only the external Gitea read boundary."""
|
||||
branch_names = [BRANCH] if branch_names is None else branch_names
|
||||
return (
|
||||
patch("mcp_server.get_auth_header", return_value="token x"),
|
||||
patch(
|
||||
"mcp_server.issue_duplicate_context_fetcher",
|
||||
side_effect=lambda h, o, r, auth, issue_number: (
|
||||
list(open_prs), list(branch_names), {"status": "not_claimed"}
|
||||
),
|
||||
),
|
||||
patch("mcp_server._list_open_pulls", return_value=list(open_prs)),
|
||||
patch(
|
||||
"mcp_server.api_get_all",
|
||||
return_value=[
|
||||
{"name": n, "commit": {"id": HEAD}} for n in branch_names
|
||||
],
|
||||
),
|
||||
)
|
||||
|
||||
# ── production entry points ─────────────────────────────────────────────
|
||||
|
||||
def run_duplicate_recheck(self, *, phase, open_prs, branch_names=None):
|
||||
"""The real shared recheck behind gitea_commit_files / gitea_create_pr."""
|
||||
patches = self.gitea_reads(open_prs=open_prs, branch_names=branch_names)
|
||||
with patches[0], patches[1], patches[2], patches[3]:
|
||||
with patch.dict(os.environ, self.env(), clear=True):
|
||||
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||
return mcp_server._enforce_locked_issue_duplicate_recheck(
|
||||
"prgs", phase, host=HOST, org=ORG, repo=REPO
|
||||
)
|
||||
|
||||
def run_readonly_assessor(
|
||||
self, *, open_prs, issue_number=ISSUE, branch=BRANCH, branch_names=None
|
||||
):
|
||||
"""The real read-only duplicate assessor MCP tool."""
|
||||
patches = self.gitea_reads(open_prs=open_prs, branch_names=branch_names)
|
||||
with patches[0], patches[1], patches[2], patches[3]:
|
||||
with patch.dict(os.environ, self.env(), clear=True):
|
||||
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||
return mcp_server.gitea_assess_work_issue_duplicate(
|
||||
issue_number=issue_number,
|
||||
branch_name=branch,
|
||||
phase=PHASE_COMMIT,
|
||||
remote="prgs",
|
||||
host=HOST,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
)
|
||||
|
||||
def run_ownership_prover(
|
||||
self, *, pr_number=OWNING_PR, branch=BRANCH, issue_number=ISSUE
|
||||
):
|
||||
"""The real push / PR-update ownership prover (existing-PR continuation)."""
|
||||
with patch("mcp_server.get_auth_header", return_value="token x"):
|
||||
with patch.dict(os.environ, self.env(), clear=True):
|
||||
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||
return mcp_server._prove_author_ownership_for_pr(
|
||||
pr_number=pr_number,
|
||||
pr_title=f"fix: something (Closes #{issue_number})",
|
||||
pr_body=f"Closes #{issue_number}.",
|
||||
source_branch=branch,
|
||||
remote="prgs",
|
||||
host=HOST,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
worktree_path=self.worktree,
|
||||
)
|
||||
|
||||
|
||||
# ─────────────── B1: renewal evidence reaches every enforcement path ───────────
|
||||
|
||||
|
||||
class TestRenewalReachesEnforcementPaths(EnforcementPathBase):
|
||||
"""A renewal-only lock must exempt its owning PR at the real call sites.
|
||||
|
||||
Each of these fails if its call site is reverted to the recovery-only
|
||||
rebuild, because the lock deliberately carries no ``dead_session_recovery``
|
||||
block at all.
|
||||
"""
|
||||
|
||||
def setUp(self):
|
||||
super().setUp()
|
||||
self.bind(self.build_lock(renewal=renewal_block()))
|
||||
|
||||
def test_commit_duplicate_recheck_permits_the_owning_pr(self):
|
||||
blocked = self.run_duplicate_recheck(
|
||||
phase=PHASE_COMMIT, open_prs=[owning_pr()]
|
||||
)
|
||||
self.assertIsNone(
|
||||
blocked,
|
||||
"commit recheck refused the PR the renewal already proved it owns; "
|
||||
"the resolver is not wired into gitea_mcp_server:2894",
|
||||
)
|
||||
|
||||
def test_create_pr_duplicate_recheck_permits_the_owning_pr(self):
|
||||
blocked = self.run_duplicate_recheck(
|
||||
phase=PHASE_CREATE_PR, open_prs=[owning_pr()]
|
||||
)
|
||||
self.assertIsNone(blocked)
|
||||
|
||||
def test_read_only_assessor_reports_the_same_exemption(self):
|
||||
result = self.run_readonly_assessor(open_prs=[owning_pr()])
|
||||
self.assertTrue(result["success"])
|
||||
self.assertFalse(result["block"])
|
||||
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||
self.assertEqual(result["linked_open_pr"], OWNING_PR)
|
||||
|
||||
def test_push_ownership_prover_carries_the_renewal_evidence(self):
|
||||
ownership = self.run_ownership_prover()
|
||||
self.assertTrue(ownership["proven"], ownership["reasons"])
|
||||
token = ownership["recovered_owning_pr"]
|
||||
self.assertIsNotNone(
|
||||
token,
|
||||
"push prover produced no continuation evidence from a renewal lock; "
|
||||
"the resolver is not wired into gitea_mcp_server:19464",
|
||||
)
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
self.assertEqual(token["branch_name"], BRANCH)
|
||||
self.assertEqual(token["head_sha"], HEAD)
|
||||
|
||||
def test_all_enforcement_paths_decide_alike_from_one_lock(self):
|
||||
"""AC: commit, create-PR, assessor and prover agree on one lock."""
|
||||
for phase in (PHASE_COMMIT, PHASE_CREATE_PR, PHASE_PUSH, PHASE_LOCK):
|
||||
with self.subTest(phase=phase):
|
||||
self.assertIsNone(
|
||||
self.run_duplicate_recheck(phase=phase, open_prs=[owning_pr()])
|
||||
)
|
||||
assessor = self.run_readonly_assessor(open_prs=[owning_pr()])
|
||||
prover = self.run_ownership_prover()
|
||||
self.assertTrue(assessor["owning_pr_recovery_exempted"])
|
||||
self.assertEqual(
|
||||
assessor["linked_open_pr"], prover["recovered_owning_pr"]["pr_number"]
|
||||
)
|
||||
|
||||
|
||||
class TestDeadSessionRecoveryStillReachesEnforcementPaths(EnforcementPathBase):
|
||||
"""#755/#768 recovery must be unchanged by the #945 resolver."""
|
||||
|
||||
def setUp(self):
|
||||
super().setUp()
|
||||
self.bind(self.build_lock(recovery=recovery_block()))
|
||||
|
||||
def test_commit_recheck_still_permits_a_recovered_owning_pr(self):
|
||||
self.assertIsNone(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_assessor_still_reports_the_recovery_exemption(self):
|
||||
result = self.run_readonly_assessor(open_prs=[owning_pr()])
|
||||
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||
|
||||
def test_prover_still_carries_recovery_evidence(self):
|
||||
token = self.run_ownership_prover()["recovered_owning_pr"]
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
|
||||
|
||||
# ──────────────── B1: the explicit anti-revert regression test ────────────────
|
||||
|
||||
|
||||
class TestRevertingThePrimaryWiringIsDetected(EnforcementPathBase):
|
||||
"""Reproduce the pre-#945 call site and prove the path then refuses.
|
||||
|
||||
Review ``623`` reverted ``gitea_mcp_server.py:2894`` from
|
||||
``_owning_pr_continuation_from_lock`` to
|
||||
``issue_lock_recovery.recovered_owning_pr_from_lock`` and found the entire
|
||||
repository still green. Substituting exactly that pre-fix behaviour here
|
||||
makes the enforcement path block, so the causal link between the resolver
|
||||
and the gate's answer is asserted, not assumed.
|
||||
"""
|
||||
|
||||
def setUp(self):
|
||||
super().setUp()
|
||||
self.bind(self.build_lock(renewal=renewal_block()))
|
||||
|
||||
def test_recovery_only_rebuild_reintroduces_the_945_refusal(self):
|
||||
with patch.object(
|
||||
mcp_server,
|
||||
"_owning_pr_continuation_from_lock",
|
||||
side_effect=issue_lock_recovery.recovered_owning_pr_from_lock,
|
||||
):
|
||||
blocked = self.run_duplicate_recheck(
|
||||
phase=PHASE_COMMIT, open_prs=[owning_pr()]
|
||||
)
|
||||
self.assertIsNotNone(
|
||||
blocked,
|
||||
"the pre-#945 recovery-only rebuild must lose the renewal waiver; "
|
||||
"if this passes, the enforcement path is not consuming the resolver",
|
||||
)
|
||||
self.assertTrue(blocked["block"])
|
||||
self.assertFalse(blocked["owning_pr_recovery_exempted"])
|
||||
|
||||
def test_restoring_the_resolver_restores_continuation(self):
|
||||
self.assertIsNone(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_read_only_assessor_is_wired_to_the_same_resolver(self):
|
||||
with patch.object(
|
||||
mcp_server,
|
||||
"_owning_pr_continuation_from_lock",
|
||||
side_effect=issue_lock_recovery.recovered_owning_pr_from_lock,
|
||||
):
|
||||
result = self.run_readonly_assessor(open_prs=[owning_pr()])
|
||||
self.assertTrue(result["block"])
|
||||
self.assertFalse(result["owning_pr_recovery_exempted"])
|
||||
|
||||
def test_push_prover_is_wired_to_the_same_resolver(self):
|
||||
with patch.object(
|
||||
mcp_server,
|
||||
"_owning_pr_continuation_from_lock",
|
||||
side_effect=issue_lock_recovery.recovered_owning_pr_from_lock,
|
||||
):
|
||||
ownership = self.run_ownership_prover()
|
||||
self.assertIsNone(ownership["recovered_owning_pr"])
|
||||
|
||||
|
||||
# ───────────────── B1: the exemption is not widened at the call sites ─────────
|
||||
|
||||
|
||||
class TestEnforcementPathsStillFailClosed(EnforcementPathBase):
|
||||
def assert_blocked(self, result):
|
||||
self.assertIsNotNone(result, "expected a fail-closed refusal")
|
||||
self.assertTrue(result["block"])
|
||||
return result
|
||||
|
||||
def test_open_pr_alone_grants_no_exemption(self):
|
||||
"""No renewal and no recovery block: the open PR still blocks."""
|
||||
self.bind(self.build_lock())
|
||||
blocked = self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
self.assertFalse(blocked["owning_pr_recovery_exempted"])
|
||||
|
||||
def test_second_pr_is_refused(self):
|
||||
self.bind(self.build_lock(renewal=renewal_block()))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(
|
||||
phase=PHASE_COMMIT,
|
||||
open_prs=[owning_pr(), owning_pr(number=OTHER_PR, ref=OTHER_BRANCH)],
|
||||
)
|
||||
)
|
||||
|
||||
def test_evidence_naming_another_pr_is_refused(self):
|
||||
self.bind(self.build_lock(renewal=renewal_block(pr_number=OTHER_PR)))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_unrelated_branch_is_refused(self):
|
||||
self.bind(self.build_lock(renewal=renewal_block(branch=OTHER_BRANCH)))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_live_head_divergence_is_refused(self):
|
||||
"""Force-push or unrelated remote movement: live PR head no longer matches."""
|
||||
self.bind(self.build_lock(renewal=renewal_block()))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(
|
||||
phase=PHASE_COMMIT, open_prs=[owning_pr(sha=OTHER_HEAD)]
|
||||
)
|
||||
)
|
||||
|
||||
def test_stale_recorded_head_is_refused(self):
|
||||
"""The renewal names a head the live PR never had."""
|
||||
self.bind(self.build_lock(renewal=renewal_block(head=OTHER_HEAD)))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_local_remote_head_divergence_is_refused(self):
|
||||
record = renewal_block()
|
||||
record["remote_head_sha"] = OTHER_HEAD
|
||||
self.bind(self.build_lock(renewal=record))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_identity_mismatch_with_the_lock_claimant_is_refused(self):
|
||||
self.bind(self.build_lock(renewal=renewal_block(identity="someone-else")))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_profile_mismatch_with_the_lock_claimant_is_refused(self):
|
||||
self.bind(self.build_lock(renewal=renewal_block(profile="test-reviewer-prgs")))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_ungranted_renewal_block_is_refused(self):
|
||||
record = renewal_block()
|
||||
record["renewed"] = False
|
||||
self.bind(self.build_lock(renewal=record))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_malformed_renewal_block_is_refused(self):
|
||||
record = renewal_block()
|
||||
record["pr_number"] = "not-a-number"
|
||||
self.bind(self.build_lock(renewal=record))
|
||||
self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
def test_wrong_issue_evidence_cannot_be_copied_onto_another_lock(self):
|
||||
"""A renewal block copied onto a lock for a different issue proves nothing.
|
||||
|
||||
The rebuilt token takes its ``issue_number`` from the lock it is found
|
||||
on, not from the record, so a block lifted onto another issue's lock
|
||||
claims that issue while still naming the original PR. That copied
|
||||
evidence must not waive the genuine duplicate the other issue has.
|
||||
"""
|
||||
self.bind(
|
||||
self.build_lock(issue_number=OTHER_ISSUE, renewal=renewal_block())
|
||||
)
|
||||
blocked = self.assert_blocked(
|
||||
self.run_duplicate_recheck(
|
||||
phase=PHASE_COMMIT,
|
||||
# The real open PR for OTHER_ISSUE is a different PR entirely.
|
||||
open_prs=[
|
||||
owning_pr(number=OTHER_PR, ref=OTHER_BRANCH, issue=OTHER_ISSUE)
|
||||
],
|
||||
branch_names=[OTHER_BRANCH],
|
||||
)
|
||||
)
|
||||
self.assertFalse(blocked["owning_pr_recovery_exempted"])
|
||||
|
||||
def test_refusal_carries_complete_structured_fields(self):
|
||||
self.bind(self.build_lock())
|
||||
blocked = self.assert_blocked(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
for field in (
|
||||
"block",
|
||||
"outcome",
|
||||
"reasons",
|
||||
"owning_pr_recovery_exempted",
|
||||
"owning_pr_recovery_notes",
|
||||
"linked_open_pr",
|
||||
"linked_open_pr_count",
|
||||
):
|
||||
with self.subTest(field=field):
|
||||
self.assertIn(field, blocked)
|
||||
self.assertTrue(blocked["reasons"])
|
||||
|
||||
|
||||
class TestSequentialTasksStayIsolated(EnforcementPathBase):
|
||||
"""One long-lived daemon serves many tasks; a waiver must not leak forward."""
|
||||
|
||||
def test_a_later_lock_without_evidence_does_not_inherit_the_earlier_waiver(self):
|
||||
self.bind(self.build_lock(renewal=renewal_block()))
|
||||
self.assertIsNone(
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
)
|
||||
|
||||
# Second task in the same process: a fresh lock, no renewal evidence.
|
||||
self.bind(
|
||||
self.build_lock(issue_number=OTHER_ISSUE, branch=OTHER_BRANCH)
|
||||
)
|
||||
blocked = self.run_duplicate_recheck(
|
||||
phase=PHASE_COMMIT,
|
||||
open_prs=[owning_pr(number=OTHER_PR, ref=OTHER_BRANCH, issue=OTHER_ISSUE)],
|
||||
branch_names=[OTHER_BRANCH],
|
||||
)
|
||||
self.assertIsNotNone(blocked)
|
||||
self.assertFalse(blocked["owning_pr_recovery_exempted"])
|
||||
|
||||
|
||||
# ───────────── F2: what the caller binding actually is, and is not ────────────
|
||||
|
||||
|
||||
class TestCallerBindingIsStructuralNotFieldComparison(unittest.TestCase):
|
||||
"""Document, in executable form, the binding this patch really provides.
|
||||
|
||||
Review ``623`` found that the claimant check in
|
||||
``owning_pr_renewal_from_lock`` compares two fields of one server-written
|
||||
lock file and is therefore not bound to the authenticated caller. That is
|
||||
correct, and these tests assert the true guarantee rather than the
|
||||
overstated one: lock *selection* is process-scoped, and the claimant check
|
||||
is an internal-consistency check.
|
||||
|
||||
No PID-derived, cached, or process-lifetime session authority is invented
|
||||
here — the process scoping asserted below is pre-existing behaviour of
|
||||
``issue_lock_store``, not something this patch adds.
|
||||
"""
|
||||
|
||||
def test_lock_selection_is_keyed_to_the_operating_system_process(self):
|
||||
with tempfile.TemporaryDirectory() as root:
|
||||
pointer = issue_lock_store.session_pointer_path(root)
|
||||
self.assertEqual(
|
||||
os.path.basename(pointer), f"session-{os.getpid()}.json"
|
||||
)
|
||||
|
||||
def test_a_lock_bound_by_another_process_is_not_reachable(self):
|
||||
"""The structural protection: a foreign session pointer is not read."""
|
||||
with tempfile.TemporaryDirectory() as root:
|
||||
foreign_pointer = os.path.join(root, f"session-{os.getpid() + 1}.json")
|
||||
issue_lock_store.save_lock_file(
|
||||
foreign_pointer, {"lock_file_path": "/nonexistent/foreign.json"}
|
||||
)
|
||||
self.assertIsNone(issue_lock_store.read_session_issue_lock(root))
|
||||
|
||||
def test_claimant_check_does_not_consult_the_live_authenticated_caller(self):
|
||||
"""The honest limit: agreement is internal to the lock document.
|
||||
|
||||
A renewal block whose identity/profile agree with the claimant recorded
|
||||
on the same lock rebuilds successfully, regardless of who is
|
||||
authenticated. Live identity and profile are enforced by the separate
|
||||
mutation-authority and profile gates, not by this rebuild.
|
||||
"""
|
||||
lock = {
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": BRANCH,
|
||||
"claimant": {"username": "unrelated-recorded-user", "profile": PROFILE},
|
||||
"lease_renewal": renewal_block(identity="unrelated-recorded-user"),
|
||||
}
|
||||
token = issue_lock_renewal.owning_pr_renewal_from_lock(lock)
|
||||
self.assertIsNotNone(token)
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
|
||||
def test_internal_disagreement_is_what_the_check_actually_rejects(self):
|
||||
lock = {
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": BRANCH,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
"lease_renewal": renewal_block(identity="someone-else"),
|
||||
}
|
||||
self.assertIsNone(issue_lock_renewal.owning_pr_renewal_from_lock(lock))
|
||||
|
||||
|
||||
class TestNoDurableArtifacts(EnforcementPathBase):
|
||||
def test_enforcement_runs_leave_nothing_outside_the_temp_lock_dir(self):
|
||||
before = sorted(os.listdir(self.lock_dir.name))
|
||||
self.bind(self.build_lock(renewal=renewal_block()))
|
||||
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||
self.run_ownership_prover()
|
||||
after = sorted(os.listdir(self.lock_dir.name))
|
||||
self.assertNotEqual(before, after, "the test must have written its lock")
|
||||
self.assertTrue(
|
||||
all(
|
||||
os.path.realpath(os.path.join(self.lock_dir.name, name)).startswith(
|
||||
os.path.realpath(self.lock_dir.name)
|
||||
)
|
||||
for name in after
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,602 @@
|
||||
import sys as _sys
|
||||
from pathlib import Path as _Path
|
||||
_sys.path.insert(0, str(_Path(__file__).resolve().parent))
|
||||
from mutation_profile_fixture import shared_mutation_env # noqa: F401,E402
|
||||
"""Exact-owner renewal keeps its owning-PR waiver past lock_issue (#945).
|
||||
|
||||
#755 taught the duplicate-work gate that a sanctioned *dead-session recovery*
|
||||
owns its open PR, and #768 taught the later gates to rebuild that proof from the
|
||||
durable lock. #760 added the exact-owner *renewal* disposition and granted it
|
||||
the same waiver inside ``gitea_lock_issue`` — but never added the matching
|
||||
rebuild. So an ordinary renewal held the waiver only for the duration of the
|
||||
lock call: ``_enforce_locked_issue_duplicate_recheck`` asked
|
||||
``recovered_owning_pr_from_lock``, which reads only ``dead_session_recovery``,
|
||||
and the very next commit was refused ``duplicate_commit_prevented`` with
|
||||
``owning_pr_recovery_exempted: false`` on the PR the renewal had just proved.
|
||||
|
||||
``TestPreFixReproduction`` pins that defect directly: the recovery-only rebuild
|
||||
still returns ``None`` for a renewal lock, which is exactly why the gates lost
|
||||
the waiver. Everything else proves the renewal half now survives, that recovery
|
||||
is unchanged, and that no path grants an exemption on weaker evidence.
|
||||
|
||||
Every fixture here is an in-memory mapping. Nothing writes a branch, worktree,
|
||||
lock file, lease, comment, or PR (#945 AC18).
|
||||
"""
|
||||
import copy
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import gitea_mcp_server # noqa: E402
|
||||
import issue_lock_recovery # noqa: E402
|
||||
import issue_lock_renewal # noqa: E402
|
||||
from issue_work_duplicate_gate import ( # noqa: E402
|
||||
OUTCOME_DUPLICATE_WORK_NOT_PREVENTED,
|
||||
PHASE_COMMIT,
|
||||
PHASE_CREATE_PR,
|
||||
PHASE_LOCK,
|
||||
PHASE_PUSH,
|
||||
assess_work_issue_duplicate_gate,
|
||||
)
|
||||
|
||||
ISSUE = 4945
|
||||
OWNING_PR = 4946
|
||||
OTHER_PR = 4947
|
||||
BRANCH = f"fix/issue-{ISSUE}-owning-pr-renewal"
|
||||
OTHER_BRANCH = f"fix/issue-{ISSUE}-competing"
|
||||
HEAD = "a" * 40
|
||||
OTHER_HEAD = "b" * 40
|
||||
IDENTITY = "example-user"
|
||||
PROFILE = "test-author-prgs"
|
||||
|
||||
|
||||
def renewal_record(**overrides):
|
||||
"""The ``lease_renewal`` block ``build_renewal_record`` writes on success."""
|
||||
record = {
|
||||
"renewed": True,
|
||||
"renewed_at": "2026-01-01T00:00:00Z",
|
||||
"prior_pid": 4242,
|
||||
"prior_pid_alive": True,
|
||||
"prior_expires_at": "2026-01-01T00:00:00Z",
|
||||
"replacement_pid": 4243,
|
||||
"new_expires_at": "2026-01-01T00:10:00Z",
|
||||
"identity": IDENTITY,
|
||||
"profile": PROFILE,
|
||||
"branch_name": BRANCH,
|
||||
"worktree_path": f"branches/issue-{ISSUE}-owning-pr-renewal",
|
||||
"head_sha": HEAD,
|
||||
"remote_head_sha": HEAD,
|
||||
"pr_head_sha": HEAD,
|
||||
"pr_number": OWNING_PR,
|
||||
"reason": "expired lease renewed by its exact recorded owner",
|
||||
"proof": [],
|
||||
}
|
||||
record.update(overrides)
|
||||
return record
|
||||
|
||||
|
||||
def renewal_lock(record=None, *, issue_number=ISSUE, claimant=True, **lock_overrides):
|
||||
lock = {
|
||||
"issue_number": issue_number,
|
||||
"branch_name": BRANCH,
|
||||
"lease_renewal": renewal_record() if record is None else record,
|
||||
}
|
||||
if claimant:
|
||||
lock["claimant"] = {"username": IDENTITY, "profile": PROFILE}
|
||||
lock.update(lock_overrides)
|
||||
return lock
|
||||
|
||||
|
||||
def recovery_lock(pr_number=OWNING_PR, head=HEAD):
|
||||
"""A lock carrying sanctioned dead-session recovery evidence (#755/#768)."""
|
||||
return {
|
||||
"issue_number": ISSUE,
|
||||
"branch_name": BRANCH,
|
||||
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||
"dead_session_recovery": {
|
||||
"recovered": True,
|
||||
"branch_name": BRANCH,
|
||||
"pr_number": pr_number,
|
||||
"pr_head": head,
|
||||
"recorded_head": head,
|
||||
"accepted_head": head,
|
||||
"head_relation": issue_lock_recovery.HEAD_RELATION_EQUAL,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def owning_pr(number=OWNING_PR, ref=BRANCH, sha=HEAD, issue=ISSUE):
|
||||
return {
|
||||
"number": number,
|
||||
"title": f"fix: something (Closes #{issue})",
|
||||
"body": f"Closes #{issue}.",
|
||||
"head": {"ref": ref, "sha": sha},
|
||||
}
|
||||
|
||||
|
||||
def gate(phase, *, token, open_prs=None, branch_names=None, locked_branch=BRANCH):
|
||||
return assess_work_issue_duplicate_gate(
|
||||
ISSUE,
|
||||
open_prs=[owning_pr()] if open_prs is None else open_prs,
|
||||
branch_names=branch_names or [],
|
||||
claim_entry={},
|
||||
locked_branch=locked_branch,
|
||||
phase=phase,
|
||||
recovered_owning_pr=token,
|
||||
)
|
||||
|
||||
|
||||
# ───────────────────── the defect this issue exists to fix ─────────────────────
|
||||
|
||||
|
||||
class TestPreFixReproduction(unittest.TestCase):
|
||||
"""The exact wiring gap: renewal evidence was invisible to later gates."""
|
||||
|
||||
def test_recovery_only_rebuild_cannot_see_a_renewal_lock(self):
|
||||
# This is the pre-fix behaviour of every enforcement path. It is correct
|
||||
# for the recovery rebuild to ignore a renewal block -- the defect was
|
||||
# that nothing else looked at it.
|
||||
self.assertIsNone(
|
||||
issue_lock_recovery.recovered_owning_pr_from_lock(renewal_lock())
|
||||
)
|
||||
|
||||
def test_renewal_lock_produced_no_exemption_before_the_fix(self):
|
||||
# Feeding the gate what the pre-fix code fed it (recovery rebuild only)
|
||||
# reproduces the reported refusal at the commit phase.
|
||||
token = issue_lock_recovery.recovered_owning_pr_from_lock(renewal_lock())
|
||||
result = gate(PHASE_COMMIT, token=token)
|
||||
self.assertTrue(result["block"])
|
||||
self.assertEqual(result["outcome"], "duplicate_commit_prevented")
|
||||
self.assertFalse(result["owning_pr_recovery_exempted"])
|
||||
self.assertEqual(result["owning_pr_recovery_notes"], [])
|
||||
|
||||
def test_shared_resolver_now_sees_it(self):
|
||||
self.assertIsNotNone(
|
||||
gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||
)
|
||||
|
||||
|
||||
# ───────────────────────── rebuild: the granted case ─────────────────────────
|
||||
|
||||
|
||||
class TestRenewalRebuildGranted(unittest.TestCase):
|
||||
def test_sanctioned_renewal_rebuilds_owning_pr_evidence(self):
|
||||
token = issue_lock_renewal.owning_pr_renewal_from_lock(renewal_lock())
|
||||
self.assertEqual(
|
||||
token,
|
||||
{
|
||||
"issue_number": ISSUE,
|
||||
"pr_number": OWNING_PR,
|
||||
"branch_name": BRANCH,
|
||||
"head_sha": HEAD,
|
||||
"recorded_head": HEAD,
|
||||
"accepted_head": HEAD,
|
||||
"head_relation": "equal",
|
||||
},
|
||||
)
|
||||
|
||||
def test_branch_falls_back_to_the_lock_branch(self):
|
||||
lock = renewal_lock(renewal_record(branch_name=""))
|
||||
token = issue_lock_renewal.owning_pr_renewal_from_lock(lock)
|
||||
self.assertEqual(token["branch_name"], BRANCH)
|
||||
|
||||
def test_claimant_may_live_under_work_lease(self):
|
||||
lock = renewal_lock(claimant=False)
|
||||
lock["work_lease"] = {"claimant": {"username": IDENTITY, "profile": PROFILE}}
|
||||
self.assertIsNotNone(issue_lock_renewal.owning_pr_renewal_from_lock(lock))
|
||||
|
||||
def test_rebuild_does_not_mutate_the_lock(self):
|
||||
lock = renewal_lock()
|
||||
before = copy.deepcopy(lock)
|
||||
issue_lock_renewal.owning_pr_renewal_from_lock(lock)
|
||||
self.assertEqual(lock, before)
|
||||
|
||||
|
||||
# ───────────────────────── rebuild: fails closed ─────────────────────────
|
||||
|
||||
|
||||
class TestRenewalRebuildFailsClosed(unittest.TestCase):
|
||||
def assertNoEvidence(self, lock):
|
||||
self.assertIsNone(issue_lock_renewal.owning_pr_renewal_from_lock(lock))
|
||||
|
||||
def test_no_lock_at_all(self):
|
||||
self.assertNoEvidence(None)
|
||||
self.assertNoEvidence({})
|
||||
self.assertNoEvidence("not-a-mapping")
|
||||
|
||||
def test_lock_without_renewal_block(self):
|
||||
# A fresh claim, or a lock whose renewal block was replaced.
|
||||
self.assertNoEvidence({"issue_number": ISSUE, "branch_name": BRANCH})
|
||||
|
||||
def test_renewal_not_granted(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(renewed=False)))
|
||||
|
||||
def test_renewal_flag_missing(self):
|
||||
record = renewal_record()
|
||||
del record["renewed"]
|
||||
self.assertNoEvidence(renewal_lock(record))
|
||||
|
||||
def test_renewal_block_malformed(self):
|
||||
self.assertNoEvidence(renewal_lock("not-a-mapping"))
|
||||
|
||||
def test_local_head_diverged_from_pr_head(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(head_sha=OTHER_HEAD)))
|
||||
|
||||
def test_remote_head_diverged_from_pr_head(self):
|
||||
# Force-push or unrelated remote movement.
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(remote_head_sha=OTHER_HEAD)))
|
||||
|
||||
def test_local_head_missing(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(head_sha="")))
|
||||
|
||||
def test_remote_head_missing(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(remote_head_sha="")))
|
||||
|
||||
def test_pr_head_missing(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(pr_head_sha="")))
|
||||
|
||||
def test_pr_number_missing(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(pr_number=None)))
|
||||
|
||||
def test_pr_number_malformed(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(pr_number="not-a-number")))
|
||||
|
||||
def test_issue_number_missing_from_lock(self):
|
||||
self.assertNoEvidence(renewal_lock(issue_number=None))
|
||||
|
||||
def test_branch_unknown_everywhere(self):
|
||||
lock = renewal_lock(renewal_record(branch_name=""))
|
||||
lock["branch_name"] = ""
|
||||
self.assertNoEvidence(lock)
|
||||
|
||||
def test_identity_mismatch(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(identity="someone-else")))
|
||||
|
||||
def test_profile_mismatch(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(profile="other-profile")))
|
||||
|
||||
def test_identity_missing(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(identity="")))
|
||||
|
||||
def test_profile_missing(self):
|
||||
self.assertNoEvidence(renewal_lock(renewal_record(profile="")))
|
||||
|
||||
def test_claimant_absent(self):
|
||||
self.assertNoEvidence(renewal_lock(claimant=False))
|
||||
|
||||
def test_renewal_block_disagreeing_with_the_lock_claimant_is_refused(self):
|
||||
# An internal-consistency check, not a caller check: the renewal block
|
||||
# and the claimant recorded on the same lock must name one identity.
|
||||
# Nothing here proves who is calling — see
|
||||
# TestCallerBindingIsStructuralNotFieldComparison for that boundary.
|
||||
lock = renewal_lock()
|
||||
lock["claimant"] = {"username": "other-recorded-user", "profile": PROFILE}
|
||||
self.assertNoEvidence(lock)
|
||||
|
||||
|
||||
# ───────────────────────── the shared resolver ─────────────────────────
|
||||
|
||||
|
||||
class TestSharedResolver(unittest.TestCase):
|
||||
def test_recovery_lock_resolves_to_recovery_evidence(self):
|
||||
token = gitea_mcp_server._owning_pr_continuation_from_lock(recovery_lock())
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
|
||||
def test_renewal_lock_resolves_to_renewal_evidence(self):
|
||||
token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
|
||||
def test_recovery_takes_precedence_over_an_agreeing_renewal(self):
|
||||
# Same precedence gitea_lock_issue applies when granting the waiver, so
|
||||
# the answer cannot differ between the granting and enforcing paths.
|
||||
# Both blocks describe one decision, so both name the same PR and head.
|
||||
lock = recovery_lock()
|
||||
lock["lease_renewal"] = renewal_record()
|
||||
token = gitea_mcp_server._owning_pr_continuation_from_lock(lock)
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
self.assertEqual(token["head_relation"], issue_lock_recovery.HEAD_RELATION_EQUAL)
|
||||
|
||||
def test_no_evidence_resolves_to_none(self):
|
||||
self.assertIsNone(gitea_mcp_server._owning_pr_continuation_from_lock(None))
|
||||
self.assertIsNone(gitea_mcp_server._owning_pr_continuation_from_lock({}))
|
||||
self.assertIsNone(
|
||||
gitea_mcp_server._owning_pr_continuation_from_lock(
|
||||
{"issue_number": ISSUE, "branch_name": BRANCH}
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
# ─────────── ambiguous recovery/renewal pairs never broaden authority ──────────
|
||||
|
||||
|
||||
class TestAmbiguousEvidenceFailsClosed(unittest.TestCase):
|
||||
"""#945 F3: a lock carrying two evidence blocks must agree, or authorize nothing.
|
||||
|
||||
Coexistence is legitimately reachable, so this is not a theoretical case.
|
||||
Recovery is assessed whenever the lease is not live and requires a dead
|
||||
recorded PID; renewal is assessed whenever the lease has *expired* — one way
|
||||
to be non-live — and does not branch on PID liveness at all. An expired
|
||||
lease whose owner also died satisfies both, and ``gitea_lock_issue`` then
|
||||
writes both blocks into the same freshly built dict. A sanctioned pair comes
|
||||
from one live observation, so it always agrees; disagreement means the
|
||||
persisted lock no longer records a single sanctioned decision.
|
||||
|
||||
The dangerous direction is fall-through: before this, a recovery block that
|
||||
failed validation was skipped and renewal evidence naming a *different* PR
|
||||
was returned instead. Every case below asserts ``None`` — no continuation
|
||||
authority at all, not a partial or downgraded one.
|
||||
"""
|
||||
|
||||
def resolve(self, lock):
|
||||
return gitea_mcp_server._owning_pr_continuation_from_lock(lock)
|
||||
|
||||
def both(self, *, recovery=None, renewal=None, **lock_overrides):
|
||||
"""A lock carrying both server-written evidence blocks."""
|
||||
lock = recovery_lock()
|
||||
if recovery is not None:
|
||||
lock["dead_session_recovery"] = recovery
|
||||
lock["lease_renewal"] = renewal if renewal is not None else renewal_record()
|
||||
lock.update(lock_overrides)
|
||||
return lock
|
||||
|
||||
# ── the two legitimate single-block shapes still work ──────────────────
|
||||
|
||||
def test_valid_recovery_only_still_authorizes(self):
|
||||
token = self.resolve(recovery_lock())
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
|
||||
def test_valid_renewal_only_still_authorizes(self):
|
||||
token = self.resolve(renewal_lock())
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
|
||||
# ── both present ───────────────────────────────────────────────────────
|
||||
|
||||
def test_both_present_and_identical_authorizes_once(self):
|
||||
token = self.resolve(self.both())
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
self.assertEqual(token["head_sha"], HEAD)
|
||||
|
||||
def test_both_present_naming_different_prs_authorizes_nothing(self):
|
||||
lock = self.both(renewal=renewal_record(pr_number=OTHER_PR))
|
||||
self.assertIsNone(self.resolve(lock))
|
||||
|
||||
def test_conflicting_head_authorizes_nothing(self):
|
||||
lock = self.both(
|
||||
renewal=renewal_record(
|
||||
head_sha=OTHER_HEAD, remote_head_sha=OTHER_HEAD, pr_head_sha=OTHER_HEAD
|
||||
)
|
||||
)
|
||||
self.assertIsNone(self.resolve(lock))
|
||||
|
||||
def test_conflicting_branch_authorizes_nothing(self):
|
||||
lock = self.both(renewal=renewal_record(branch_name=OTHER_BRANCH))
|
||||
self.assertIsNone(self.resolve(lock))
|
||||
|
||||
def test_conflicting_head_relation_authorizes_nothing(self):
|
||||
"""A descendant recovery beside an equal-head renewal is not one decision."""
|
||||
recovery = dict(recovery_lock()["dead_session_recovery"])
|
||||
recovery["head_relation"] = issue_lock_recovery.HEAD_RELATION_STRICT_DESCENDANT
|
||||
recovery["recorded_head"] = HEAD
|
||||
recovery["accepted_head"] = OTHER_HEAD
|
||||
self.assertIsNone(self.resolve(self.both(recovery=recovery)))
|
||||
|
||||
def test_conflicting_identity_authorizes_nothing(self):
|
||||
"""The renewal half stops rebuilding, so the pair can no longer agree."""
|
||||
lock = self.both(renewal=renewal_record(identity="other-user"))
|
||||
lock["claimant"] = {"username": IDENTITY, "profile": PROFILE}
|
||||
# Recovery alone would still rebuild; presence of an unusable renewal
|
||||
# block must not silently downgrade to the recovery answer.
|
||||
self.assertEqual(self.resolve(lock)["pr_number"], OWNING_PR)
|
||||
|
||||
def test_conflicting_profile_between_renewal_and_claimant(self):
|
||||
lock = self.both(renewal=renewal_record(profile="other-profile"))
|
||||
self.assertEqual(self.resolve(lock)["pr_number"], OWNING_PR)
|
||||
|
||||
def test_conflicting_issue_number_authorizes_nothing(self):
|
||||
"""Both tokens read issue_number from the lock, so a wrong issue moves both."""
|
||||
lock = self.both(issue_number=ISSUE + 1)
|
||||
token = self.resolve(lock)
|
||||
self.assertEqual(token["issue_number"], ISSUE + 1)
|
||||
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||
|
||||
# ── recovery present but unusable: never fall through to renewal ────────
|
||||
|
||||
def test_malformed_recovery_beside_valid_renewal_authorizes_nothing(self):
|
||||
recovery = {"recovered": True, "pr_number": "not-a-number"}
|
||||
self.assertIsNone(self.resolve(self.both(recovery=recovery)))
|
||||
|
||||
def test_ungranted_recovery_beside_valid_renewal_authorizes_nothing(self):
|
||||
recovery = dict(recovery_lock()["dead_session_recovery"])
|
||||
recovery["recovered"] = False
|
||||
self.assertIsNone(self.resolve(self.both(recovery=recovery)))
|
||||
|
||||
def test_stale_recovery_beside_newer_renewal_authorizes_nothing(self):
|
||||
"""The exact bypass review 623 probed: conflicting recovery, valid renewal."""
|
||||
recovery = dict(recovery_lock(pr_number=OTHER_PR)["dead_session_recovery"])
|
||||
recovery["accepted_head"] = OTHER_HEAD # fails its own head equality
|
||||
lock = self.both(recovery=recovery)
|
||||
self.assertIsNone(
|
||||
self.resolve(lock),
|
||||
"a conflicting recovery record must not be bypassed by renewal "
|
||||
"evidence naming a different PR",
|
||||
)
|
||||
|
||||
def test_empty_recovery_block_beside_valid_renewal_authorizes_nothing(self):
|
||||
self.assertIsNone(self.resolve(self.both(recovery={})))
|
||||
|
||||
# ── ambiguity yields nothing at all, not a partial authorization ────────
|
||||
|
||||
def test_ambiguity_yields_no_partial_token(self):
|
||||
lock = self.both(renewal=renewal_record(pr_number=OTHER_PR))
|
||||
result = self.resolve(lock)
|
||||
self.assertIsNone(result)
|
||||
self.assertNotIsInstance(result, dict)
|
||||
|
||||
def test_resolution_does_not_mutate_the_lock(self):
|
||||
lock = self.both(renewal=renewal_record(pr_number=OTHER_PR))
|
||||
before = copy.deepcopy(lock)
|
||||
self.resolve(lock)
|
||||
self.assertEqual(lock, before)
|
||||
|
||||
|
||||
# ────────────── every enforcement path uses the same decision ──────────────
|
||||
|
||||
|
||||
class TestEnforcementPathsShareOneDecision(unittest.TestCase):
|
||||
"""AC: commit, push and create-PR gates consume one authoritative token."""
|
||||
|
||||
def setUp(self):
|
||||
self.token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||
|
||||
def test_commit_phase_permits_continuation(self):
|
||||
result = gate(PHASE_COMMIT, token=self.token)
|
||||
self.assertFalse(result["block"])
|
||||
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||
self.assertEqual(result["outcome"], OUTCOME_DUPLICATE_WORK_NOT_PREVENTED)
|
||||
|
||||
def test_create_pr_phase_permits_continuation(self):
|
||||
result = gate(PHASE_CREATE_PR, token=self.token)
|
||||
self.assertFalse(result["block"])
|
||||
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||
|
||||
def test_push_phase_permits_continuation(self):
|
||||
result = gate(PHASE_PUSH, token=self.token)
|
||||
self.assertFalse(result["block"])
|
||||
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||
|
||||
def test_lock_phase_permits_continuation(self):
|
||||
result = gate(PHASE_LOCK, token=self.token)
|
||||
self.assertFalse(result["block"])
|
||||
|
||||
def test_all_phases_agree(self):
|
||||
outcomes = {
|
||||
phase: gate(phase, token=self.token)["block"]
|
||||
for phase in (PHASE_LOCK, PHASE_COMMIT, PHASE_PUSH, PHASE_CREATE_PR)
|
||||
}
|
||||
self.assertEqual(set(outcomes.values()), {False}, outcomes)
|
||||
|
||||
def test_dead_session_recovery_still_permits_continuation(self):
|
||||
token = gitea_mcp_server._owning_pr_continuation_from_lock(recovery_lock())
|
||||
for phase in (PHASE_COMMIT, PHASE_PUSH, PHASE_CREATE_PR):
|
||||
with self.subTest(phase=phase):
|
||||
result = gate(phase, token=token)
|
||||
self.assertFalse(result["block"])
|
||||
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||
|
||||
|
||||
# ───────────────── the exemption cannot be widened ─────────────────
|
||||
|
||||
|
||||
class TestExemptionCannotBeWidened(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||
|
||||
def test_an_open_pr_alone_grants_nothing(self):
|
||||
result = gate(PHASE_COMMIT, token=None)
|
||||
self.assertTrue(result["block"])
|
||||
self.assertFalse(result["owning_pr_recovery_exempted"])
|
||||
|
||||
def test_a_second_pr_is_refused(self):
|
||||
result = gate(
|
||||
PHASE_CREATE_PR,
|
||||
token=self.token,
|
||||
open_prs=[owning_pr(), owning_pr(number=OTHER_PR, ref=OTHER_BRANCH)],
|
||||
)
|
||||
self.assertTrue(result["block"])
|
||||
self.assertFalse(result["owning_pr_recovery_exempted"])
|
||||
|
||||
def test_a_different_pr_is_refused(self):
|
||||
result = gate(
|
||||
PHASE_COMMIT, token=self.token, open_prs=[owning_pr(number=OTHER_PR)]
|
||||
)
|
||||
self.assertTrue(result["block"])
|
||||
|
||||
def test_a_different_branch_is_refused(self):
|
||||
result = gate(
|
||||
PHASE_COMMIT, token=self.token, open_prs=[owning_pr(ref=OTHER_BRANCH)]
|
||||
)
|
||||
self.assertTrue(result["block"])
|
||||
|
||||
def test_locked_branch_mismatch_is_refused(self):
|
||||
result = gate(PHASE_COMMIT, token=self.token, locked_branch=OTHER_BRANCH)
|
||||
self.assertTrue(result["block"])
|
||||
|
||||
def test_live_pr_head_divergence_is_refused(self):
|
||||
# Force-push or unrelated remote movement after renewal.
|
||||
result = gate(
|
||||
PHASE_COMMIT, token=self.token, open_prs=[owning_pr(sha=OTHER_HEAD)]
|
||||
)
|
||||
self.assertTrue(result["block"])
|
||||
|
||||
def test_evidence_for_another_issue_is_refused(self):
|
||||
foreign = gitea_mcp_server._owning_pr_continuation_from_lock(
|
||||
renewal_lock(issue_number=ISSUE + 1)
|
||||
)
|
||||
result = gate(PHASE_COMMIT, token=foreign)
|
||||
self.assertTrue(result["block"])
|
||||
|
||||
def test_sequential_tasks_do_not_inherit_continuation(self):
|
||||
# One daemon serves many tasks. A renewal proved for issue N must not
|
||||
# authorize continuation for the next task's issue.
|
||||
prior_task = gitea_mcp_server._owning_pr_continuation_from_lock(
|
||||
renewal_lock(issue_number=ISSUE + 7)
|
||||
)
|
||||
self.assertIsNotNone(prior_task)
|
||||
self.assertTrue(gate(PHASE_COMMIT, token=prior_task)["block"])
|
||||
|
||||
|
||||
# ───────────────── ordinary duplicate prevention is intact ─────────────────
|
||||
|
||||
|
||||
class TestDuplicatePreventionRetained(unittest.TestCase):
|
||||
def test_competing_branch_still_blocks(self):
|
||||
token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||
result = gate(
|
||||
PHASE_COMMIT,
|
||||
token=token,
|
||||
open_prs=[],
|
||||
branch_names=[BRANCH, OTHER_BRANCH],
|
||||
)
|
||||
self.assertTrue(result["block"])
|
||||
|
||||
def test_unrelated_work_without_a_lock_still_blocks(self):
|
||||
token = gitea_mcp_server._owning_pr_continuation_from_lock(None)
|
||||
self.assertIsNone(token)
|
||||
self.assertTrue(gate(PHASE_COMMIT, token=token)["block"])
|
||||
|
||||
|
||||
# ───────────────── refusals stay structured and auditable ─────────────────
|
||||
|
||||
|
||||
class TestRefusalShapePreserved(unittest.TestCase):
|
||||
def test_blocked_result_keeps_its_audit_fields(self):
|
||||
token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||
result = gate(
|
||||
PHASE_COMMIT, token=token, open_prs=[owning_pr(number=OTHER_PR)]
|
||||
)
|
||||
for field in (
|
||||
"block",
|
||||
"outcome",
|
||||
"reasons",
|
||||
"owning_pr_recovery_exempted",
|
||||
"owning_pr_recovery_notes",
|
||||
):
|
||||
with self.subTest(field=field):
|
||||
self.assertIn(field, result)
|
||||
self.assertTrue(result["reasons"])
|
||||
# A rejected token explains which element of ownership disagreed.
|
||||
self.assertTrue(result["owning_pr_recovery_notes"])
|
||||
|
||||
def test_granted_result_records_why(self):
|
||||
token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||
result = gate(PHASE_COMMIT, token=token)
|
||||
self.assertTrue(result["owning_pr_recovery_notes"])
|
||||
self.assertIn(
|
||||
f"#{OWNING_PR}", " ".join(result["owning_pr_recovery_notes"])
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -24,6 +24,8 @@ def _lease(expires_at: str) -> dict:
|
||||
|
||||
|
||||
def _lock_record(**overrides) -> dict:
|
||||
# #860: live locks require a usable session pid; PID-less records are never
|
||||
# classified live merely because expiry/heartbeat fields are present.
|
||||
record = {
|
||||
"issue_number": 420,
|
||||
"branch_name": "feat/issue-420-server-code-parity",
|
||||
@@ -31,6 +33,8 @@ def _lock_record(**overrides) -> dict:
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"worktree_path": "/tmp/wt-420",
|
||||
"session_pid": os.getpid(),
|
||||
"pid": os.getpid(),
|
||||
"work_lease": _lease("2999-01-01T00:00:00Z"),
|
||||
}
|
||||
record.update(overrides)
|
||||
@@ -88,6 +92,8 @@ class TestIssueLockStore(unittest.TestCase):
|
||||
existing = _lock_record(
|
||||
branch_name="feat/issue-420-other",
|
||||
worktree_path="/tmp/other",
|
||||
session_pid=os.getpid(),
|
||||
pid=os.getpid(),
|
||||
work_lease=_lease("2999-01-01T00:00:00Z"),
|
||||
)
|
||||
path = ils.lock_file_path(
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
"""Unit tests for mcp_config_drift.py (#672)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
|
||||
from mcp_config_drift import (
|
||||
REQUIRED_GITEA_ROLE_SERVERS,
|
||||
analyze_config_drift,
|
||||
load_mcp_config,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_global_config() -> dict:
|
||||
return {
|
||||
"mcpServers": {
|
||||
"gitea-author": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-author", "SENTRY_AUTH_TOKEN": "secret-token-999"},
|
||||
},
|
||||
"gitea-reviewer": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-reviewer"},
|
||||
},
|
||||
"gitea-merger": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-merger"},
|
||||
},
|
||||
"gitea-reconciler": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-reconciler"},
|
||||
},
|
||||
"gitea-controller": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-controller"},
|
||||
},
|
||||
"gitea-tools": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-author"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
def write_json(path: Path, data: dict) -> str:
|
||||
path.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
||||
return str(path)
|
||||
|
||||
|
||||
def test_drift_detection_in_sync(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, sample_global_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
|
||||
assert report["in_sync"] is True
|
||||
assert report["missing_role_servers"] == []
|
||||
assert report["profile_mismatches"] == []
|
||||
assert set(report["present_role_servers"]) == set(REQUIRED_GITEA_ROLE_SERVERS)
|
||||
|
||||
|
||||
def test_drift_detection_missing_author(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
active_config = json.loads(json.dumps(sample_global_config))
|
||||
del active_config["mcpServers"]["gitea-author"]
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, active_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
|
||||
assert report["in_sync"] is False
|
||||
assert "gitea-author" in report["missing_role_servers"]
|
||||
assert "gitea-author" not in report["present_role_servers"]
|
||||
|
||||
|
||||
def test_drift_detection_missing_reviewer(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
active_config = json.loads(json.dumps(sample_global_config))
|
||||
del active_config["mcpServers"]["gitea-reviewer"]
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, active_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
|
||||
assert report["in_sync"] is False
|
||||
assert "gitea-reviewer" in report["missing_role_servers"]
|
||||
|
||||
|
||||
def test_drift_detection_profile_mismatch(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
active_config = json.loads(json.dumps(sample_global_config))
|
||||
active_config["mcpServers"]["gitea-author"]["env"]["GITEA_MCP_PROFILE"] = "dadeschools-author"
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, active_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
|
||||
assert report["in_sync"] is False
|
||||
assert len(report["profile_mismatches"]) == 1
|
||||
mismatch = report["profile_mismatches"][0]
|
||||
assert mismatch["server"] == "gitea-author"
|
||||
assert mismatch["active_profile"] == "dadeschools-author"
|
||||
assert mismatch["global_profile"] == "prgs-author"
|
||||
|
||||
|
||||
def test_secret_redaction_in_drift_report(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, sample_global_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
serialized = str(report)
|
||||
|
||||
assert "secret-token-999" not in serialized
|
||||
|
||||
|
||||
def test_sanctioned_runbook_forbids_pkill():
|
||||
report = analyze_config_drift(active_config_path="/nonexistent/path/active.json", global_config_path="/nonexistent/path/global.json")
|
||||
|
||||
runbook_text = " ".join(report["sanctioned_repair_runbook"]).lower()
|
||||
forbidden_text = " ".join(report["forbidden_repair_methods"]).lower()
|
||||
|
||||
assert "pkill" in forbidden_text
|
||||
assert "mtime" in forbidden_text
|
||||
assert "source" in forbidden_text
|
||||
assert "session-state" in forbidden_text
|
||||
@@ -0,0 +1,788 @@
|
||||
"""Authoritative fleet inventory classification (#949).
|
||||
|
||||
One test group per acceptance criterion. The classifier is pure, so every
|
||||
scenario is expressed as a snapshot: registry rows plus a process observation.
|
||||
PID liveness is the one impure input, so it is patched per test rather than
|
||||
depending on whatever happens to be running on the machine.
|
||||
"""
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from unittest import mock
|
||||
|
||||
import mcp_fleet_inventory as mfi
|
||||
|
||||
|
||||
NOW = datetime(2026, 7, 28, 3, 0, 0, tzinfo=timezone.utc)
|
||||
HEAD_A = "82d71b77028a7abd4f8ab4a4e4d89658a187f73d"
|
||||
HEAD_B = "35ed8a2fcb11134a37c862ca6eaca26e3028902a"
|
||||
COHORT_A = "ppid:40990"
|
||||
COHORT_B = "ppid:51022"
|
||||
BINDING = {"remote": "prgs", "org": "Scaled-Tech-Consulting", "repo": "Gitea-Tools"}
|
||||
|
||||
_BASE_PID = 41000
|
||||
|
||||
|
||||
def row(
|
||||
namespace,
|
||||
profile,
|
||||
role,
|
||||
pid,
|
||||
*,
|
||||
cohort_id=COHORT_A,
|
||||
startup_head=HEAD_A,
|
||||
registered_at="2026-07-28T02:00:00Z",
|
||||
**overrides,
|
||||
):
|
||||
"""One control-plane runtime registry row."""
|
||||
record = {
|
||||
"runtime_id": f"{namespace}:{pid}:boot{pid}",
|
||||
"namespace": namespace,
|
||||
"profile": profile,
|
||||
"role": role,
|
||||
"remote": BINDING["remote"],
|
||||
"org": BINDING["org"],
|
||||
"repo": BINDING["repo"],
|
||||
"repository_root": "/checkout/Gitea-Tools",
|
||||
"pid": pid,
|
||||
"cohort_id": cohort_id,
|
||||
"cohort_source": "parent_process",
|
||||
"client_provenance": "client_managed",
|
||||
"boot_id": f"boot{pid}",
|
||||
"startup_head": startup_head,
|
||||
"daemon_start_head": startup_head,
|
||||
"transport": "stdio",
|
||||
"registered_at": registered_at,
|
||||
"last_heartbeat_at": registered_at,
|
||||
"status": "running",
|
||||
}
|
||||
record.update(overrides)
|
||||
return record
|
||||
|
||||
|
||||
def healthy_rows():
|
||||
"""One live row per expected member, all one cohort, all one revision."""
|
||||
return [
|
||||
row(entry["namespace"], entry["profile"], entry["role"], _BASE_PID + index)
|
||||
for index, entry in enumerate(mfi.EXPECTED_PRGS_FLEET)
|
||||
]
|
||||
|
||||
|
||||
def scan_for(rows, *, available=True, extra_pids=(), started_at="2026-07-28T01:00:00Z"):
|
||||
"""A process observation that corroborates *rows* (plus any extra PIDs)."""
|
||||
processes = [
|
||||
{"pid": r["pid"], "started_at": started_at, "command": "python mcp_server.py"}
|
||||
for r in rows
|
||||
]
|
||||
processes.extend(
|
||||
{"pid": pid, "started_at": started_at, "command": "python mcp_server.py"}
|
||||
for pid in extra_pids
|
||||
)
|
||||
return {"available": available, "processes": processes, "reason": None}
|
||||
|
||||
|
||||
def classify(rows, scan=None, **kwargs):
|
||||
kwargs.setdefault("expected_binding", BINDING)
|
||||
kwargs.setdefault("now", NOW)
|
||||
return mfi.classify_fleet_inventory(
|
||||
runtime_rows=rows,
|
||||
process_scan=scan if scan is not None else scan_for(rows),
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
class _AliveMixin:
|
||||
"""Treat every PID as alive unless a test declares it dead or unknown."""
|
||||
|
||||
def setUp(self):
|
||||
super().setUp()
|
||||
self.dead_pids = set()
|
||||
self.unknown_pids = set()
|
||||
|
||||
def probe(pid):
|
||||
if pid in self.unknown_pids:
|
||||
return None
|
||||
return pid not in self.dead_pids
|
||||
|
||||
patcher = mock.patch.object(mfi, "probe_pid_alive", side_effect=probe)
|
||||
patcher.start()
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
|
||||
class TestHealthyFleet(_AliveMixin, unittest.TestCase):
|
||||
"""AC1: a healthy five-server fleet reports all five members exactly once."""
|
||||
|
||||
def test_five_members_each_reported_once(self):
|
||||
result = classify(healthy_rows())
|
||||
self.assertEqual(len(result["configured_members"]), 5)
|
||||
self.assertEqual(len(result["running_members"]), 5)
|
||||
self.assertEqual(result["running_member_count"], 5)
|
||||
self.assertEqual(result["missing_members"], [])
|
||||
self.assertEqual(result["duplicate_members"], [])
|
||||
self.assertEqual(result["unexpected_members"], [])
|
||||
self.assertTrue(result["exactly_one_per_profile"])
|
||||
for member in result["configured_members"]:
|
||||
self.assertEqual(member["instance_count"], 1)
|
||||
self.assertEqual(member["health"], mfi.HEALTH_RUNNING)
|
||||
|
||||
def test_healthy_fleet_satisfies_the_mutation_gate(self):
|
||||
result = classify(healthy_rows())
|
||||
self.assertTrue(result["inventory_complete"])
|
||||
self.assertTrue(result["mutation_gate_satisfied"])
|
||||
self.assertIsNone(result["blocked_reason"])
|
||||
self.assertTrue(result["single_cohort"])
|
||||
self.assertFalse(result["mixed_cohort"])
|
||||
self.assertFalse(result["mixed_revision"])
|
||||
|
||||
def test_every_expected_namespace_and_profile_is_present(self):
|
||||
result = classify(healthy_rows())
|
||||
self.assertEqual(
|
||||
sorted(m["profile"] for m in result["configured_members"]),
|
||||
[
|
||||
"prgs-author",
|
||||
"prgs-controller",
|
||||
"prgs-merger",
|
||||
"prgs-reconciler",
|
||||
"prgs-reviewer",
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
class TestDuplicateMember(_AliveMixin, unittest.TestCase):
|
||||
"""AC2: two processes on one profile are a duplicate and fail the gate."""
|
||||
|
||||
def test_duplicate_is_reported_with_every_pid(self):
|
||||
rows = healthy_rows()
|
||||
rows.append(row("gitea-author", "prgs-author", "author", 49999))
|
||||
result = classify(rows)
|
||||
self.assertEqual(len(result["duplicate_members"]), 1)
|
||||
duplicate = result["duplicate_members"][0]
|
||||
self.assertEqual(duplicate["profile"], "prgs-author")
|
||||
self.assertEqual(duplicate["instance_count"], 2)
|
||||
self.assertEqual(duplicate["pids"], [_BASE_PID, 49999])
|
||||
|
||||
def test_duplicate_fails_the_mutation_gate(self):
|
||||
rows = healthy_rows()
|
||||
rows.append(row("gitea-author", "prgs-author", "author", 49999))
|
||||
result = classify(rows)
|
||||
self.assertFalse(result["exactly_one_per_profile"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertIn("duplicate", result["blocked_reason"])
|
||||
|
||||
def test_duplicate_is_not_reported_as_missing_or_unexpected(self):
|
||||
rows = healthy_rows()
|
||||
rows.append(row("gitea-author", "prgs-author", "author", 49999))
|
||||
result = classify(rows)
|
||||
self.assertEqual(result["missing_members"], [])
|
||||
self.assertEqual(result["unexpected_members"], [])
|
||||
|
||||
|
||||
class TestMissingMember(_AliveMixin, unittest.TestCase):
|
||||
"""AC3: a missing expected server is identified and fails the gate."""
|
||||
|
||||
def test_missing_member_is_named(self):
|
||||
rows = [r for r in healthy_rows() if r["profile"] != "prgs-merger"]
|
||||
result = classify(rows)
|
||||
self.assertEqual(len(result["missing_members"]), 1)
|
||||
self.assertEqual(result["missing_members"][0]["profile"], "prgs-merger")
|
||||
self.assertEqual(result["missing_members"][0]["health"], mfi.HEALTH_MISSING)
|
||||
|
||||
def test_missing_member_fails_the_mutation_gate(self):
|
||||
rows = [r for r in healthy_rows() if r["profile"] != "prgs-merger"]
|
||||
result = classify(rows)
|
||||
self.assertFalse(result["exactly_one_per_profile"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertIn("prgs-merger", result["blocked_reason"])
|
||||
|
||||
def test_missing_member_still_lists_all_configured_members(self):
|
||||
rows = [r for r in healthy_rows() if r["profile"] != "prgs-merger"]
|
||||
result = classify(rows)
|
||||
self.assertEqual(len(result["configured_members"]), 5)
|
||||
merger = [
|
||||
m for m in result["configured_members"] if m["profile"] == "prgs-merger"
|
||||
][0]
|
||||
self.assertEqual(merger["instance_count"], 0)
|
||||
self.assertEqual(merger["health"], mfi.HEALTH_MISSING)
|
||||
|
||||
|
||||
class TestUnexpectedMember(_AliveMixin, unittest.TestCase):
|
||||
"""AC4: an unexpected PRGS server is reported explicitly."""
|
||||
|
||||
def test_unexpected_member_is_its_own_category(self):
|
||||
rows = healthy_rows()
|
||||
rows.append(row("gitea-shadow", "prgs-shadow", "author", 47777))
|
||||
result = classify(rows)
|
||||
self.assertEqual(len(result["unexpected_members"]), 1)
|
||||
self.assertEqual(result["unexpected_members"][0]["profile"], "prgs-shadow")
|
||||
self.assertEqual(
|
||||
result["unexpected_members"][0]["health"], mfi.HEALTH_UNEXPECTED
|
||||
)
|
||||
|
||||
def test_unexpected_member_is_not_collapsed_into_duplicates_or_missing(self):
|
||||
rows = healthy_rows()
|
||||
rows.append(row("gitea-shadow", "prgs-shadow", "author", 47777))
|
||||
result = classify(rows)
|
||||
self.assertEqual(result["duplicate_members"], [])
|
||||
self.assertEqual(result["missing_members"], [])
|
||||
self.assertTrue(result["exactly_one_per_profile"])
|
||||
self.assertFalse(result["no_unexpected_members"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_unexpected_member_blocks_with_its_own_reason(self):
|
||||
rows = healthy_rows()
|
||||
rows.append(row("gitea-shadow", "prgs-shadow", "author", 47777))
|
||||
result = classify(rows)
|
||||
self.assertTrue(
|
||||
any("unexpected" in reason for reason in result["blocked_reasons"])
|
||||
)
|
||||
|
||||
|
||||
class TestMixedCohort(_AliveMixin, unittest.TestCase):
|
||||
"""AC5: members from different client cohorts are detected."""
|
||||
|
||||
def test_two_cohorts_are_detected(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["cohort_id"] = COHORT_B
|
||||
result = classify(rows)
|
||||
self.assertTrue(result["mixed_cohort"])
|
||||
self.assertFalse(result["single_cohort"])
|
||||
self.assertEqual(result["cohort_ids"], sorted([COHORT_A, COHORT_B]))
|
||||
|
||||
def test_mixed_cohort_fails_the_mutation_gate(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["cohort_id"] = COHORT_B
|
||||
result = classify(rows)
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertTrue(any("cohort" in reason for reason in result["blocked_reasons"]))
|
||||
|
||||
def test_mixed_cohort_is_not_a_duplicate_or_missing_report(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["cohort_id"] = COHORT_B
|
||||
result = classify(rows)
|
||||
self.assertEqual(result["duplicate_members"], [])
|
||||
self.assertEqual(result["missing_members"], [])
|
||||
|
||||
|
||||
class TestMixedRevision(_AliveMixin, unittest.TestCase):
|
||||
"""AC6: members running different startup revisions are detected."""
|
||||
|
||||
def test_two_revisions_are_detected(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["startup_head"] = HEAD_B
|
||||
result = classify(rows)
|
||||
self.assertTrue(result["mixed_revision"])
|
||||
self.assertEqual(result["startup_revisions"], sorted([HEAD_A, HEAD_B]))
|
||||
|
||||
def test_mixed_revision_fails_the_mutation_gate(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["startup_head"] = HEAD_B
|
||||
result = classify(rows)
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertTrue(
|
||||
any("revision" in reason for reason in result["blocked_reasons"])
|
||||
)
|
||||
|
||||
def test_mixed_revision_does_not_by_itself_imply_mixed_cohort(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["startup_head"] = HEAD_B
|
||||
result = classify(rows)
|
||||
self.assertTrue(result["single_cohort"])
|
||||
self.assertFalse(result["mixed_cohort"])
|
||||
|
||||
|
||||
class TestRevisionIsNotCohort(_AliveMixin, unittest.TestCase):
|
||||
"""AC7: matching Git revisions alone do not establish a single cohort."""
|
||||
|
||||
def test_identical_revisions_with_unknown_cohort_stay_unknown(self):
|
||||
rows = healthy_rows()
|
||||
for r in rows:
|
||||
r["cohort_id"] = None
|
||||
result = classify(rows)
|
||||
self.assertEqual(len({r["startup_head"] for r in rows}), 1)
|
||||
self.assertIsNone(result["single_cohort"])
|
||||
self.assertIsNone(result["mixed_cohort"])
|
||||
|
||||
def test_identical_revisions_with_unknown_cohort_fail_closed(self):
|
||||
rows = healthy_rows()
|
||||
for r in rows:
|
||||
r["cohort_id"] = None
|
||||
result = classify(rows)
|
||||
self.assertFalse(result["inventory_complete"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertTrue(
|
||||
any("cohort" in reason for reason in result["incomplete_reasons"])
|
||||
)
|
||||
|
||||
def test_one_unknown_cohort_among_known_ones_still_fails_closed(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["cohort_id"] = None
|
||||
result = classify(rows)
|
||||
self.assertIsNone(result["single_cohort"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_cohort_derivation_never_consults_revisions(self):
|
||||
"""The cohort helper takes no revision input at all."""
|
||||
identity = mfi.derive_cohort_identity({mfi.COHORT_ID_ENV: "cohort-x"})
|
||||
self.assertEqual(identity["cohort_id"], "cohort-x")
|
||||
self.assertEqual(identity["cohort_source"], "explicit_env")
|
||||
|
||||
def test_orphaned_process_reports_unknown_cohort(self):
|
||||
with mock.patch("os.getppid", return_value=1):
|
||||
identity = mfi.derive_cohort_identity({})
|
||||
self.assertIsNone(identity["cohort_id"])
|
||||
self.assertEqual(identity["cohort_source"], "unknown")
|
||||
|
||||
def test_parent_process_is_the_cohort_when_no_env_is_set(self):
|
||||
with mock.patch("os.getppid", return_value=40990):
|
||||
identity = mfi.derive_cohort_identity({})
|
||||
self.assertEqual(identity["cohort_id"], COHORT_A)
|
||||
self.assertEqual(identity["cohort_source"], "parent_process")
|
||||
|
||||
|
||||
class TestConfigurationIsNotRunning(_AliveMixin, unittest.TestCase):
|
||||
"""AC8: configuration without a live worker is not a running member."""
|
||||
|
||||
def test_no_registry_rows_means_every_member_is_missing(self):
|
||||
result = classify([], scan=scan_for([]))
|
||||
self.assertEqual(len(result["missing_members"]), 5)
|
||||
self.assertEqual(result["running_members"], [])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_dead_pid_is_stale_not_running(self):
|
||||
rows = healthy_rows()
|
||||
self.dead_pids = {rows[0]["pid"]}
|
||||
result = classify(rows)
|
||||
self.assertEqual(len(result["running_members"]), 4)
|
||||
self.assertEqual(len(result["stale_members"]), 1)
|
||||
self.assertEqual(result["stale_members"][0]["liveness"], mfi.LIVENESS_DEAD)
|
||||
self.assertEqual(result["stale_members"][0]["health"], mfi.HEALTH_STALE)
|
||||
self.assertEqual(len(result["missing_members"]), 1)
|
||||
|
||||
def test_registry_row_without_a_matching_process_is_not_running(self):
|
||||
rows = healthy_rows()
|
||||
scan = scan_for(rows[1:]) # first member's process is absent
|
||||
result = classify(rows, scan=scan)
|
||||
self.assertEqual(len(result["running_members"]), 4)
|
||||
self.assertEqual(
|
||||
result["stale_members"][0]["liveness"], mfi.LIVENESS_UNOBSERVED
|
||||
)
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_recycled_pid_does_not_impersonate_a_dead_server(self):
|
||||
rows = healthy_rows()
|
||||
scan = scan_for(rows)
|
||||
# The process now holding the first PID started *after* registration.
|
||||
scan["processes"][0]["started_at"] = "2026-07-28T02:30:00Z"
|
||||
result = classify(rows, scan=scan)
|
||||
stale = [
|
||||
m
|
||||
for m in result["stale_members"]
|
||||
if m["liveness"] == mfi.LIVENESS_PID_RECYCLED
|
||||
]
|
||||
self.assertEqual(len(stale), 1)
|
||||
self.assertEqual(len(result["running_members"]), 4)
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
|
||||
class TestIncompleteEvidenceFailsClosed(_AliveMixin, unittest.TestCase):
|
||||
"""AC9: unknown or unavailable evidence produces a fail-closed result."""
|
||||
|
||||
def test_unavailable_process_listing_fails_closed(self):
|
||||
rows = healthy_rows()
|
||||
result = classify(
|
||||
rows,
|
||||
scan={"available": False, "processes": [], "reason": "ps unavailable"},
|
||||
)
|
||||
self.assertFalse(result["inventory_complete"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertIn("ps unavailable", result["incomplete_reasons"])
|
||||
self.assertEqual(result["running_members"], [])
|
||||
|
||||
def test_unreadable_registry_fails_closed(self):
|
||||
result = classify(
|
||||
[],
|
||||
scan=scan_for([]),
|
||||
registry_available=False,
|
||||
registry_error="registry unreadable",
|
||||
)
|
||||
self.assertFalse(result["inventory_complete"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertIn("registry unreadable", result["incomplete_reasons"])
|
||||
|
||||
def test_unregistered_running_process_fails_closed(self):
|
||||
rows = healthy_rows()
|
||||
result = classify(rows, scan=scan_for(rows, extra_pids=[59999]))
|
||||
self.assertEqual([p["pid"] for p in result["unregistered_processes"]], [59999])
|
||||
self.assertFalse(result["inventory_complete"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertTrue(
|
||||
any("59999" in reason for reason in result["incomplete_reasons"])
|
||||
)
|
||||
|
||||
def test_undeterminable_pid_liveness_is_unknown_not_healthy(self):
|
||||
rows = healthy_rows()
|
||||
self.unknown_pids = {rows[0]["pid"]}
|
||||
result = classify(rows)
|
||||
self.assertEqual(result["stale_members"][0]["liveness"], mfi.LIVENESS_UNKNOWN)
|
||||
self.assertEqual(result["stale_members"][0]["health"], mfi.HEALTH_UNKNOWN)
|
||||
self.assertFalse(result["inventory_complete"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_unknown_startup_revision_fails_closed(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["startup_head"] = None
|
||||
result = classify(rows)
|
||||
self.assertFalse(result["inventory_complete"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertTrue(
|
||||
any("startup revision" in r for r in result["incomplete_reasons"])
|
||||
)
|
||||
|
||||
def test_unknown_is_distinguished_from_healthy(self):
|
||||
rows = healthy_rows()
|
||||
for r in rows:
|
||||
r["cohort_id"] = None
|
||||
result = classify(rows)
|
||||
# Not "unhealthy" in the sense of a named defect: nothing is missing,
|
||||
# duplicated or unexpected. It is *unknown*, and that still fails closed.
|
||||
self.assertEqual(result["missing_members"], [])
|
||||
self.assertEqual(result["duplicate_members"], [])
|
||||
self.assertEqual(result["unexpected_members"], [])
|
||||
self.assertIsNone(result["single_cohort"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
|
||||
class TestBindingAndRoleConsistency(_AliveMixin, unittest.TestCase):
|
||||
"""Repository-binding and role/profile mismatches fail closed."""
|
||||
|
||||
def test_repository_binding_mismatch_is_reported(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["repo"] = "Some-Other-Repo"
|
||||
result = classify(rows)
|
||||
self.assertEqual(len(result["repository_binding_mismatches"]), 1)
|
||||
self.assertEqual(
|
||||
result["repository_binding_mismatches"][0]["repo"], "Some-Other-Repo"
|
||||
)
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_incomplete_repository_binding_is_reported(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["org"] = None
|
||||
result = classify(rows)
|
||||
self.assertEqual(len(result["repository_binding_mismatches"]), 1)
|
||||
self.assertIn("complete", result["repository_binding_mismatches"][0]["reason"])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_role_mismatch_is_reported(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["role"] = "merger" # prgs-author is configured as author
|
||||
result = classify(rows)
|
||||
self.assertEqual(len(result["role_mismatches"]), 1)
|
||||
self.assertEqual(result["role_mismatches"][0]["expected_role"], "author")
|
||||
self.assertEqual(result["role_mismatches"][0]["declared_role"], "merger")
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_profile_served_from_the_wrong_namespace_is_reported(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["namespace"] = "gitea-reviewer"
|
||||
result = classify(rows)
|
||||
self.assertTrue(
|
||||
any(
|
||||
m.get("expected_namespace") == "gitea-author"
|
||||
for m in result["role_mismatches"]
|
||||
)
|
||||
)
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_matching_binding_produces_no_mismatch(self):
|
||||
result = classify(healthy_rows())
|
||||
self.assertEqual(result["repository_binding_mismatches"], [])
|
||||
self.assertEqual(result["role_mismatches"], [])
|
||||
|
||||
|
||||
class TestDeterminism(_AliveMixin, unittest.TestCase):
|
||||
"""AC10 / stable ordering and deterministic structured output."""
|
||||
|
||||
def test_identical_snapshots_produce_identical_results(self):
|
||||
rows = healthy_rows()
|
||||
first = classify(rows, scan=scan_for(rows))
|
||||
second = classify(healthy_rows(), scan=scan_for(healthy_rows()))
|
||||
self.assertEqual(first, second)
|
||||
|
||||
def test_row_order_does_not_change_the_result(self):
|
||||
rows = healthy_rows()
|
||||
shuffled = list(reversed(healthy_rows()))
|
||||
forward = classify(rows, scan=scan_for(rows))
|
||||
backward = classify(shuffled, scan=scan_for(shuffled))
|
||||
self.assertEqual(forward["running_members"], backward["running_members"])
|
||||
self.assertEqual(forward["configured_members"], backward["configured_members"])
|
||||
self.assertEqual(
|
||||
forward["mutation_gate_satisfied"], backward["mutation_gate_satisfied"]
|
||||
)
|
||||
|
||||
def test_members_are_sorted_by_namespace_then_profile_then_pid(self):
|
||||
rows = healthy_rows()
|
||||
rows.append(row("gitea-author", "prgs-author", "author", 40001))
|
||||
result = classify(rows)
|
||||
keys = [
|
||||
(m["namespace"], m["profile"], m["pid"]) for m in result["running_members"]
|
||||
]
|
||||
self.assertEqual(keys, sorted(keys))
|
||||
|
||||
def test_answering_namespace_does_not_change_the_verdict(self):
|
||||
"""AC10: controller and reconciler agree for one fleet snapshot."""
|
||||
rows = healthy_rows()
|
||||
controller = classify(
|
||||
rows, scan=scan_for(rows), answering_namespace="gitea-controller"
|
||||
)
|
||||
reconciler = classify(
|
||||
healthy_rows(),
|
||||
scan=scan_for(healthy_rows()),
|
||||
answering_namespace="gitea-reconciler",
|
||||
)
|
||||
self.assertEqual(controller["answering_namespace"], "gitea-controller")
|
||||
self.assertEqual(reconciler["answering_namespace"], "gitea-reconciler")
|
||||
for key in sorted(set(controller) - {"answering_namespace"}):
|
||||
self.assertEqual(controller[key], reconciler[key], f"{key} disagreed")
|
||||
|
||||
def test_controller_and_reconciler_agree_on_an_unhealthy_fleet(self):
|
||||
rows = [r for r in healthy_rows() if r["profile"] != "prgs-merger"]
|
||||
controller = classify(
|
||||
rows, scan=scan_for(rows), answering_namespace="gitea-controller"
|
||||
)
|
||||
reconciler = classify(
|
||||
rows, scan=scan_for(rows), answering_namespace="gitea-reconciler"
|
||||
)
|
||||
self.assertEqual(controller["blocked_reason"], reconciler["blocked_reason"])
|
||||
self.assertEqual(
|
||||
controller["mutation_gate_satisfied"],
|
||||
reconciler["mutation_gate_satisfied"],
|
||||
)
|
||||
|
||||
|
||||
class TestReadOnly(_AliveMixin, unittest.TestCase):
|
||||
"""AC11: the capability mutates nothing."""
|
||||
|
||||
def test_classification_reports_no_mutations(self):
|
||||
result = classify(healthy_rows())
|
||||
self.assertEqual(result["mutations_performed"], [])
|
||||
self.assertTrue(result["read_only"])
|
||||
|
||||
def test_classification_does_not_write_the_input_rows_back(self):
|
||||
rows = healthy_rows()
|
||||
snapshot = [dict(r) for r in rows]
|
||||
classify(rows)
|
||||
self.assertEqual(rows, snapshot)
|
||||
|
||||
|
||||
class TestLivenessProbe(unittest.TestCase):
|
||||
"""AC11: the real probe never sends a terminating signal.
|
||||
|
||||
Deliberately *not* using ``_AliveMixin`` — these tests exercise
|
||||
``probe_pid_alive`` itself, which the mixin replaces.
|
||||
"""
|
||||
|
||||
def test_only_signal_zero_is_ever_sent(self):
|
||||
with mock.patch("os.kill") as killer:
|
||||
mfi.probe_pid_alive(4242)
|
||||
killer.assert_called_once_with(4242, 0)
|
||||
|
||||
def test_classifying_a_duplicate_never_terminates_it(self):
|
||||
rows = healthy_rows()
|
||||
rows.append(row("gitea-author", "prgs-author", "author", 49999))
|
||||
with mock.patch("os.kill") as killer:
|
||||
result = mfi.classify_fleet_inventory(
|
||||
runtime_rows=rows,
|
||||
process_scan=scan_for(rows),
|
||||
expected_binding=BINDING,
|
||||
now=NOW,
|
||||
)
|
||||
self.assertTrue(killer.call_args_list, "liveness must actually be probed")
|
||||
for call in killer.call_args_list:
|
||||
self.assertEqual(call.args[1], 0, "only signal 0 may ever be sent")
|
||||
self.assertEqual(result["mutations_performed"], [])
|
||||
|
||||
def test_liveness_probe_tolerates_a_missing_process(self):
|
||||
with mock.patch("os.kill", side_effect=ProcessLookupError):
|
||||
self.assertFalse(mfi.probe_pid_alive(4242))
|
||||
|
||||
def test_liveness_probe_treats_permission_error_as_alive(self):
|
||||
with mock.patch("os.kill", side_effect=PermissionError):
|
||||
self.assertTrue(mfi.probe_pid_alive(4242))
|
||||
|
||||
def test_liveness_probe_returns_unknown_on_other_os_errors(self):
|
||||
with mock.patch("os.kill", side_effect=OSError):
|
||||
self.assertIsNone(mfi.probe_pid_alive(4242))
|
||||
|
||||
def test_invalid_pid_is_unknown_and_probes_nothing(self):
|
||||
with mock.patch("os.kill") as killer:
|
||||
self.assertIsNone(mfi.probe_pid_alive(None))
|
||||
self.assertIsNone(mfi.probe_pid_alive(0))
|
||||
killer.assert_not_called()
|
||||
|
||||
|
||||
class TestMultiClientRegression(_AliveMixin, unittest.TestCase):
|
||||
"""The multi-LLM duplicate-server scenario that motivated #949."""
|
||||
|
||||
@staticmethod
|
||||
def _two_client_rows():
|
||||
first = healthy_rows()
|
||||
second = [
|
||||
row(
|
||||
entry["namespace"],
|
||||
entry["profile"],
|
||||
entry["role"],
|
||||
50000 + index,
|
||||
cohort_id=COHORT_B,
|
||||
)
|
||||
for index, entry in enumerate(mfi.EXPECTED_PRGS_FLEET)
|
||||
]
|
||||
return first + second
|
||||
|
||||
def test_second_client_running_the_same_five_profiles_is_caught(self):
|
||||
"""Two clients, ten servers, same profiles, same revision.
|
||||
|
||||
Every member self-reports the same parity, which is exactly why the old
|
||||
per-process surfaces reported success. The fleet inventory must report
|
||||
five duplicates, two cohorts, and a closed gate.
|
||||
"""
|
||||
rows = self._two_client_rows()
|
||||
result = classify(rows, scan=scan_for(rows))
|
||||
|
||||
self.assertEqual(len(result["duplicate_members"]), 5)
|
||||
self.assertFalse(result["exactly_one_per_profile"])
|
||||
self.assertTrue(result["mixed_cohort"])
|
||||
self.assertFalse(result["single_cohort"])
|
||||
self.assertFalse(result["mixed_revision"], "both clients share a revision")
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
self.assertEqual(result["missing_members"], [])
|
||||
|
||||
def test_identical_parity_across_ten_servers_is_not_health(self):
|
||||
rows = self._two_client_rows()
|
||||
result = classify(rows, scan=scan_for(rows))
|
||||
self.assertEqual(result["startup_revisions"], [HEAD_A])
|
||||
self.assertFalse(result["mutation_gate_satisfied"])
|
||||
|
||||
def test_every_duplicate_pid_is_named_for_the_operator(self):
|
||||
rows = self._two_client_rows()
|
||||
result = classify(rows, scan=scan_for(rows))
|
||||
for duplicate in result["duplicate_members"]:
|
||||
self.assertEqual(len(duplicate["pids"]), 2)
|
||||
|
||||
|
||||
class TestProcessScan(unittest.TestCase):
|
||||
"""The process observation is server-side and fails closed."""
|
||||
|
||||
def test_scan_parses_mcp_server_processes(self):
|
||||
stdout = (
|
||||
" PID STARTED COMMAND\n"
|
||||
" 41000 Mon Jul 27 20:00:00 2026 python /path/mcp_server.py\n"
|
||||
" 41001 Mon Jul 27 20:00:01 2026 python /path/other_server.py\n"
|
||||
)
|
||||
result = mfi.scan_mcp_server_processes(
|
||||
runner=lambda *a, **k: mock.Mock(stdout=stdout)
|
||||
)
|
||||
self.assertTrue(result["available"])
|
||||
self.assertEqual([p["pid"] for p in result["processes"]], [41000])
|
||||
|
||||
def test_scan_failure_reports_unavailable_rather_than_empty(self):
|
||||
def boom(*args, **kwargs):
|
||||
raise OSError("ps missing")
|
||||
|
||||
result = mfi.scan_mcp_server_processes(runner=boom)
|
||||
self.assertFalse(result["available"])
|
||||
self.assertEqual(result["processes"], [])
|
||||
self.assertIn("ps missing", result["reason"])
|
||||
|
||||
def test_scan_results_are_sorted_by_pid(self):
|
||||
stdout = (
|
||||
" PID STARTED COMMAND\n"
|
||||
" 41005 Mon Jul 27 20:00:00 2026 python /path/mcp_server.py\n"
|
||||
" 41001 Mon Jul 27 20:00:01 2026 python /path/mcp_server.py\n"
|
||||
)
|
||||
result = mfi.scan_mcp_server_processes(
|
||||
runner=lambda *a, **k: mock.Mock(stdout=stdout)
|
||||
)
|
||||
self.assertEqual([p["pid"] for p in result["processes"]], [41001, 41005])
|
||||
|
||||
|
||||
class TestRuntimeRecord(unittest.TestCase):
|
||||
"""The row a server writes about itself."""
|
||||
|
||||
def test_record_describes_the_calling_process(self):
|
||||
record = mfi.build_process_runtime_record(
|
||||
namespace="gitea-controller",
|
||||
profile="prgs-controller",
|
||||
role="controller",
|
||||
remote="prgs",
|
||||
org=BINDING["org"],
|
||||
repo=BINDING["repo"],
|
||||
pid=41022,
|
||||
startup_head=HEAD_A,
|
||||
transport="stdio",
|
||||
client_provenance="client_managed",
|
||||
env={mfi.COHORT_ID_ENV: COHORT_A},
|
||||
boot_id="deadbeefcafe0001",
|
||||
)
|
||||
self.assertEqual(
|
||||
record["runtime_id"], "gitea-controller:41022:deadbeefcafe0001"
|
||||
)
|
||||
self.assertEqual(record["namespace"], "gitea-controller")
|
||||
self.assertEqual(record["cohort_id"], COHORT_A)
|
||||
self.assertEqual(record["cohort_source"], "explicit_env")
|
||||
self.assertEqual(record["startup_head"], HEAD_A)
|
||||
self.assertEqual(record["status"], "running")
|
||||
|
||||
def test_record_defaults_pid_to_the_current_process(self):
|
||||
record = mfi.build_process_runtime_record(
|
||||
namespace="gitea-author", profile="prgs-author", role="author"
|
||||
)
|
||||
self.assertEqual(record["pid"], os.getpid())
|
||||
|
||||
def test_each_boot_gets_a_distinct_runtime_id(self):
|
||||
first = mfi.build_process_runtime_record(
|
||||
namespace="gitea-author", profile="prgs-author", role="author", pid=1
|
||||
)
|
||||
second = mfi.build_process_runtime_record(
|
||||
namespace="gitea-author", profile="prgs-author", role="author", pid=1
|
||||
)
|
||||
self.assertNotEqual(first["runtime_id"], second["runtime_id"])
|
||||
|
||||
def test_registered_at_uses_the_control_plane_timestamp_format(self):
|
||||
record = mfi.build_process_runtime_record(
|
||||
namespace="gitea-author", profile="prgs-author", role="author", pid=1
|
||||
)
|
||||
datetime.strptime(record["registered_at"], "%Y-%m-%dT%H:%M:%SZ")
|
||||
|
||||
|
||||
class TestSummary(_AliveMixin, unittest.TestCase):
|
||||
def test_healthy_summary_names_the_counts(self):
|
||||
result = classify(healthy_rows())
|
||||
self.assertIn("5 of 5", mfi.summarize(result))
|
||||
|
||||
def test_blocked_summary_repeats_the_blocked_reason(self):
|
||||
rows = [r for r in healthy_rows() if r["profile"] != "prgs-merger"]
|
||||
result = classify(rows)
|
||||
self.assertIn(result["blocked_reason"], mfi.summarize(result))
|
||||
|
||||
|
||||
class TestHeartbeatAge(_AliveMixin, unittest.TestCase):
|
||||
def test_heartbeat_age_is_reported_for_diagnosis(self):
|
||||
rows = healthy_rows()
|
||||
stamp = (NOW - timedelta(minutes=30)).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
rows[0]["last_heartbeat_at"] = stamp
|
||||
result = classify(rows)
|
||||
member = [m for m in result["running_members"] if m["pid"] == rows[0]["pid"]][0]
|
||||
self.assertEqual(member["heartbeat_age_seconds"], 1800)
|
||||
|
||||
def test_missing_heartbeat_is_reported_as_unknown_age(self):
|
||||
rows = healthy_rows()
|
||||
rows[0]["last_heartbeat_at"] = None
|
||||
result = classify(rows)
|
||||
member = [m for m in result["running_members"] if m["pid"] == rows[0]["pid"]][0]
|
||||
self.assertIsNone(member["heartbeat_age_seconds"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user