Compare commits
11
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7e19079b5c | ||
|
|
32ab839289 | ||
|
|
03b434a0b6 | ||
|
|
2d5d5c9d17 | ||
|
|
108cbfa173 | ||
|
|
0dbff6dcd5 | ||
|
|
4a1cc63e94 | ||
|
|
6596b259fc | ||
|
|
b0868be6b3 | ||
|
|
1a38ef95e3 | ||
|
|
324a0c8a8d |
@@ -355,6 +355,10 @@ def assess_anti_stomp_preflight(
|
||||
root_head_sha: str | None = None,
|
||||
root_porcelain: str | None = None,
|
||||
remote_master_sha: str | None = None,
|
||||
# #983: the tracking integration ref the SHA above came from, so root-checkout
|
||||
# contamination names the ref actually compared instead of a hardcoded
|
||||
# 'prgs/master'. None preserves the previous generic wording.
|
||||
remote_master_ref: str | None = None,
|
||||
check_root_checkout: bool = True,
|
||||
check_worktree: bool = True,
|
||||
create_issue_bootstrap_assessment: dict[str, Any] | None = None,
|
||||
@@ -553,6 +557,7 @@ def assess_anti_stomp_preflight(
|
||||
remote_master_sha=remote_master_sha,
|
||||
resolved_role=req_role or role,
|
||||
actual_role=role,
|
||||
remote_master_ref=remote_master_ref,
|
||||
)
|
||||
checks["root_checkout"] = {
|
||||
"block": bool(root_assessment.get("block")),
|
||||
|
||||
+456
-14
@@ -37,6 +37,34 @@ CANONICAL_ROOT_ENV = "GITEA_CANONICAL_REPOSITORY_ROOT"
|
||||
# Candidate git remote names probed when deriving repository identity.
|
||||
_IDENTITY_REMOTE_CANDIDATES = ("prgs", "origin", "dadeschools", "mdcps")
|
||||
|
||||
# Fallback integration-branch names, probed only when the checkout declares no
|
||||
# configured upstream (#983). Mirrors the stable base branches recognised
|
||||
# elsewhere in the workflow (``stacked_pr_support``, ``root_checkout_guard``).
|
||||
# The fallback is deliberately *not* ordered-first-wins: when more than one of
|
||||
# these refs exists and the checkout records no upstream, the integration branch
|
||||
# is genuinely ambiguous and resolution fails closed instead of guessing.
|
||||
INTEGRATION_BRANCH_CANDIDATES: tuple[str, ...] = ("master", "main", "dev")
|
||||
|
||||
# Sources for a *proven* base-ref derivation (#983 B1).
|
||||
#
|
||||
# ``refs/remotes/<remote>/HEAD`` is deliberately absent from this list. It is a
|
||||
# local symbolic-ref *cache* written once at clone time and refreshed only by an
|
||||
# explicit ``git remote set-head``; an ordinary fetch never updates it. When the
|
||||
# upstream default branch changes afterwards the cache keeps naming the old
|
||||
# branch, so trusting it derives the wrong integration branch for a checkout
|
||||
# that is sitting exactly on its tip. The cache is still read, but only as a
|
||||
# corroborating observation reported back to the caller — never as an authority,
|
||||
# and never as a tie-breaker between otherwise ambiguous candidates.
|
||||
BASE_REF_SOURCE_CONFIGURED_UPSTREAM = "configured_branch_upstream"
|
||||
BASE_REF_SOURCE_UNIQUE_CANDIDATE = "unique_integration_branch_ref"
|
||||
|
||||
# Reason codes for base-ref derivation outcomes (#983), so callers and tests can
|
||||
# assert the refusal cause instead of string-matching prose.
|
||||
DENY_NO_IDENTITY_REMOTE = "no_identity_remote"
|
||||
DENY_AMBIGUOUS_REMOTE = "ambiguous_identity_remote"
|
||||
DENY_AMBIGUOUS_BASE_BRANCH = "ambiguous_integration_branch"
|
||||
DENY_NO_BASE_BRANCH = "no_integration_branch_ref"
|
||||
|
||||
# Repository-authority modes (#973 B10). Exactly two values are supported.
|
||||
# ``mode`` selects how repository authority is established, so an unrecognised
|
||||
# value must never be normalised onto one of these: aliasing a trusted mode is
|
||||
@@ -120,17 +148,15 @@ def resolve_repo_toplevel(path: str) -> str | None:
|
||||
return os.path.realpath(top) if top else None
|
||||
|
||||
|
||||
def repository_identity_slug(path: str, *, remote: str | None = None) -> str | None:
|
||||
"""``owner/repository`` derived from a git remote configured at *path*.
|
||||
def _identity_remote_candidates(path: str, remote: str | None) -> list[str]:
|
||||
"""Ordered remote names to probe for identity at *path*.
|
||||
|
||||
Tries the caller-named remote first, then a small set of known remote names,
|
||||
then whatever remote the repository actually has. Returns None when no remote
|
||||
URL is parseable (identity cannot be proven).
|
||||
Names are used verbatim — never case-folded. Git config subsection names are
|
||||
case-sensitive, so a repository whose remote is ``MDCPS`` is reached only by
|
||||
the exact string ``MDCPS``; the lowercase entry in
|
||||
:data:`_IDENTITY_REMOTE_CANDIDATES` simply does not resolve, and the exact
|
||||
name arrives from ``git remote`` below (#983).
|
||||
"""
|
||||
text = (path or "").strip()
|
||||
if not text:
|
||||
return None
|
||||
|
||||
ordered: list[str] = []
|
||||
for name in (remote, *_IDENTITY_REMOTE_CANDIDATES):
|
||||
clean = (name or "").strip()
|
||||
@@ -139,7 +165,7 @@ def repository_identity_slug(path: str, *, remote: str | None = None) -> str | N
|
||||
|
||||
try:
|
||||
listed = subprocess.run(
|
||||
["git", "-C", text, "remote"],
|
||||
["git", "-C", path, "remote"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
@@ -149,11 +175,20 @@ def repository_identity_slug(path: str, *, remote: str | None = None) -> str | N
|
||||
for name in listed:
|
||||
if name and name not in ordered:
|
||||
ordered.append(name)
|
||||
return ordered
|
||||
|
||||
for name in ordered:
|
||||
|
||||
def _configured_remote_identities(path: str, remote: str | None) -> list[tuple[str, str]]:
|
||||
"""``(remote_name, owner/repository)`` for every probe name that resolves.
|
||||
|
||||
Remote names are returned exactly as configured so downstream tracking refs
|
||||
(``refs/remotes/<remote>/<branch>``) address the real ref (#983).
|
||||
"""
|
||||
found: list[tuple[str, str]] = []
|
||||
for name in _identity_remote_candidates(path, remote):
|
||||
try:
|
||||
url = subprocess.run(
|
||||
["git", "-C", text, "remote", "get-url", name],
|
||||
["git", "-C", path, "remote", "get-url", name],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
@@ -162,8 +197,415 @@ def repository_identity_slug(path: str, *, remote: str | None = None) -> str | N
|
||||
continue
|
||||
parsed = remote_repo_guard.parse_org_repo_from_remote_url(url)
|
||||
if parsed:
|
||||
return f"{parsed[0]}/{parsed[1]}"
|
||||
return None
|
||||
found.append((name, f"{parsed[0]}/{parsed[1]}"))
|
||||
return found
|
||||
|
||||
|
||||
def _configured_branch_upstream(path: str) -> tuple[str | None, str | None]:
|
||||
"""``(remote_name, branch)`` the checked-out branch is configured to track.
|
||||
|
||||
Reads ``branch.<current>.remote`` and ``branch.<current>.merge`` — the
|
||||
checkout's own explicitly configured integration target, equivalent to
|
||||
``@{upstream}``. Unlike ``refs/remotes/<remote>/HEAD`` this is not a
|
||||
clone-time cache: it is written when the branch is set up to track an
|
||||
upstream and rewritten whenever that tracking changes, so it states what the
|
||||
checkout actually integrates onto today (#983 B1).
|
||||
|
||||
Returns ``(None, None)`` on a detached HEAD or an untracked branch. Names are
|
||||
returned verbatim; git remote names are case-sensitive.
|
||||
"""
|
||||
text = (path or "").strip()
|
||||
if not text:
|
||||
return None, None
|
||||
|
||||
branch = _git_read(text, "symbolic-ref", "--quiet", "--short", "HEAD")
|
||||
if not branch:
|
||||
return None, None
|
||||
|
||||
remote = _git_read(text, "config", "--get", f"branch.{branch}.remote")
|
||||
merge = _git_read(text, "config", "--get", f"branch.{branch}.merge")
|
||||
if not remote or not merge:
|
||||
return None, None
|
||||
|
||||
prefix = "refs/heads/"
|
||||
upstream_branch = merge[len(prefix):].strip() if merge.startswith(prefix) else merge.strip()
|
||||
if not upstream_branch:
|
||||
return None, None
|
||||
return remote, upstream_branch
|
||||
|
||||
|
||||
def _cached_remote_head_branch(path: str, remote: str) -> str | None:
|
||||
"""Branch named by the *cached* ``refs/remotes/<remote>/HEAD`` symref.
|
||||
|
||||
Read for observability only. This value is never authoritative (see
|
||||
:data:`BASE_REF_SOURCE_CONFIGURED_UPSTREAM`); it is surfaced so an operator
|
||||
can see that the local cache disagrees with the configured upstream, and so
|
||||
a refusal can name the misleading signal explicitly.
|
||||
"""
|
||||
prefix = f"refs/remotes/{remote}/"
|
||||
symref = _git_read(path, "symbolic-ref", "--quiet", f"{prefix}HEAD")
|
||||
if not symref or not symref.startswith(prefix):
|
||||
return None
|
||||
branch = symref[len(prefix):].strip()
|
||||
return branch or None
|
||||
|
||||
|
||||
def assess_identity_remote(
|
||||
path: str, *, explicit_remote: str | None = None
|
||||
) -> dict:
|
||||
"""Resolve the identity remote, failing closed when the target is ambiguous.
|
||||
|
||||
This is the ambiguity-aware counterpart to :func:`resolve_identity_remote`,
|
||||
and the reason gating and reporting can no longer disagree (#983 B2).
|
||||
|
||||
*explicit_remote* is a **caller-supplied disambiguation** and nothing else.
|
||||
It must come from an operator or an explicitly sanctioned repository
|
||||
context; a value this module inferred while probing must never be handed
|
||||
back in through it, because doing so re-labels an internal first-wins guess
|
||||
as deliberate caller intent and silently suppresses the ambiguity gate.
|
||||
|
||||
Decision table:
|
||||
|
||||
* No remote yields a parseable identity -> :data:`DENY_NO_IDENTITY_REMOTE`.
|
||||
* *explicit_remote* names one of the resolving remotes -> that remote is
|
||||
authoritative, ``explicit`` True.
|
||||
* Otherwise, when every resolving remote claims the **same** repository the
|
||||
target is unambiguous and is accepted, ``explicit`` False. The remote
|
||||
named by the checkout's configured upstream is preferred among equals so
|
||||
the choice is deterministic rather than probe-order dependent.
|
||||
* Otherwise distinct remotes claim different repositories and there is no
|
||||
sanctioned disambiguation -> :data:`DENY_AMBIGUOUS_REMOTE`.
|
||||
|
||||
Returns a dict with ``remote``, ``slug``, ``identities``, ``ambiguous``,
|
||||
``explicit``, ``reason_code`` and ``reasons``. ``remote``/``slug`` are None
|
||||
on any refusal, so a caller cannot report a repository the gate refuses.
|
||||
"""
|
||||
text = (path or "").strip()
|
||||
result: dict = {
|
||||
"remote": None,
|
||||
"slug": None,
|
||||
"identities": [],
|
||||
"ambiguous": False,
|
||||
"explicit": False,
|
||||
"reason_code": None,
|
||||
"reasons": [],
|
||||
}
|
||||
if not text:
|
||||
result["reason_code"] = DENY_NO_IDENTITY_REMOTE
|
||||
result["reasons"].append(
|
||||
"no repository path supplied for identity-remote resolution (fail closed)"
|
||||
)
|
||||
return result
|
||||
|
||||
named = (explicit_remote or "").strip() or None
|
||||
identities = _configured_remote_identities(text, named)
|
||||
result["identities"] = list(identities)
|
||||
if not identities:
|
||||
result["reason_code"] = DENY_NO_IDENTITY_REMOTE
|
||||
result["reasons"].append(
|
||||
f"no git remote at '{text}' yields a parseable repository identity "
|
||||
"(fail closed)"
|
||||
)
|
||||
return result
|
||||
|
||||
if named:
|
||||
for name, slug in identities:
|
||||
if name == named:
|
||||
result["remote"] = name
|
||||
result["slug"] = slug
|
||||
result["explicit"] = True
|
||||
return result
|
||||
|
||||
distinct = {slug for _, slug in identities}
|
||||
if len(distinct) > 1:
|
||||
listed = ", ".join(f"{name} -> {slug}" for name, slug in identities)
|
||||
result["ambiguous"] = True
|
||||
result["reason_code"] = DENY_AMBIGUOUS_REMOTE
|
||||
detail = (
|
||||
f"explicitly named remote '{named}' does not resolve a repository identity "
|
||||
"there, so it cannot disambiguate; "
|
||||
if named
|
||||
else ""
|
||||
)
|
||||
result["reasons"].append(
|
||||
f"ambiguous repository identity at '{text}': {detail}remotes resolve to "
|
||||
f"different repositories ({listed}); no single authoritative target can "
|
||||
"be established (fail closed)"
|
||||
)
|
||||
return result
|
||||
|
||||
# One repository, possibly reachable through several remote names (a mirror).
|
||||
# Prefer the remote the checkout is actually configured to track so the
|
||||
# choice is deterministic instead of probe-order dependent.
|
||||
upstream_remote, _ = _configured_branch_upstream(text)
|
||||
chosen = identities[0]
|
||||
if upstream_remote:
|
||||
for entry in identities:
|
||||
if entry[0] == upstream_remote:
|
||||
chosen = entry
|
||||
break
|
||||
result["remote"], result["slug"] = chosen
|
||||
return result
|
||||
|
||||
|
||||
def resolve_identity_remote(
|
||||
path: str, *, remote: str | None = None
|
||||
) -> tuple[str | None, str | None]:
|
||||
"""``(remote_name, owner/repository)`` for the remote that proves identity.
|
||||
|
||||
The remote *name* is the piece historically thrown away by
|
||||
:func:`repository_identity_slug`, even though resolving the slug already
|
||||
required discovering it. Cross-repository base-ref derivation needs that
|
||||
name to build ``refs/remotes/<remote>/<branch>``, so it is now returned
|
||||
rather than discarded (#983). Returns ``(None, None)`` when no remote URL is
|
||||
parseable (identity cannot be proven).
|
||||
|
||||
First-wins by design: this is the identity lookup behind
|
||||
:func:`repository_identity_slug` and the #706/#973 canonical-root
|
||||
validation, which compare an observed slug against an independently trusted
|
||||
expected slug and therefore do not need an ambiguity verdict. Callers that
|
||||
*derive* a target rather than validate one — the mutation guard and the
|
||||
parity report — must use :func:`assess_identity_remote`, which fails closed
|
||||
on ambiguity (#983 B2).
|
||||
"""
|
||||
text = (path or "").strip()
|
||||
if not text:
|
||||
return None, None
|
||||
for name, slug in _configured_remote_identities(text, remote):
|
||||
return name, slug
|
||||
return None, None
|
||||
|
||||
|
||||
def repository_identity_slug(path: str, *, remote: str | None = None) -> str | None:
|
||||
"""``owner/repository`` derived from a git remote configured at *path*.
|
||||
|
||||
Tries the caller-named remote first, then a small set of known remote names,
|
||||
then whatever remote the repository actually has. Returns None when no remote
|
||||
URL is parseable (identity cannot be proven).
|
||||
"""
|
||||
return resolve_identity_remote(path, remote=remote)[1]
|
||||
|
||||
|
||||
def _base_ref_result(
|
||||
*,
|
||||
proven: bool,
|
||||
reasons: list[str],
|
||||
remote: str | None = None,
|
||||
branch: str | None = None,
|
||||
repository_slug: str | None = None,
|
||||
source: str | None = None,
|
||||
reason_code: str | None = None,
|
||||
identity_explicit: bool = False,
|
||||
configured_upstream_remote: str | None = None,
|
||||
configured_upstream_branch: str | None = None,
|
||||
cached_remote_head_branch: str | None = None,
|
||||
) -> dict:
|
||||
"""Build the base-ref derivation payload.
|
||||
|
||||
``tracking_refs`` is the ordered probe tuple downstream guards hand to
|
||||
``git rev-parse``: the ``<remote>/<branch>`` shorthand first, then the fully
|
||||
qualified ``refs/remotes/<remote>/<branch>``. It is empty whenever the
|
||||
derivation is not ``proven``, so an unresolved target can never be probed
|
||||
against some other repository's ref.
|
||||
|
||||
``cached_remote_head_branch`` reports what the local
|
||||
``refs/remotes/<remote>/HEAD`` cache claims, and
|
||||
``cached_remote_head_conflicts`` whether that claim disagrees with the branch
|
||||
actually derived. Both are observability only: the cache never decides the
|
||||
outcome (#983 B1).
|
||||
"""
|
||||
tracking_ref = f"refs/remotes/{remote}/{branch}" if proven and remote and branch else None
|
||||
tracking_refs: tuple[str, ...] = (
|
||||
(f"{remote}/{branch}", tracking_ref) if tracking_ref else ()
|
||||
)
|
||||
return {
|
||||
"proven": proven,
|
||||
"block": not proven,
|
||||
"remote": remote,
|
||||
"branch": branch,
|
||||
"repository_slug": repository_slug,
|
||||
"tracking_ref": tracking_ref,
|
||||
"tracking_refs": tracking_refs,
|
||||
"source": source,
|
||||
"reason_code": reason_code,
|
||||
"identity_explicit": identity_explicit,
|
||||
"configured_upstream_remote": configured_upstream_remote,
|
||||
"configured_upstream_branch": configured_upstream_branch,
|
||||
"cached_remote_head_branch": cached_remote_head_branch,
|
||||
"cached_remote_head_conflicts": bool(
|
||||
cached_remote_head_branch and branch and cached_remote_head_branch != branch
|
||||
),
|
||||
"reasons": list(reasons),
|
||||
}
|
||||
|
||||
|
||||
def _git_read(path: str, *args: str) -> str | None:
|
||||
"""Run a read-only git command in *path*; ``None`` on any failure."""
|
||||
try:
|
||||
res = subprocess.run(
|
||||
["git", "-C", path, *args],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
except Exception:
|
||||
return None
|
||||
if res.returncode != 0:
|
||||
return None
|
||||
return (res.stdout or "").strip() or None
|
||||
|
||||
|
||||
def _ref_exists(path: str, ref: str) -> bool:
|
||||
"""Whether *ref* resolves in the checkout at *path*."""
|
||||
try:
|
||||
res = subprocess.run(
|
||||
["git", "-C", path, "rev-parse", "--verify", "--quiet", ref],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
except Exception:
|
||||
return False
|
||||
return res.returncode == 0 and bool((res.stdout or "").strip())
|
||||
|
||||
|
||||
def resolve_target_base_ref(path: str, *, explicit_remote: str | None = None) -> dict:
|
||||
"""Derive the authoritative integration base ref for the checkout at *path*.
|
||||
|
||||
This is the single resolved target shared by cross-repository mutation
|
||||
gating and parity reporting, so the two can never disagree about which
|
||||
commit a checkout is supposed to match (#983).
|
||||
|
||||
*explicit_remote* is a caller-supplied disambiguation only; see
|
||||
:func:`assess_identity_remote` for why an internally inferred remote must
|
||||
never be passed back in here (#983 B2).
|
||||
|
||||
Resolution order:
|
||||
|
||||
1. **Identity remote** — via :func:`assess_identity_remote`, with its exact
|
||||
configured case preserved (``MDCPS`` stays ``MDCPS``). Ambiguous identity
|
||||
fails closed here rather than resolving to whichever remote probed first.
|
||||
2. **The checkout's configured upstream** — ``branch.<current>.remote`` plus
|
||||
``branch.<current>.merge``, accepted when it names the identity remote
|
||||
and its remote-tracking ref actually exists. This is the checkout's own
|
||||
declaration of what it integrates onto, and unlike the remote-HEAD cache
|
||||
it is rewritten whenever that tracking changes.
|
||||
3. **Fallback** — only when the checkout declares no usable upstream,
|
||||
exactly one of :data:`INTEGRATION_BRANCH_CANDIDATES` present as a
|
||||
remote-tracking ref.
|
||||
|
||||
``refs/remotes/<remote>/HEAD`` is **not** a step. It is read for reporting
|
||||
(``cached_remote_head_branch`` / ``cached_remote_head_conflicts``) and never
|
||||
decides the branch, because it is a clone-time cache that an ordinary fetch
|
||||
does not refresh: a checkout whose upstream default moved on still has the
|
||||
old branch cached, and trusting it gates that checkout against a ref it does
|
||||
not integrate onto (#983 B1).
|
||||
|
||||
Fails closed — ``proven`` False, empty ``tracking_refs``, and a
|
||||
``reason_code`` — when identity is unprovable, when distinct remotes claim
|
||||
different repositories, when several candidate branches exist with no
|
||||
configured upstream, or when no candidate exists at all. Nothing here
|
||||
invents a branch, writes a ref, runs a fetch, or falls back to another
|
||||
repository's base. Every ``proven`` result names a tracking ref that
|
||||
resolves in this checkout.
|
||||
"""
|
||||
text = (path or "").strip()
|
||||
if not text:
|
||||
return _base_ref_result(
|
||||
proven=False,
|
||||
reasons=["no repository path supplied for base-ref derivation (fail closed)"],
|
||||
reason_code=DENY_NO_IDENTITY_REMOTE,
|
||||
)
|
||||
|
||||
identity = assess_identity_remote(text, explicit_remote=explicit_remote)
|
||||
remote_name, slug = identity["remote"], identity["slug"]
|
||||
if not remote_name:
|
||||
return _base_ref_result(
|
||||
proven=False,
|
||||
reasons=list(identity["reasons"]),
|
||||
reason_code=identity["reason_code"],
|
||||
identity_explicit=bool(identity["explicit"]),
|
||||
)
|
||||
|
||||
prefix = f"refs/remotes/{remote_name}/"
|
||||
cached = _cached_remote_head_branch(text, remote_name)
|
||||
upstream_remote, upstream_branch = _configured_branch_upstream(text)
|
||||
common = {
|
||||
"repository_slug": slug,
|
||||
"identity_explicit": bool(identity["explicit"]),
|
||||
"configured_upstream_remote": upstream_remote,
|
||||
"configured_upstream_branch": upstream_branch,
|
||||
"cached_remote_head_branch": cached,
|
||||
}
|
||||
|
||||
# 2. The checkout's configured upstream, when it belongs to the identity
|
||||
# remote and its tracking ref is actually present. Requiring the ref to
|
||||
# exist keeps 'proven' honest: a proven target is always resolvable.
|
||||
if (
|
||||
upstream_remote == remote_name
|
||||
and upstream_branch
|
||||
and _ref_exists(text, f"{prefix}{upstream_branch}")
|
||||
):
|
||||
return _base_ref_result(
|
||||
proven=True,
|
||||
reasons=[],
|
||||
remote=remote_name,
|
||||
branch=upstream_branch,
|
||||
source=BASE_REF_SOURCE_CONFIGURED_UPSTREAM,
|
||||
**common,
|
||||
)
|
||||
|
||||
# 3. Exactly one known integration branch present as a remote-tracking ref.
|
||||
present = [
|
||||
candidate
|
||||
for candidate in INTEGRATION_BRANCH_CANDIDATES
|
||||
if _ref_exists(text, f"{prefix}{candidate}")
|
||||
]
|
||||
if len(present) == 1:
|
||||
return _base_ref_result(
|
||||
proven=True,
|
||||
reasons=[],
|
||||
remote=remote_name,
|
||||
branch=present[0],
|
||||
source=BASE_REF_SOURCE_UNIQUE_CANDIDATE,
|
||||
**common,
|
||||
)
|
||||
|
||||
# The cached remote HEAD is named in the refusal so the operator can see the
|
||||
# signal that looks authoritative but is not, and is told the read-only fix.
|
||||
cache_note = (
|
||||
f" the cached '{prefix}HEAD' names '{cached}', but that cache is written at "
|
||||
"clone time and is not refreshed by fetch, so it cannot break the tie;"
|
||||
if cached
|
||||
else ""
|
||||
)
|
||||
remedy = (
|
||||
f" Configure the checkout's upstream (git branch --set-upstream-to={remote_name}/"
|
||||
"<branch>) so the integration target is declared rather than guessed."
|
||||
)
|
||||
if len(present) > 1:
|
||||
return _base_ref_result(
|
||||
proven=False,
|
||||
reasons=[
|
||||
f"the checkout at '{text}' declares no upstream on remote "
|
||||
f"'{remote_name}' and several integration branches exist "
|
||||
f"({', '.join(present)});{cache_note} the integration base ref is "
|
||||
f"ambiguous (fail closed).{remedy}"
|
||||
],
|
||||
reason_code=DENY_AMBIGUOUS_BASE_BRANCH,
|
||||
**common,
|
||||
)
|
||||
return _base_ref_result(
|
||||
proven=False,
|
||||
reasons=[
|
||||
f"the checkout at '{text}' declares no upstream on remote '{remote_name}' "
|
||||
f"and none of {'/'.join(INTEGRATION_BRANCH_CANDIDATES)} exists as a "
|
||||
f"remote-tracking ref under '{prefix}';{cache_note} the integration base "
|
||||
f"ref cannot be derived (fail closed; no fetch is performed here).{remedy}"
|
||||
],
|
||||
reason_code=DENY_NO_BASE_BRANCH,
|
||||
**common,
|
||||
)
|
||||
|
||||
|
||||
def assess_canonical_repository_root(
|
||||
|
||||
@@ -0,0 +1,286 @@
|
||||
# Instance-level fleet identity and health snapshots
|
||||
|
||||
Issue **#978**. Companion primitives: **#948** (worker ownership), **#975**
|
||||
(heartbeat lifecycle), **#951** (restart receipts).
|
||||
|
||||
## Why this exists
|
||||
|
||||
The multi-instance fleet gate (#963) needs a *native* answer to:
|
||||
|
||||
* which application launches are live;
|
||||
* which of the five namespace workers belong to each launch;
|
||||
* whether two Codex (or Claude, Grok, …) launches are distinct;
|
||||
* whether any live collision is real rather than a shared client type.
|
||||
|
||||
Process tables, PID proximity, configuration files, and direct SQLite access
|
||||
are **not** production evidence. The sanctioned surface is the read-only MCP
|
||||
tool `gitea_snapshot_instance_fleet` on **controller** and **reconciler**
|
||||
namespaces.
|
||||
|
||||
## Identity hierarchy
|
||||
|
||||
| Identity | Scope | Who generates it | Lifetime |
|
||||
| --- | --- | --- | --- |
|
||||
| `client_type` | Application family (`codex`, `claude_code`, `gemini`, `grok`, …) | Launcher sets `GITEA_MCP_CLIENT` | Stable for the product |
|
||||
| `client_instance_id` | One running application launch | **Trusted launcher**, once per launch, as `GITEA_MCP_CLIENT_INSTANCE` | Fresh launch → new ID; reconnect of same launch → same ID; full app restart → new ID |
|
||||
| `fleet_run_id` | Operator-approved enrollment / canary cohort | Operator / controller sets `GITEA_MCP_FLEET_RUN_ID` | Duration of the approved rollout |
|
||||
| `namespace` | `author` \| `reviewer` \| `merger` \| `controller` \| `reconciler` | Profile / MCP server binding | Process lifetime |
|
||||
| `worker_id` / `worker_identity` | One namespace worker process | Runtime registry at first registration | New on worker restart; not reused |
|
||||
| `session_id` | Client session ownership | Launcher `GITEA_MCP_CLIENT_SESSION` or runtime | Session lifetime |
|
||||
| `generation_id` | One daemon launch | Runtime at process boot | New on process restart |
|
||||
| `process_identity` / PID | OS process | Runtime | Process lifetime |
|
||||
|
||||
### Rules
|
||||
|
||||
1. Multiple active instances **may** share the same `client_type`.
|
||||
2. Every application launch receives a **distinct** `client_instance_id`.
|
||||
3. All namespace workers of one launch report the **same**
|
||||
`client_instance_id`. For a whole-fleet launch that is five workers; for a
|
||||
project-scoped launch it is exactly the namespaces that launch started
|
||||
(#985).
|
||||
4. Each namespace worker has a **distinct** `worker_identity`, process
|
||||
identity, generation, and PID.
|
||||
5. Instance identity is **never** inferred from PID proximity, timestamps, or
|
||||
client type alone.
|
||||
6. Live reuse of one `client_instance_id` with conflicting workers fails closed.
|
||||
7. Sharing only a profile or `client_type` is **not** a duplicate.
|
||||
|
||||
This deliberately **replaces** any permanent `exactly_one_per_profile` fleet
|
||||
model (#949 assumption) as the operating rule for multi-instance fleets.
|
||||
|
||||
## Project-scoped launches and the runnable entry point (#985)
|
||||
|
||||
Not every project exposes all five namespaces. A project-scoped configuration
|
||||
such as Weekly Briefings intentionally has **author, reviewer, and merger
|
||||
only**, and has no controller or reconciler profile to supply. Requiring all
|
||||
five would force operators to invent profiles that must not exist, so the
|
||||
launcher accepts an explicitly validated subset.
|
||||
|
||||
### Running a launch
|
||||
|
||||
```bash
|
||||
python3 -m mcp_application_launcher \
|
||||
--client claude_code \
|
||||
--namespaces author,reviewer,merger \
|
||||
--profile author=prgs-author \
|
||||
--profile reviewer=prgs-reviewer \
|
||||
--profile merger=prgs-merger
|
||||
```
|
||||
|
||||
That mints one trusted `client_instance_id`, writes a per-launch `mcpServers`
|
||||
config, and executes:
|
||||
|
||||
```text
|
||||
claude --mcp-config <per-launch.json> --strict-mcp-config
|
||||
```
|
||||
|
||||
Add `--dry-run` to print the plan (namespaces, excluded namespaces, minted ID,
|
||||
argv, config path) as JSON and start nothing. Omit `--namespaces` to launch all
|
||||
five exactly as before. Arguments after the flags are forwarded to the client.
|
||||
|
||||
Other supported clients register one entry in
|
||||
`mcp_application_launcher.CLIENT_LAUNCH_SPECS`; because the argv builder only
|
||||
ever receives a config this module wrote, every client necessarily goes through
|
||||
the same mint-once/propagate-to-all mechanism.
|
||||
|
||||
### Why the config is per-launch and never `.mcp.json`
|
||||
|
||||
Persisting a trusted ID into a shared, reused `.mcp.json` would give two
|
||||
concurrent sessions **the same** trusted identity — precisely the reuse case
|
||||
the duplicate gate must reject. The launcher therefore writes a fresh
|
||||
owner-readable-only (`0600`) config per launch. Do not commit one, and do not
|
||||
copy a minted `GITEA_MCP_CLIENT_INSTANCE` into any checked-in configuration.
|
||||
|
||||
### Provenance sealing
|
||||
|
||||
The `inst-…` format is public and reproducible, so format alone cannot
|
||||
establish trust: anyone able to set one environment variable could otherwise
|
||||
hand-write a valid-looking ID and be believed. Trust therefore requires **both**
|
||||
the format and `GITEA_MCP_INSTANCE_PROVENANCE=trusted_launcher`, which only the
|
||||
launcher writes. A well-formed but unsealed identity is classified
|
||||
`unsealed_launcher` and **fails closed** — it is still reported for diagnosis,
|
||||
but never authorizes trusted attribution. Manually asserted trust is not
|
||||
possible.
|
||||
|
||||
### Migration and coordinated relaunch
|
||||
|
||||
Existing configurations that predate this work set no instance key at all, so
|
||||
their workers register under `legacy-pid-*` (`legacy_incomplete`) and remain
|
||||
mutually indistinguishable when two launches share a profile.
|
||||
|
||||
Migrating is a **coordinated relaunch**, not an in-place edit — a running
|
||||
worker cannot acquire an identity it was not started with:
|
||||
|
||||
1. Stop every LLM client currently running Gitea MCP workers. A single
|
||||
surviving legacy cohort keeps the fleet ambiguous.
|
||||
2. Relaunch each application through the command above.
|
||||
3. Repeat per application. Concurrent launches are expected and safe: each
|
||||
receives its own trusted ID.
|
||||
|
||||
### Post-launch verification
|
||||
|
||||
* `gitea_get_runtime_context` → `provenance_assessment.attachment` should show
|
||||
an `inst-…` `client_instance_id` with
|
||||
`instance_id_provenance: trusted_launcher`, not `legacy-pid-*` /
|
||||
`legacy_incomplete`.
|
||||
* Every namespace of the same launch must report that **same** ID; two
|
||||
different launches must report different IDs.
|
||||
* `gitea_resolve_task_capability` should return
|
||||
`exact_safe_next_action: "None; ready for operations."` with no
|
||||
`blocker_kind`. A `runtime_reconnect_required` naming duplicate PIDs per
|
||||
profile means at least one cohort is still on legacy identity, or a second
|
||||
cohort is genuinely running.
|
||||
* `mcp_application_launcher.collect_instance_ids_from_mcp_servers(servers,
|
||||
namespaces=[...])` returns `shared_single_trusted_id` for a built config, and
|
||||
reports `missing_servers` when an expected namespace entry is absent.
|
||||
|
||||
## How five workers join one instance
|
||||
|
||||
1. The host starts one application instance (for example one Codex session).
|
||||
2. The **production application launcher**
|
||||
(`mcp_application_launcher.build_application_mcp_servers` /
|
||||
`gitea_config.multi_namespace_launcher_entries`) mints exactly one
|
||||
`client_instance_id` via `mcp_fleet_snapshot.generate_client_instance_id`
|
||||
and injects the same env into every namespace worker:
|
||||
|
||||
```text
|
||||
GITEA_MCP_CLIENT=<client_type>
|
||||
GITEA_MCP_CLIENT_INSTANCE=<client_instance_id>
|
||||
GITEA_MCP_INSTANCE_PROVENANCE=trusted_launcher
|
||||
GITEA_MCP_CLIENT_SESSION=<session_id> # optional but recommended
|
||||
GITEA_MCP_FLEET_RUN_ID=<enrollment id> # when on an approved canary
|
||||
GITEA_CLIENT_MANAGED=1
|
||||
GITEA_MCP_PROFILE=<role-profile>
|
||||
```
|
||||
|
||||
3. The MCP client starts the five namespace processes (`gitea-author`,
|
||||
`gitea-reviewer`, `gitea-merger`, `gitea-controller`, `gitea-reconciler`)
|
||||
from that generated `mcpServers` map — each entry carries the **same**
|
||||
instance ID.
|
||||
4. Each worker registers once into the worker registry with its own
|
||||
`worker_identity`, `namespace`, `generation_id`, and PID, but the shared
|
||||
`client_instance_id`. Workers never mint a trusted instance ID themselves.
|
||||
|
||||
Only IDs matching the launcher format `inst-<client>-<timestamp>-<digest>`
|
||||
are trusted. Missing, `legacy-pid-*` / `pid-*` placeholders, and ordinary
|
||||
user-supplied strings fail closed as untrusted and cannot authorize
|
||||
multi-instance fleet mutation safety.
|
||||
|
||||
Worker reconnect (same process, same registration) keeps the instance ID.
|
||||
Worker restart (new process) mints a new worker identity and generation but
|
||||
must still receive the same `GITEA_MCP_CLIENT_INSTANCE` from the parent
|
||||
application if it is the same launch. Full application restart mints a new
|
||||
`client_instance_id` (call the launcher without a prior ID).
|
||||
|
||||
### Mutation gate (instance-aware)
|
||||
|
||||
The capability / runtime diagnostic gate
|
||||
(`_check_mcp_runtimes_diagnostics`) is **instance-aware**:
|
||||
|
||||
* Two legitimate application instances that share a profile (same
|
||||
`GITEA_MCP_PROFILE`) are allowed when each has a distinct trusted
|
||||
`client_instance_id`.
|
||||
* More than one live worker for the same
|
||||
`(client_instance_id, profile/namespace)` fails closed as a duplicate
|
||||
namespace worker.
|
||||
* Multiple processes sharing a profile **without** trusted instance
|
||||
evidence still fail closed (indistinguishable from a duplicate).
|
||||
|
||||
## Approved fleet manifest
|
||||
|
||||
An operator (or controller enrollment step) obtains an approved manifest as a
|
||||
list of expected instances, for example:
|
||||
|
||||
```json
|
||||
{
|
||||
"instances": [
|
||||
{
|
||||
"client_type": "codex",
|
||||
"client_instance_id": "inst-codex-20260730T120000Z-abc123def456",
|
||||
"fleet_run_id": "canary-963-2026-07-30",
|
||||
"namespaces": ["author", "reviewer", "merger", "controller", "reconciler"]
|
||||
},
|
||||
{
|
||||
"client_type": "codex",
|
||||
"client_instance_id": "inst-codex-20260730T120100Z-fed654cba321",
|
||||
"fleet_run_id": "canary-963-2026-07-30",
|
||||
"namespaces": ["author", "reviewer", "merger", "controller", "reconciler"]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Pass that list as `expected_manifest` to `gitea_snapshot_instance_fleet`.
|
||||
When a manifest is supplied:
|
||||
|
||||
* instances on the manifest but not live → `missing_expected`;
|
||||
* live instances not on the manifest → `unmanifested`;
|
||||
* both are **active blockers** for fleet-gate safety.
|
||||
|
||||
Without a manifest, the snapshot still enumerates the live fleet and classifies
|
||||
identity collisions; it does not invent enrollment policy.
|
||||
|
||||
## Snapshot consistency
|
||||
|
||||
Each snapshot includes:
|
||||
|
||||
* `snapshot_at` — UTC timestamp;
|
||||
* `consistency_token` / `registry_revision` — content digest over worker
|
||||
identity, instance id, status, heartbeat, fencing, and generation.
|
||||
|
||||
Two successive snapshots can prove heartbeat continuity via
|
||||
`mcp_fleet_snapshot.compare_snapshot_heartbeats`.
|
||||
|
||||
## Classification (active vs historical)
|
||||
|
||||
**Active blockers** (make `live_fleet_safe=false`):
|
||||
|
||||
* missing expected instance;
|
||||
* unmanifested extra instance;
|
||||
* duplicate namespace worker within one instance;
|
||||
* live `client_instance_id` collision;
|
||||
* reused worker / session / generation / process / PID / fencing identity;
|
||||
* orphaned or unowned workers;
|
||||
* unknown client;
|
||||
* foreign-repository workers;
|
||||
* old-revision workers;
|
||||
* stale workers still marked active;
|
||||
* legacy incomplete instance identity.
|
||||
|
||||
**Historical dead rows** (`status` released/superseded) are reported as
|
||||
`historical_dead` findings with `active_blocker=false`. They never
|
||||
automatically make the live fleet unsafe.
|
||||
|
||||
## Fail-closed behaviour and recovery
|
||||
|
||||
| Condition | Mutation safety | Diagnostic reads | Recovery |
|
||||
| --- | --- | --- | --- |
|
||||
| Two instances, same `client_type`, distinct IDs | Safe (if otherwise healthy) | Available | None needed |
|
||||
| Live reuse of one `client_instance_id` | Unsafe | Available | Stop the colliding launch or re-issue a distinct ID |
|
||||
| Two workers, same namespace, one instance | Unsafe | Available | Stop the extra worker |
|
||||
| Legacy registration without trusted instance ID | Unsafe for fleet mutation | Available | Relaunch with launcher-issued `GITEA_MCP_CLIENT_INSTANCE` |
|
||||
| Historical dead row only | Does not block alone | Available | No action required for fleet safety |
|
||||
| Registry unavailable | Snapshot fails closed | N/A | Repair registry path / reconnect namespaces |
|
||||
|
||||
Incomplete or untrusted instance identity **cannot** authorize unsafe
|
||||
mutation. Diagnostic reads remain available where `gitea.read` allows.
|
||||
|
||||
## Permissions
|
||||
|
||||
* **Allowed:** `controller`, `reconciler` with `gitea.read`.
|
||||
* **Denied:** author, reviewer, merger (even with `gitea.read` for other tools).
|
||||
* **No new mutation permissions** are granted to any role.
|
||||
|
||||
## Related surfaces
|
||||
|
||||
* `mcp_worker_identity` — registry, worker identity, heartbeats (#948, #975).
|
||||
* `mcp_fleet_snapshot` — pure snapshot + classification (#978).
|
||||
* `gitea_snapshot_instance_fleet` — sanctioned MCP tool (#978).
|
||||
* `gitea_get_runtime_context` — single-process view (not fleet-wide).
|
||||
|
||||
## Non-goals
|
||||
|
||||
* Starting the #963 canary.
|
||||
* Purging historical registry rows.
|
||||
* Preserving “exactly one process per profile” as the permanent model.
|
||||
* Using shell / process-table / SQLite inspection as production fleet evidence.
|
||||
@@ -159,6 +159,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_sentry_reconcile_issue`
|
||||
- `gitea_sentry_watchdog`
|
||||
- `gitea_set_issue_labels`
|
||||
- `gitea_snapshot_instance_fleet`
|
||||
- `gitea_submit_pr_review`
|
||||
- `gitea_update_pr_branch_by_merge`
|
||||
- `gitea_validate_review_final_report`
|
||||
|
||||
+90
-15
@@ -1193,6 +1193,31 @@ RECOGNIZED_GITEA_ENV_KEYS = frozenset({
|
||||
"GITEA_IRRECOVERABLE_HMAC_SECRET",
|
||||
"GITEA_FORCE_MCP_RUNTIME_CHECK",
|
||||
"GITEA_FORCE_CLIENT_MANAGED",
|
||||
# #975: the client-identity inputs the server actually consumes at startup
|
||||
# (CLIENT_NAME_ENV / CLIENT_INSTANCE_ENV / CLIENT_SESSION_ENV in
|
||||
# gitea_mcp_server). Production read them while this allowlist omitted them,
|
||||
# so the peer-env scan classified them as unsupported overrides and the
|
||||
# capability resolver refused every mutation fleet-wide. Named individually
|
||||
# on purpose: no prefix is added, so an unrecognised GITEA_* override is
|
||||
# still refused exactly as it was before.
|
||||
"GITEA_MCP_CLIENT",
|
||||
"GITEA_MCP_CLIENT_INSTANCE",
|
||||
"GITEA_MCP_CLIENT_SESSION",
|
||||
# #978: operator-approved fleet enrollment identity shared by one launch.
|
||||
"GITEA_MCP_FLEET_RUN_ID",
|
||||
"GITEA_MCP_PROCESS_IDENTITY",
|
||||
# #978 B1/B2: launcher-sealed provenance + peer-visible worker identity so
|
||||
# the instance-aware mutation gate can distinguish two legitimate
|
||||
# application instances that share a profile from a true duplicate worker.
|
||||
"GITEA_MCP_INSTANCE_PROVENANCE",
|
||||
"GITEA_MCP_WORKER_IDENTITY",
|
||||
"GITEA_MCP_GENERATION_ID",
|
||||
# #975 review 652 B1: production also consumes HEARTBEAT_INTERVAL_ENV from
|
||||
# mcp_worker_identity via gitea_mcp_server._start_worker_heartbeat. Omitting
|
||||
# it reproduced the same unsupported-env → runtime_reconnect_required
|
||||
# failure mode for the documented operator override. Named individually;
|
||||
# no GITEA_* / GITEA_WORKER_* prefix is added.
|
||||
"GITEA_WORKER_HEARTBEAT_INTERVAL_SECONDS",
|
||||
})
|
||||
|
||||
RECOGNIZED_GITEA_ENV_PREFIXES = (
|
||||
@@ -1220,24 +1245,74 @@ def get_unconsumed_gitea_env_overrides(env=None) -> dict[str, str]:
|
||||
return unconsumed
|
||||
|
||||
|
||||
def launcher_entry(profile_name, config_path=None):
|
||||
def launcher_entry(
|
||||
profile_name,
|
||||
config_path=None,
|
||||
*,
|
||||
client_type=None,
|
||||
client_instance_id=None,
|
||||
fleet_run_id=None,
|
||||
session_id=None,
|
||||
launch_nonce=None,
|
||||
):
|
||||
"""Return a thin MCP launcher entry for *profile_name*.
|
||||
|
||||
Contains command/args and the GITEA_MCP_* / GITEA_CLIENT_MANAGED env vars — never a token
|
||||
or password. Suitable for Claude / Gemini / Codex ``mcpServers`` blocks.
|
||||
Contains command/args and the GITEA_MCP_* / GITEA_CLIENT_MANAGED env vars —
|
||||
never a token or password. Suitable for Claude / Gemini / Codex
|
||||
``mcpServers`` blocks.
|
||||
|
||||
#978 B1: production launches mint (or reuse) a trusted
|
||||
``GITEA_MCP_CLIENT_INSTANCE`` so workers never fall back to the legacy
|
||||
placeholder identity on the real serve path. Pass *client_instance_id* to
|
||||
resume the same launch; omit it for a fresh launch (new ID).
|
||||
"""
|
||||
command, args = server_command()
|
||||
return {
|
||||
"gitea-tools": {
|
||||
"command": command,
|
||||
"args": args,
|
||||
"env": {
|
||||
"GITEA_MCP_CONFIG": config_path or DEFAULT_CONFIG_PATH,
|
||||
"GITEA_MCP_PROFILE": profile_name,
|
||||
"GITEA_CLIENT_MANAGED": "1",
|
||||
},
|
||||
}
|
||||
}
|
||||
import mcp_application_launcher as app_launcher
|
||||
|
||||
built = app_launcher.launcher_entry_for_profile(
|
||||
profile_name,
|
||||
client_type=client_type or "unknown",
|
||||
config_path=config_path or DEFAULT_CONFIG_PATH,
|
||||
client_instance_id=client_instance_id,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
launch_nonce=launch_nonce,
|
||||
server_key="gitea-tools",
|
||||
)
|
||||
# Public shape stays {server_key: {command, args, env}} for existing callers.
|
||||
return {"gitea-tools": built["gitea-tools"]}
|
||||
|
||||
|
||||
def multi_namespace_launcher_entries(
|
||||
profile_by_namespace,
|
||||
*,
|
||||
client_type,
|
||||
namespaces=None,
|
||||
config_path=None,
|
||||
client_instance_id=None,
|
||||
fleet_run_id=None,
|
||||
session_id=None,
|
||||
launch_nonce=None,
|
||||
):
|
||||
"""Build production mcpServers for the namespaces of one application launch.
|
||||
|
||||
One shared trusted ``client_instance_id`` is minted (or reused) and
|
||||
propagated to every namespace worker env. *namespaces* selects a validated
|
||||
project-scoped subset (#985); omitting it launches all five sanctioned
|
||||
namespaces as before. See
|
||||
:func:`mcp_application_launcher.build_application_mcp_servers`.
|
||||
"""
|
||||
import mcp_application_launcher as app_launcher
|
||||
|
||||
return app_launcher.build_application_mcp_servers(
|
||||
profile_by_namespace,
|
||||
client_type=client_type,
|
||||
namespaces=namespaces,
|
||||
config_path=config_path or DEFAULT_CONFIG_PATH,
|
||||
client_instance_id=client_instance_id,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
launch_nonce=launch_nonce,
|
||||
)
|
||||
|
||||
|
||||
|
||||
|
||||
+489
-47
@@ -1052,9 +1052,16 @@ def _create_issue_bootstrap_assessment(
|
||||
git_state = issue_lock_worktree.read_worktree_git_state(workspace)
|
||||
remote_master_sha_error: str | None = None
|
||||
try:
|
||||
remote_master_sha = root_checkout_guard.resolve_remote_master_sha(
|
||||
# #983: consume the resolved-target state so an *underivable* base ref
|
||||
# reaches the assessor as named missing evidence rather than a bare
|
||||
# None, which the bootstrap would otherwise report only as "live
|
||||
# master tip is unknown" with no cause.
|
||||
_base_state = root_checkout_guard.resolve_remote_master_ref_state(
|
||||
ctx["canonical_repo_root"]
|
||||
)
|
||||
remote_master_sha = _base_state["sha"]
|
||||
if not remote_master_sha and _base_state.get("reasons"):
|
||||
remote_master_sha_error = "; ".join(_base_state["reasons"])
|
||||
except Exception as exc:
|
||||
remote_master_sha = None
|
||||
remote_master_sha_error = (
|
||||
@@ -1082,9 +1089,14 @@ def _create_issue_bootstrap_assessment(
|
||||
# closed instead of proceeding without base-equivalence proof.
|
||||
remote_master_sha_error: str | None = None
|
||||
try:
|
||||
remote_master_sha = root_checkout_guard.resolve_remote_master_sha(
|
||||
# #983: same resolved-target consumption as the author bootstrap above —
|
||||
# a target whose base ref cannot be derived must say why.
|
||||
_base_state = root_checkout_guard.resolve_remote_master_ref_state(
|
||||
ctx["canonical_repo_root"]
|
||||
)
|
||||
remote_master_sha = _base_state["sha"]
|
||||
if not remote_master_sha and _base_state.get("reasons"):
|
||||
remote_master_sha_error = "; ".join(_base_state["reasons"])
|
||||
except Exception as exc:
|
||||
remote_master_sha = None
|
||||
remote_master_sha_error = f"{type(exc).__name__}: {exc}".strip() or "resolver failed"
|
||||
@@ -1303,7 +1315,12 @@ def _run_anti_stomp_preflight(
|
||||
workspace = ctx["workspace_path"]
|
||||
canonical_root = ctx["canonical_repo_root"]
|
||||
git_state = issue_lock_worktree.read_worktree_git_state(canonical_root)
|
||||
remote_master_sha = root_checkout_guard.resolve_remote_master_sha(canonical_root)
|
||||
# #983: derive the target base ref for this repository and carry the ref
|
||||
# itself alongside the SHA, so the anti-stomp root-checkout check reports the
|
||||
# ref it actually compared.
|
||||
_root_base_state = root_checkout_guard.resolve_remote_master_ref_state(canonical_root)
|
||||
remote_master_sha = _root_base_state["sha"]
|
||||
remote_master_ref = _root_base_state.get("ref")
|
||||
|
||||
# Repo/org facts (best-effort; explicit org/repo when provided).
|
||||
resolved_org = org
|
||||
@@ -1432,6 +1449,7 @@ def _run_anti_stomp_preflight(
|
||||
root_head_sha=git_state.get("head_sha"),
|
||||
root_porcelain=git_state.get("porcelain_status") or "",
|
||||
remote_master_sha=remote_master_sha,
|
||||
remote_master_ref=remote_master_ref,
|
||||
startup_head=startup_head,
|
||||
current_code_head=current_code_head,
|
||||
lease_required=lease_required,
|
||||
@@ -1916,8 +1934,16 @@ def _verify_role_mutation_workspace(
|
||||
if runtime_reasons:
|
||||
raise RuntimeError("; ".join(runtime_reasons))
|
||||
except Exception as exc:
|
||||
if "stale-runtime:" in str(exc):
|
||||
raise RuntimeError(str(exc))
|
||||
# #975: ``_check_mcp_runtimes_diagnostics`` raises every reason it
|
||||
# produces through this one RuntimeError, but only ``stale-runtime:``
|
||||
# was re-raised here — an ``unsupported-env:`` reason was swallowed
|
||||
# while still failing the capability resolver, so this preflight and
|
||||
# the resolver disagreed about the identical diagnostic. Both
|
||||
# authoritative prefixes now propagate the same way. This can only ever
|
||||
# widen what is refused, never widen what is permitted.
|
||||
message = str(exc)
|
||||
if any(prefix in message for prefix in RUNTIME_DIAGNOSTIC_HARD_PREFIXES):
|
||||
raise RuntimeError(message)
|
||||
pass
|
||||
|
||||
role = _effective_workspace_role()
|
||||
@@ -1979,7 +2005,11 @@ def _enforce_root_checkout_guard(worktree_path: str | None = None) -> None:
|
||||
canonical_root = ctx["canonical_repo_root"]
|
||||
workspace = ctx["workspace_path"]
|
||||
git_state = issue_lock_worktree.read_worktree_git_state(canonical_root)
|
||||
remote_master_sha = root_checkout_guard.resolve_remote_master_sha(canonical_root)
|
||||
# #983: consume the resolved target so the contamination message names the
|
||||
# ref that was actually compared (refs/remotes/<remote>/<branch>) instead of
|
||||
# a hardcoded 'prgs/master' the target repository may not have.
|
||||
base_state = root_checkout_guard.resolve_remote_master_ref_state(canonical_root)
|
||||
remote_master_sha = base_state["sha"]
|
||||
assessment = root_checkout_guard.assess_root_checkout_guard(
|
||||
workspace_path=workspace,
|
||||
canonical_repo_root=canonical_root,
|
||||
@@ -1989,6 +2019,7 @@ def _enforce_root_checkout_guard(worktree_path: str | None = None) -> None:
|
||||
remote_master_sha=remote_master_sha,
|
||||
resolved_role=_preflight_resolved_role,
|
||||
actual_role=_actual_profile_role(),
|
||||
remote_master_ref=base_state.get("ref"),
|
||||
)
|
||||
if assessment["block"]:
|
||||
raise RuntimeError(root_checkout_guard.format_root_checkout_guard_error(assessment))
|
||||
@@ -15714,6 +15745,10 @@ _WORKER_REGISTRY = None
|
||||
_WORKER_IDENTITY: str | None = None
|
||||
_WORKER_GENERATION: str | None = None
|
||||
_WORKER_REGISTRATION_ATTEMPTED = False
|
||||
#: #975: the one heartbeat supervisor for this process's registration. One
|
||||
#: process registers exactly one worker identity, so there is exactly one
|
||||
#: supervisor and it is never replaced.
|
||||
_WORKER_HEARTBEAT_SUPERVISOR = None
|
||||
|
||||
#: Env a client launcher may set to name itself and its session. Absent values
|
||||
#: are reported as unknown; they are never guessed at, because guessing is what
|
||||
@@ -15721,6 +15756,10 @@ _WORKER_REGISTRATION_ATTEMPTED = False
|
||||
CLIENT_NAME_ENV = "GITEA_MCP_CLIENT"
|
||||
CLIENT_INSTANCE_ENV = "GITEA_MCP_CLIENT_INSTANCE"
|
||||
CLIENT_SESSION_ENV = "GITEA_MCP_CLIENT_SESSION"
|
||||
FLEET_RUN_ENV = "GITEA_MCP_FLEET_RUN_ID"
|
||||
INSTANCE_PROVENANCE_ENV = "GITEA_MCP_INSTANCE_PROVENANCE"
|
||||
WORKER_IDENTITY_ENV = "GITEA_MCP_WORKER_IDENTITY"
|
||||
GENERATION_ID_ENV = "GITEA_MCP_GENERATION_ID"
|
||||
|
||||
|
||||
def _worker_registry():
|
||||
@@ -15744,13 +15783,57 @@ def _worker_registry():
|
||||
|
||||
|
||||
def _client_identity_hints() -> dict:
|
||||
"""What the launcher told us about itself. Unset fields stay unset."""
|
||||
"""What the launcher told us about itself (#948 / #975 / #978).
|
||||
|
||||
``client_instance_id`` is established by the trusted application launcher
|
||||
once per launch and shared by all five namespace workers. Workers inherit
|
||||
that value from the production launcher env and never mint a trusted ID
|
||||
themselves. When the launcher key is absent or untrusted we still register
|
||||
under an explicit *legacy* placeholder so diagnostic reads work, but the
|
||||
registration is marked incomplete and cannot authorize multi-instance fleet
|
||||
mutation safety.
|
||||
"""
|
||||
import mcp_application_launcher as app_launcher
|
||||
import mcp_fleet_snapshot
|
||||
|
||||
inherited = app_launcher.inherit_or_refuse_client_instance(os.environ)
|
||||
client_name = (os.environ.get(CLIENT_NAME_ENV) or "").strip() or None
|
||||
instance = {
|
||||
"client_instance_id": inherited.get("client_instance_id"),
|
||||
"provenance": inherited.get("provenance"),
|
||||
"trusted": bool(inherited.get("trusted")),
|
||||
"complete": bool(inherited.get("complete")),
|
||||
"launcher_sealed": bool(inherited.get("launcher_sealed")),
|
||||
}
|
||||
# Legacy placeholder only when the launcher omitted a usable key — never
|
||||
# invent a trusted ID from PID proximity or ordinary user env.
|
||||
if not instance["client_instance_id"] or not instance["trusted"]:
|
||||
if not instance["client_instance_id"]:
|
||||
placeholder = f"legacy-pid-{os.getpid()}"
|
||||
fallback = mcp_fleet_snapshot.assess_instance_identity(
|
||||
placeholder, source="pid_fallback"
|
||||
)
|
||||
instance = {
|
||||
"client_instance_id": fallback["client_instance_id"],
|
||||
"provenance": fallback["provenance"],
|
||||
"trusted": False,
|
||||
"complete": False,
|
||||
"launcher_sealed": False,
|
||||
}
|
||||
else:
|
||||
# Malformed / untrusted user-supplied value: keep it for diagnosis
|
||||
# but never mark trusted.
|
||||
instance["trusted"] = False
|
||||
instance["launcher_sealed"] = False
|
||||
return {
|
||||
"client_name": (os.environ.get(CLIENT_NAME_ENV) or "").strip() or None,
|
||||
"client_instance_id": (os.environ.get(CLIENT_INSTANCE_ENV) or "").strip()
|
||||
or f"pid-{os.getpid()}",
|
||||
"client_name": client_name,
|
||||
"client_instance_id": instance["client_instance_id"],
|
||||
"instance_id_provenance": instance["provenance"],
|
||||
"instance_identity_trusted": instance["trusted"],
|
||||
"instance_launcher_sealed": instance.get("launcher_sealed", False),
|
||||
"session_id": (os.environ.get(CLIENT_SESSION_ENV) or "").strip()
|
||||
or f"proc-{os.getpid()}-{_process_boot_head_sha or 'nohead'}",
|
||||
"fleet_run_id": (os.environ.get(FLEET_RUN_ENV) or "").strip() or None,
|
||||
}
|
||||
|
||||
|
||||
@@ -15782,24 +15865,71 @@ def _active_worker_identity() -> str | None:
|
||||
identity = mcp_worker_identity.generate_worker_identity(
|
||||
hints["client_name"], hints["session_id"]
|
||||
)
|
||||
role = _active_role_kind_safe()
|
||||
profile_name = (
|
||||
(os.environ.get(gitea_config.ENV_PROFILE) or "").strip() or None
|
||||
)
|
||||
# Namespace is the role namespace (author/reviewer/…) when known.
|
||||
namespace = role if role in {
|
||||
"author", "reviewer", "merger", "controller", "reconciler"
|
||||
} else None
|
||||
parity = None
|
||||
try:
|
||||
parity = _current_master_parity()
|
||||
except Exception:
|
||||
parity = None
|
||||
outcome = registry.register(
|
||||
worker_identity=identity,
|
||||
client_name=hints["client_name"],
|
||||
client_instance_id=hints["client_instance_id"],
|
||||
session_id=hints["session_id"],
|
||||
generation_id=generation,
|
||||
role=_active_role_kind_safe(),
|
||||
profile=(os.environ.get(gitea_config.ENV_PROFILE) or "").strip() or None,
|
||||
role=role,
|
||||
profile=profile_name,
|
||||
namespace=namespace,
|
||||
remote=(os.environ.get("GITEA_MCP_REMOTE") or "").strip() or None,
|
||||
repository_binding=PROJECT_ROOT,
|
||||
pid=os.getpid(),
|
||||
transport=native.get("bound_transport"),
|
||||
token_fingerprint=native.get("token_fingerprint"),
|
||||
pid_alive_probe=issue_lock_store.is_process_alive,
|
||||
fleet_run_id=hints.get("fleet_run_id"),
|
||||
process_identity=f"pid-{os.getpid()}",
|
||||
startup_revision=(parity or {}).get("startup_head")
|
||||
or _process_boot_head_sha,
|
||||
loaded_revision=(parity or {}).get("local_head")
|
||||
or (parity or {}).get("current_head"),
|
||||
parity_revision=(parity or {}).get("current_head"),
|
||||
live_revision=(parity or {}).get("live_remote_head"),
|
||||
instance_id_provenance=hints.get("instance_id_provenance"),
|
||||
)
|
||||
if outcome.get("registered"):
|
||||
_WORKER_IDENTITY = identity
|
||||
_WORKER_GENERATION = generation
|
||||
# #978 B2: export worker/generation/instance into this process
|
||||
# env so peer process scans (and the instance-aware mutation
|
||||
# gate) can attribute live workers without treating shared
|
||||
# profiles as duplicates. Never overwrite a launcher-sealed
|
||||
# client_instance_id with a different value.
|
||||
os.environ[WORKER_IDENTITY_ENV] = identity
|
||||
os.environ[GENERATION_ID_ENV] = generation
|
||||
if hints.get("client_instance_id") and not (
|
||||
os.environ.get(CLIENT_INSTANCE_ENV) or ""
|
||||
).strip():
|
||||
os.environ[CLIENT_INSTANCE_ENV] = str(
|
||||
hints["client_instance_id"]
|
||||
)
|
||||
# #975: registration is the only moment identity and fencing
|
||||
# epoch are both known, so the heartbeat supervisor is started
|
||||
# here. Without it ``last_heartbeat_at`` never left
|
||||
# ``started_at`` and every healthy client lost ownership at the
|
||||
# TTL.
|
||||
_start_worker_heartbeat(
|
||||
identity=identity,
|
||||
fencing_epoch=outcome.get("fencing_epoch"),
|
||||
hints=hints,
|
||||
generation=generation,
|
||||
)
|
||||
return identity
|
||||
if not outcome.get("collision"):
|
||||
return None
|
||||
@@ -15810,6 +15940,78 @@ def _active_worker_identity() -> str | None:
|
||||
return None
|
||||
|
||||
|
||||
def _start_worker_heartbeat(
|
||||
*,
|
||||
identity: str,
|
||||
fencing_epoch,
|
||||
hints: dict,
|
||||
generation: str | None,
|
||||
) -> None:
|
||||
"""Attach a heartbeat supervisor to the registration just created (#975).
|
||||
|
||||
Never raises: a supervisor that cannot start leaves the row exactly as
|
||||
``register()`` wrote it, which is the pre-#975 behaviour, rather than
|
||||
failing the tool call that happened to trigger lazy registration.
|
||||
|
||||
Under pytest the thread is deliberately not started. Tests drive
|
||||
``WorkerHeartbeatSupervisor`` directly with an injected clock, so the suite
|
||||
proves the lifecycle without leaving background sqlite writers behind.
|
||||
"""
|
||||
global _WORKER_HEARTBEAT_SUPERVISOR
|
||||
if _WORKER_HEARTBEAT_SUPERVISOR is not None:
|
||||
return
|
||||
registry = _worker_registry()
|
||||
if registry is None or fencing_epoch is None:
|
||||
return
|
||||
try:
|
||||
supervisor = mcp_worker_identity.WorkerHeartbeatSupervisor(
|
||||
registry,
|
||||
worker_identity=identity,
|
||||
fencing_epoch=int(fencing_epoch),
|
||||
session_id=hints.get("session_id"),
|
||||
generation_id=generation,
|
||||
client_name=hints.get("client_name"),
|
||||
pid=os.getpid(),
|
||||
ttl_seconds=mcp_worker_identity.DEFAULT_HEARTBEAT_TTL_SECONDS,
|
||||
interval_seconds=os.environ.get(
|
||||
mcp_worker_identity.HEARTBEAT_INTERVAL_ENV
|
||||
),
|
||||
)
|
||||
_WORKER_HEARTBEAT_SUPERVISOR = supervisor
|
||||
if not mcp_daemon_guard.is_pytest_runtime():
|
||||
supervisor.start()
|
||||
except Exception:
|
||||
return
|
||||
|
||||
|
||||
def _worker_heartbeat_status() -> dict:
|
||||
"""Read-only heartbeat observability for ``gitea_get_runtime_context`` (#975).
|
||||
|
||||
Reports the unsupervised case explicitly rather than omitting the key, so an
|
||||
operator can tell "no supervisor" apart from "supervisor with no beats yet".
|
||||
"""
|
||||
supervisor = _WORKER_HEARTBEAT_SUPERVISOR
|
||||
if supervisor is None:
|
||||
return {
|
||||
"supervised": False,
|
||||
"running": False,
|
||||
"reasons": [
|
||||
"no worker heartbeat supervisor is attached to this process; the "
|
||||
"registration is not being renewed and will go stale at its TTL"
|
||||
],
|
||||
}
|
||||
try:
|
||||
status = dict(supervisor.status())
|
||||
except Exception as exc:
|
||||
return {
|
||||
"supervised": True,
|
||||
"running": False,
|
||||
"reasons": [f"heartbeat status unavailable: {type(exc).__name__}: {exc}"],
|
||||
}
|
||||
status.setdefault("reasons", [])
|
||||
return status
|
||||
|
||||
|
||||
def _active_role_kind_safe() -> str | None:
|
||||
"""Best-effort role for the registry record; never raises into a tool call.
|
||||
|
||||
@@ -19278,6 +19480,172 @@ def gitea_validate_review_final_report(
|
||||
)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_snapshot_instance_fleet(
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
expected_manifest: list | dict | str | None = None,
|
||||
include_historical: bool = True,
|
||||
canonical_repository: str | None = None,
|
||||
expected_live_revision: str | None = None,
|
||||
) -> dict:
|
||||
"""Read-only: authoritative instance-level fleet identity and health snapshot (#978).
|
||||
|
||||
Controller and reconciler only. Returns a point-in-time snapshot of every
|
||||
registered namespace worker with instance attribution, heartbeat freshness,
|
||||
and structured classification (expected/missing/unmanifested, duplicate
|
||||
namespace workers, identity collisions, foreign repository, old revision,
|
||||
historical dead rows). Sharing only a client type is never a duplicate.
|
||||
|
||||
Does not grant any mutation capability. Does not scan process tables or
|
||||
open foreign databases for production evidence — the worker registry is
|
||||
the sole authority.
|
||||
|
||||
Args:
|
||||
remote: Known instance — 'dadeschools' or 'prgs'.
|
||||
host: Optional host override.
|
||||
org: Optional org override (audit context only).
|
||||
repo: Optional repo override (audit context only).
|
||||
expected_manifest: Optional approved fleet enrollment list
|
||||
(list of dicts with client_instance_id / client_type / fleet_run_id).
|
||||
When supplied, missing and unmanifested instances are classified.
|
||||
include_historical: Include released/superseded rows (default True);
|
||||
historical rows never make the live fleet unsafe by themselves.
|
||||
canonical_repository: Expected repository binding path/slug for
|
||||
foreign-repository classification.
|
||||
expected_live_revision: When set, workers whose recorded revisions
|
||||
differ are classified as old-revision.
|
||||
"""
|
||||
import json as _json
|
||||
|
||||
import mcp_fleet_snapshot
|
||||
|
||||
read_block = _profile_operation_gate("gitea.read")
|
||||
if read_block:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": read_block,
|
||||
"permission_report": _permission_block_report("gitea.read"),
|
||||
}
|
||||
|
||||
profile = get_profile()
|
||||
role = _profile_role_kind(profile)
|
||||
if role not in {"controller", "reconciler"}:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"allowed": False,
|
||||
"denied_role": role,
|
||||
"required_roles": ["controller", "reconciler"],
|
||||
"mutation_performed": False,
|
||||
"reasons": [
|
||||
f"gitea_snapshot_instance_fleet is restricted to controller and "
|
||||
f"reconciler roles; active role_kind is {role!r}. Author, "
|
||||
"reviewer, and merger profiles keep gitea.read for diagnosis "
|
||||
"elsewhere but do not receive this fleet surface, and no "
|
||||
"unrelated mutation permission is granted."
|
||||
],
|
||||
"exact_next_action": (
|
||||
"Re-run from a prgs-controller or prgs-reconciler namespace."
|
||||
),
|
||||
}
|
||||
|
||||
manifest: list | None = None
|
||||
if expected_manifest is not None:
|
||||
raw = expected_manifest
|
||||
if isinstance(raw, str):
|
||||
try:
|
||||
raw = _json.loads(raw)
|
||||
except Exception as exc:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": [
|
||||
f"expected_manifest is not valid JSON: {type(exc).__name__}"
|
||||
],
|
||||
}
|
||||
if isinstance(raw, dict):
|
||||
# Accept {"instances": [...]} or a single instance object.
|
||||
if "instances" in raw and isinstance(raw["instances"], list):
|
||||
raw = raw["instances"]
|
||||
else:
|
||||
raw = [raw]
|
||||
if not isinstance(raw, list):
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": ["expected_manifest must be a list of instance objects"],
|
||||
}
|
||||
manifest = raw
|
||||
|
||||
registry = _worker_registry()
|
||||
if registry is None:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": [
|
||||
"worker registry is unavailable; cannot produce an authoritative "
|
||||
"fleet snapshot (fail closed)"
|
||||
],
|
||||
"exact_next_action": (
|
||||
"Ensure GITEA_WORKER_REGISTRY_DB is writable and re-run after "
|
||||
"workers have registered."
|
||||
),
|
||||
}
|
||||
|
||||
try:
|
||||
if include_historical:
|
||||
records = registry.list_workers(status=None)
|
||||
else:
|
||||
records = registry.list_workers(status=mcp_worker_identity.STATUS_ACTIVE)
|
||||
except Exception as exc:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": [
|
||||
f"failed to read worker registry: {type(exc).__name__}: {_redact(str(exc))}"
|
||||
],
|
||||
}
|
||||
|
||||
parity = None
|
||||
try:
|
||||
parity = _current_master_parity()
|
||||
except Exception:
|
||||
parity = None
|
||||
live_rev = expected_live_revision or (parity or {}).get("live_remote_head")
|
||||
canon = canonical_repository or PROJECT_ROOT
|
||||
|
||||
snapshot = mcp_fleet_snapshot.snapshot_instance_fleet(
|
||||
records,
|
||||
expected_manifest=manifest,
|
||||
pid_alive_probe=issue_lock_store.is_process_alive,
|
||||
canonical_repository=canon,
|
||||
expected_live_revision=live_rev,
|
||||
)
|
||||
snapshot["role_kind"] = role
|
||||
snapshot["profile"] = profile.get("profile_name")
|
||||
snapshot["remote"] = _effective_remote(remote)
|
||||
snapshot["repository"] = {
|
||||
"org": org,
|
||||
"repo": repo,
|
||||
"canonical_repository": canon,
|
||||
}
|
||||
snapshot["mutation_performed"] = False
|
||||
snapshot["permission_scope"] = {
|
||||
"read_only": True,
|
||||
"granted_operations": ["gitea.read"],
|
||||
"denied_unrelated_mutations": True,
|
||||
"note": (
|
||||
"This capability is strictly observational. It does not authorize "
|
||||
"branch, issue, PR, review, merge, or restart mutations."
|
||||
),
|
||||
}
|
||||
return snapshot
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_get_runtime_context(
|
||||
remote: str = "dadeschools",
|
||||
@@ -19451,6 +19819,11 @@ def gitea_get_runtime_context(
|
||||
"fencing_epoch": provenance_assessment["fencing_epoch"],
|
||||
"conflicting_live_sessions": provenance_assessment["conflicting_live_sessions"],
|
||||
"provenance_assessment": provenance_assessment,
|
||||
# #975: whether this registration is actually being renewed. Read-only,
|
||||
# and it grants nothing — ownership still comes from the attachment
|
||||
# record above. It exists so "my heartbeat stopped" is diagnosable
|
||||
# before the TTL turns it into session_attachment_missing.
|
||||
"worker_heartbeat": _worker_heartbeat_status(),
|
||||
"unconsumed_gitea_env": unconsumed_env,
|
||||
"preflight_ready": preflight["preflight_ready"],
|
||||
"preflight_block_reasons": preflight["preflight_block_reasons"],
|
||||
@@ -21959,6 +22332,17 @@ def gitea_route_task_session(
|
||||
# self-recovery was removed from the read-only path (was _trigger_mcp_auto_restart).
|
||||
# Recovery is owned exclusively by the IDE/client reconnect path.
|
||||
|
||||
#: Every reason prefix ``_check_mcp_runtimes_diagnostics`` can emit. Callers
|
||||
#: raise its reasons as one RuntimeError, and #975 found that the preflight
|
||||
#: re-raise recognised only ``stale-runtime:``, silently dropping
|
||||
#: ``unsupported-env:`` while the capability resolver still failed on it. This
|
||||
#: lives beside the producer so a newly added reason family cannot be forgotten
|
||||
#: by a distant re-raise predicate again.
|
||||
RUNTIME_DIAGNOSTIC_HARD_PREFIXES: tuple[str, ...] = (
|
||||
"stale-runtime:",
|
||||
"unsupported-env:",
|
||||
)
|
||||
|
||||
|
||||
def _check_mcp_runtimes_diagnostics(task: str, matching_profiles: list[str]) -> list[str]:
|
||||
"""Read-only: report missing or stale MCP runtimes (no config or process mutation).
|
||||
@@ -22053,6 +22437,20 @@ def _check_mcp_runtimes_diagnostics(task: str, matching_profiles: list[str]) ->
|
||||
)
|
||||
peer_worker_identity = peer_env.get("GITEA_MCP_WORKER_IDENTITY")
|
||||
peer_generation = peer_env.get("GITEA_MCP_GENERATION_ID")
|
||||
# #978 B2: instance attribution for the live fleet gate. Two legitimate
|
||||
# application instances may share a profile when each has a distinct
|
||||
# trusted client_instance_id; only same-(instance, profile/namespace)
|
||||
# duplicates remain blocked.
|
||||
import mcp_fleet_snapshot as _fleet
|
||||
|
||||
peer_instance_raw = peer_env.get("GITEA_MCP_CLIENT_INSTANCE")
|
||||
peer_instance_assess = _fleet.assess_instance_identity(peer_instance_raw)
|
||||
peer_client_instance = (
|
||||
peer_instance_assess["client_instance_id"]
|
||||
if peer_instance_assess.get("trusted")
|
||||
else None
|
||||
)
|
||||
peer_instance_trusted = bool(peer_instance_assess.get("trusted"))
|
||||
|
||||
for env_match in re.finditer(r'\b(GITEA_[A-Z0-9_]+)=([^\s]+)', env_out):
|
||||
k, v = env_match.group(1), env_match.group(2)
|
||||
@@ -22070,6 +22468,9 @@ def _check_mcp_runtimes_diagnostics(task: str, matching_profiles: list[str]) ->
|
||||
"is_client_managed": is_client_managed,
|
||||
"worker_identity": peer_worker_identity,
|
||||
"generation_id": peer_generation,
|
||||
"client_instance_id": peer_client_instance,
|
||||
"instance_identity_trusted": peer_instance_trusted,
|
||||
"raw_client_instance_id": peer_instance_raw,
|
||||
}
|
||||
if profile not in all_profile_procs:
|
||||
all_profile_procs[profile] = []
|
||||
@@ -22077,44 +22478,85 @@ def _check_mcp_runtimes_diagnostics(task: str, matching_profiles: list[str]) ->
|
||||
|
||||
running_profiles = {}
|
||||
for profile, procs in all_profile_procs.items():
|
||||
# #948 AC40/AC43: sharing a profile is legitimate — profile is a
|
||||
# reusable capability definition, not a worker identity. What is *not*
|
||||
# legitimate is reusing one worker identity, or two live sessions
|
||||
# claiming one generation. Distinctness has to be proven, though:
|
||||
# processes carrying no identity evidence are indistinguishable, so
|
||||
# they stay classified as duplicates and keep the #686 wall intact.
|
||||
identified = [p for p in procs if p.get("worker_identity")]
|
||||
distinct_identities = {p["worker_identity"] for p in identified}
|
||||
all_identified = len(identified) == len(procs)
|
||||
duplicate_identity = len(identified) != len(distinct_identities)
|
||||
contested_generation = any(
|
||||
len({p["worker_identity"] for p in identified if p.get("generation_id") == gen}) > 1
|
||||
for gen in {p.get("generation_id") for p in identified if p.get("generation_id")}
|
||||
)
|
||||
|
||||
if len(procs) > 1 and (
|
||||
not all_identified or duplicate_identity or contested_generation
|
||||
):
|
||||
pids_str = ", ".join(str(p["pid"]) for p in procs)
|
||||
if duplicate_identity or contested_generation:
|
||||
detail = (
|
||||
"The same worker identity or generation is claimed more than once, "
|
||||
"so these are genuine duplicates rather than independent workers."
|
||||
)
|
||||
# #978 B2 / #948 AC40: sharing a profile is legitimate when each live
|
||||
# process belongs to a distinct trusted application instance. The
|
||||
# permanent model is instance-aware — not exactly_one_per_profile.
|
||||
# Fail closed only when:
|
||||
# * more than one live worker claims the same (client_instance_id,
|
||||
# profile/namespace), or
|
||||
# * worker identity / generation is reused, or
|
||||
# * multiple processes share a profile without trusted instance
|
||||
# evidence (indistinguishable → treat as duplicate).
|
||||
# Group by trusted client_instance_id (untrusted/missing → one bucket).
|
||||
by_instance: dict[str, list[dict]] = {}
|
||||
for p in procs:
|
||||
if p.get("instance_identity_trusted") and p.get("client_instance_id"):
|
||||
key = str(p["client_instance_id"])
|
||||
else:
|
||||
detail = (
|
||||
"Manual or duplicate launches defeat staleness detection and cannot "
|
||||
"receive client stdio."
|
||||
key = "__untrusted_or_missing__"
|
||||
by_instance.setdefault(key, []).append(p)
|
||||
|
||||
for instance_key, group in by_instance.items():
|
||||
if len(group) <= 1:
|
||||
continue
|
||||
identified = [p for p in group if p.get("worker_identity")]
|
||||
distinct_identities = {p["worker_identity"] for p in identified}
|
||||
all_identified = len(identified) == len(group)
|
||||
duplicate_identity = len(identified) != len(distinct_identities)
|
||||
contested_generation = any(
|
||||
len(
|
||||
{
|
||||
p["worker_identity"]
|
||||
for p in identified
|
||||
if p.get("generation_id") == gen
|
||||
}
|
||||
)
|
||||
reasons.append(
|
||||
f"stale-runtime: Duplicate MCP server process(es) detected for profile '{profile}' (PIDs: {pids_str}). "
|
||||
+ detail
|
||||
> 1
|
||||
for gen in {
|
||||
p.get("generation_id")
|
||||
for p in identified
|
||||
if p.get("generation_id")
|
||||
}
|
||||
)
|
||||
# Otherwise: several independently identified workers share one profile.
|
||||
# No reason is appended, deliberately. Every reason this function
|
||||
# returns is raised as a hard RuntimeError by its callers, so recording
|
||||
# legitimate concurrency here as "informational" would block exactly the
|
||||
# case #948 exists to permit.
|
||||
# Same trusted (instance, profile) with >1 live process is always a
|
||||
# duplicate-namespace-worker, even if worker identities differ —
|
||||
# one application launch gets exactly one worker per namespace.
|
||||
same_trusted_instance = instance_key != "__untrusted_or_missing__"
|
||||
if same_trusted_instance or (
|
||||
not all_identified or duplicate_identity or contested_generation
|
||||
):
|
||||
pids_str = ", ".join(str(p["pid"]) for p in group)
|
||||
if same_trusted_instance:
|
||||
detail = (
|
||||
f"More than one live worker for profile '{profile}' under "
|
||||
f"client_instance_id {instance_key!r} (duplicate namespace "
|
||||
"worker). Two legitimate application instances with "
|
||||
"distinct trusted client_instance_id values may share a "
|
||||
"profile; this collision does not."
|
||||
)
|
||||
elif duplicate_identity or contested_generation:
|
||||
detail = (
|
||||
"The same worker identity or generation is claimed more "
|
||||
"than once, so these are genuine duplicates rather than "
|
||||
"independent workers."
|
||||
)
|
||||
else:
|
||||
detail = (
|
||||
"Multiple processes share this profile without distinct "
|
||||
"trusted client_instance_id evidence, so they cannot be "
|
||||
"told apart from a duplicate MCP process. Relaunch via "
|
||||
"the production application launcher so each instance "
|
||||
"receives a trusted GITEA_MCP_CLIENT_INSTANCE."
|
||||
)
|
||||
reasons.append(
|
||||
f"stale-runtime: Duplicate MCP server process(es) detected "
|
||||
f"for profile '{profile}' (PIDs: {pids_str}). "
|
||||
+ detail
|
||||
)
|
||||
# Multiple trusted instances sharing one profile intentionally produce
|
||||
# no reason here. Callers raise every reason as a hard RuntimeError, so
|
||||
# treating legitimate multi-instance concurrency as "informational"
|
||||
# would re-introduce the profile-only wall #978 removes.
|
||||
client_procs = [p for p in procs if p["is_client_managed"]]
|
||||
if client_procs:
|
||||
client_procs.sort(key=lambda p: p["start_time"], reverse=True)
|
||||
|
||||
+60
-11
@@ -378,6 +378,11 @@ def format_parity(assessment: dict) -> str:
|
||||
# feeds the mutation gate and never changes startup_head/current_head.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Legacy hardcoded target tracking ref. Retained for callers that still pass an
|
||||
# explicit ref, but no longer the default: assuming a remote named ``origin``
|
||||
# read an unrelated (often orphaned) remote-tracking ref in any checkout whose
|
||||
# remote is named something else, and reported the target stale against a commit
|
||||
# from a remote that may no longer even be configured (#983).
|
||||
DEFAULT_TARGET_TRACKING_REF = "refs/remotes/origin/master"
|
||||
|
||||
|
||||
@@ -408,7 +413,7 @@ def assess_target_repository_parity(
|
||||
*,
|
||||
canonical_root: str | None,
|
||||
source: str | None,
|
||||
tracking_ref: str = DEFAULT_TARGET_TRACKING_REF,
|
||||
tracking_ref: str | None = None,
|
||||
) -> dict:
|
||||
"""Assess the configured cross-repository target checkout.
|
||||
|
||||
@@ -421,6 +426,11 @@ def assess_target_repository_parity(
|
||||
An unconfigured namespace is ``configured=False`` and never ``stale`` — the
|
||||
single-repository default has no second dimension to be stale about. A
|
||||
configured root that cannot be read is ``determinable=False`` with reasons.
|
||||
|
||||
*tracking_ref* defaults to None, meaning **derive the target from the
|
||||
repository itself** through the same resolver the mutation guard uses, so
|
||||
gating and reporting can never disagree about which ref is authoritative
|
||||
(#983). Passing an explicit ref preserves the previous behaviour verbatim.
|
||||
"""
|
||||
result = {
|
||||
"configured": bool(canonical_root),
|
||||
@@ -429,9 +439,13 @@ def assess_target_repository_parity(
|
||||
"repository_slug": None,
|
||||
"checkout_head": None,
|
||||
"tracking_ref": tracking_ref,
|
||||
"base_remote": None,
|
||||
"base_branch": None,
|
||||
"tracking_ref_source": "explicit_tracking_ref" if tracking_ref else None,
|
||||
"remote_tracking_head": None,
|
||||
"determinable": False,
|
||||
"stale": False,
|
||||
"reason_code": None,
|
||||
"reasons": [],
|
||||
}
|
||||
if not canonical_root:
|
||||
@@ -463,18 +477,53 @@ def assess_target_repository_parity(
|
||||
result["checkout_head"] = head
|
||||
result["determinable"] = True
|
||||
|
||||
remote_url = _git_capture(canonical_root, "remote", "get-url", "origin")
|
||||
if remote_url:
|
||||
# Local import keeps this module dependency-light for its startup role.
|
||||
import remote_repo_guard
|
||||
# Local import keeps this module dependency-light for its startup role.
|
||||
import canonical_repository_root as _crr
|
||||
|
||||
parsed = remote_repo_guard.parse_org_repo_from_remote_url(remote_url)
|
||||
if parsed:
|
||||
result["repository_slug"] = f"{parsed[0]}/{parsed[1]}"
|
||||
if not result["repository_slug"]:
|
||||
result["reasons"].append(
|
||||
"target repository identity could not be derived from its git remote"
|
||||
# #983: identity comes from whichever remote actually proves it, not from a
|
||||
# remote assumed to be named 'origin'. In a checkout whose only remote is
|
||||
# 'prgs', the old lookup failed outright and reported the identity as
|
||||
# underivable while a leftover refs/remotes/origin/master still resolved.
|
||||
#
|
||||
# #983 B2: this must be the *ambiguity-aware* resolver. The first-wins
|
||||
# `resolve_identity_remote` picks whichever remote probes first, so on a
|
||||
# target where distinct remotes claim different repositories the report
|
||||
# confidently named one of them while the mutation gate refused the same
|
||||
# target — gating and reporting evaluating different repositories, which is
|
||||
# precisely the divergence this issue exists to end.
|
||||
identity = _crr.assess_identity_remote(canonical_root)
|
||||
result["repository_slug"] = identity["slug"]
|
||||
if not identity["slug"]:
|
||||
result["reason_code"] = identity["reason_code"]
|
||||
result["reasons"].extend(
|
||||
identity["reasons"]
|
||||
or ["target repository identity could not be derived from its git remote"]
|
||||
)
|
||||
if identity["ambiguous"]:
|
||||
# An ambiguous target has no single authoritative base, so reporting one
|
||||
# would be a guess. Fail closed here exactly as the gate does.
|
||||
return result
|
||||
|
||||
# Identity is otherwise resolved independently of the base ref: a target that
|
||||
# has never been fetched still has a provable repository identity, and
|
||||
# reporting it as unidentifiable would lose real information over an
|
||||
# unrelated missing ref.
|
||||
if not tracking_ref:
|
||||
# No remote argument. The identity remote resolved just above was
|
||||
# *inferred here*, and feeding it back in would tell the resolver a
|
||||
# caller had explicitly disambiguated the repository, suppressing its
|
||||
# ambiguity gate (#983 B2). Only an operator-supplied remote may do that,
|
||||
# and this call site has none.
|
||||
base = _crr.resolve_target_base_ref(canonical_root)
|
||||
if not base.get("proven"):
|
||||
result["reason_code"] = base.get("reason_code")
|
||||
result["reasons"].extend(base.get("reasons") or [])
|
||||
return result
|
||||
tracking_ref = base["tracking_ref"]
|
||||
result["tracking_ref"] = tracking_ref
|
||||
result["tracking_ref_source"] = base.get("source")
|
||||
result["base_remote"] = base.get("remote")
|
||||
result["base_branch"] = base.get("branch")
|
||||
|
||||
tracking_head = _git_capture(canonical_root, "rev-parse", tracking_ref)
|
||||
if not tracking_head:
|
||||
|
||||
@@ -0,0 +1,753 @@
|
||||
"""Production application launcher for multi-namespace MCP fleets (#978 B1).
|
||||
|
||||
One real LLM application launch mints exactly one trusted
|
||||
``client_instance_id`` and propagates it to every Gitea MCP namespace worker
|
||||
started for that launch. Workers never invent a trusted instance identity from
|
||||
PID proximity, timestamps, or ordinary untrusted environment values.
|
||||
|
||||
This module is the production serve-path authority for instance identity.
|
||||
Tests and fixtures may call the same functions, but production registration
|
||||
receives the identity from the env this launcher builds — not from a hand-set
|
||||
test-only helper that bypasses it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import secrets
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
import mcp_fleet_snapshot as fleet
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
#: Canonical MCP server names for the five role namespaces.
|
||||
NAMESPACE_SERVER_NAMES: tuple[str, ...] = (
|
||||
"gitea-author",
|
||||
"gitea-reviewer",
|
||||
"gitea-merger",
|
||||
"gitea-controller",
|
||||
"gitea-reconciler",
|
||||
)
|
||||
|
||||
#: Map MCP server name → role/namespace kind.
|
||||
SERVER_TO_NAMESPACE: dict[str, str] = {
|
||||
"gitea-author": "author",
|
||||
"gitea-reviewer": "reviewer",
|
||||
"gitea-merger": "merger",
|
||||
"gitea-controller": "controller",
|
||||
"gitea-reconciler": "reconciler",
|
||||
}
|
||||
|
||||
SANCTIONED_NAMESPACES: tuple[str, ...] = (
|
||||
"author",
|
||||
"reviewer",
|
||||
"merger",
|
||||
"controller",
|
||||
"reconciler",
|
||||
)
|
||||
|
||||
# Env the launcher may set on every worker of one application launch.
|
||||
CLIENT_NAME_ENV = "GITEA_MCP_CLIENT"
|
||||
CLIENT_INSTANCE_ENV = fleet.CLIENT_INSTANCE_ENV
|
||||
FLEET_RUN_ENV = fleet.FLEET_RUN_ENV
|
||||
CLIENT_SESSION_ENV = "GITEA_MCP_CLIENT_SESSION"
|
||||
CLIENT_MANAGED_ENV = "GITEA_CLIENT_MANAGED"
|
||||
PROFILE_ENV = "GITEA_MCP_PROFILE"
|
||||
CONFIG_ENV = "GITEA_MCP_CONFIG"
|
||||
INSTANCE_PROVENANCE_ENV = "GITEA_MCP_INSTANCE_PROVENANCE"
|
||||
WORKER_IDENTITY_ENV = "GITEA_MCP_WORKER_IDENTITY"
|
||||
GENERATION_ID_ENV = "GITEA_MCP_GENERATION_ID"
|
||||
|
||||
# Marker the trusted launcher alone writes; ordinary user env without this
|
||||
# marker is never classified as launcher-trusted provenance.
|
||||
LAUNCHER_PROVENANCE_VALUE = fleet.INSTANCE_ID_PROVENANCE_TRUSTED
|
||||
|
||||
|
||||
def mint_application_launch(
|
||||
client_type: str | None,
|
||||
*,
|
||||
launch_nonce: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Mint one trusted application-instance identity for a production launch.
|
||||
|
||||
Called exactly once per real application launch. The returned
|
||||
``client_instance_id`` is injected into every namespace worker environment
|
||||
for that launch. A second call (separate launch) yields a different ID.
|
||||
"""
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
instance_id = fleet.generate_client_instance_id(
|
||||
client, launch_nonce=launch_nonce, now=now
|
||||
)
|
||||
assessment = fleet.assess_instance_identity(instance_id)
|
||||
if not assessment["trusted"]:
|
||||
# generate_client_instance_id always produces a trusted format; fail
|
||||
# closed if that invariant ever breaks rather than shipping untrusted.
|
||||
raise RuntimeError(
|
||||
f"launcher produced untrusted client_instance_id {instance_id!r}: "
|
||||
f"{assessment.get('reasons')}"
|
||||
)
|
||||
session = (session_id or "").strip() or f"launch-{secrets.token_hex(12)}"
|
||||
return {
|
||||
"client_type": client,
|
||||
"client_instance_id": instance_id,
|
||||
"instance_id_provenance": LAUNCHER_PROVENANCE_VALUE,
|
||||
"instance_identity_trusted": True,
|
||||
"fleet_run_id": (fleet_run_id or "").strip() or None,
|
||||
"session_id": session,
|
||||
"namespaces": list(SANCTIONED_NAMESPACES),
|
||||
"namespace_server_names": list(NAMESPACE_SERVER_NAMES),
|
||||
}
|
||||
|
||||
|
||||
def namespace_worker_env(
|
||||
*,
|
||||
profile_name: str,
|
||||
client_type: str | None,
|
||||
client_instance_id: str,
|
||||
config_path: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
extra_env: Mapping[str, str] | None = None,
|
||||
) -> dict[str, str]:
|
||||
"""Build the environment for one namespace worker of a trusted launch.
|
||||
|
||||
Never invents a client_instance_id. The caller must supply the launch-minted
|
||||
identity so all five workers receive the same value.
|
||||
"""
|
||||
assessment = fleet.assess_instance_identity(client_instance_id)
|
||||
if not assessment["trusted"]:
|
||||
raise ValueError(
|
||||
"namespace_worker_env refuses untrusted client_instance_id "
|
||||
f"{client_instance_id!r}: {assessment.get('reasons')}"
|
||||
)
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
env: dict[str, str] = {
|
||||
PROFILE_ENV: str(profile_name),
|
||||
CLIENT_MANAGED_ENV: "1",
|
||||
"GITEA_MCP_CLIENT": client,
|
||||
CLIENT_INSTANCE_ENV: assessment["client_instance_id"],
|
||||
INSTANCE_PROVENANCE_ENV: LAUNCHER_PROVENANCE_VALUE,
|
||||
}
|
||||
if config_path:
|
||||
env[CONFIG_ENV] = str(config_path)
|
||||
if fleet_run_id:
|
||||
env[FLEET_RUN_ENV] = str(fleet_run_id)
|
||||
if session_id:
|
||||
env[CLIENT_SESSION_ENV] = str(session_id)
|
||||
if extra_env:
|
||||
# Never let untrusted callers override the trusted instance keys.
|
||||
protected = {
|
||||
CLIENT_INSTANCE_ENV,
|
||||
INSTANCE_PROVENANCE_ENV,
|
||||
"GITEA_MCP_CLIENT",
|
||||
CLIENT_MANAGED_ENV,
|
||||
}
|
||||
for key, value in extra_env.items():
|
||||
if key in protected:
|
||||
continue
|
||||
env[str(key)] = str(value)
|
||||
return env
|
||||
|
||||
|
||||
def resolve_launch_namespaces(
|
||||
namespaces: "Sequence[str] | None" = None,
|
||||
) -> tuple[str, ...]:
|
||||
"""Validate an explicit project-scoped namespace subset (#985).
|
||||
|
||||
``None`` keeps the historical whole-fleet behaviour and resolves to all five
|
||||
sanctioned namespaces. An explicit sequence is validated against
|
||||
:data:`SANCTIONED_NAMESPACES` and returned in canonical fleet order, so a
|
||||
project-scoped configuration (for example author/reviewer/merger only) can
|
||||
launch without inventing controller or reconciler profiles.
|
||||
|
||||
This is an allow-list, not a filter: anything outside the sanctioned set is
|
||||
refused rather than silently dropped, so a typo can never quietly shrink a
|
||||
launch to fewer workers than the operator intended.
|
||||
"""
|
||||
if namespaces is None:
|
||||
return tuple(SANCTIONED_NAMESPACES)
|
||||
|
||||
requested = [str(ns).strip().lower() for ns in namespaces]
|
||||
if not requested or any(not ns for ns in requested):
|
||||
raise ValueError(
|
||||
"namespaces must be a non-empty sequence of sanctioned namespace "
|
||||
f"names; choose from {list(SANCTIONED_NAMESPACES)}"
|
||||
)
|
||||
|
||||
unknown = sorted({ns for ns in requested if ns not in SANCTIONED_NAMESPACES})
|
||||
if unknown:
|
||||
raise ValueError(
|
||||
f"unsanctioned namespace(s) {unknown} requested; sanctioned "
|
||||
f"namespaces are {list(SANCTIONED_NAMESPACES)}"
|
||||
)
|
||||
|
||||
duplicates = sorted({ns for ns in requested if requested.count(ns) > 1})
|
||||
if duplicates:
|
||||
raise ValueError(
|
||||
f"duplicate namespace(s) {duplicates} requested; each sanctioned "
|
||||
"namespace may appear at most once in one launch"
|
||||
)
|
||||
|
||||
selected = set(requested)
|
||||
return tuple(ns for ns in SANCTIONED_NAMESPACES if ns in selected)
|
||||
|
||||
|
||||
def build_application_mcp_servers(
|
||||
profile_by_namespace: Mapping[str, str],
|
||||
*,
|
||||
client_type: str | None,
|
||||
namespaces: "Sequence[str] | None" = None,
|
||||
config_path: str | None = None,
|
||||
client_instance_id: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
launch_nonce: str | None = None,
|
||||
command: str | None = None,
|
||||
args: list[str] | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build a production ``mcpServers`` map for one application launch.
|
||||
|
||||
Mints one ``client_instance_id`` when *client_instance_id* is omitted (fresh
|
||||
launch). When the caller supplies a previously minted trusted ID (resume of
|
||||
the same launch / config rewrite), that ID is reused so reconnect keeps
|
||||
attribution. A full new application restart omits the ID and receives a
|
||||
fresh mint.
|
||||
|
||||
Every namespace server entry receives the **same** instance ID. Separate
|
||||
calls with no supplied ID receive distinct IDs.
|
||||
|
||||
*namespaces* selects an explicitly validated project-scoped subset (#985).
|
||||
Omitting it preserves the original whole-fleet behaviour, so existing
|
||||
five-namespace callers are unaffected. Only the resolved namespaces need a
|
||||
profile, and only they are started; excluded namespaces are reported so a
|
||||
caller can prove a controller/reconciler worker was never launched.
|
||||
"""
|
||||
resolved_namespaces = resolve_launch_namespaces(namespaces)
|
||||
|
||||
missing = [
|
||||
ns for ns in resolved_namespaces if not profile_by_namespace.get(ns)
|
||||
]
|
||||
if missing:
|
||||
raise ValueError(
|
||||
"build_application_mcp_servers requires a profile for every "
|
||||
f"requested namespace; missing: {missing}"
|
||||
)
|
||||
|
||||
if client_instance_id is None:
|
||||
launch = mint_application_launch(
|
||||
client_type,
|
||||
launch_nonce=launch_nonce,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
now=now,
|
||||
)
|
||||
else:
|
||||
assessment = fleet.assess_instance_identity(client_instance_id)
|
||||
if not assessment["trusted"]:
|
||||
raise ValueError(
|
||||
"refusing to propagate untrusted client_instance_id "
|
||||
f"{client_instance_id!r} into production launch envs: "
|
||||
f"{assessment.get('reasons')}"
|
||||
)
|
||||
launch = {
|
||||
"client_type": mwi.normalize_client_name(client_type),
|
||||
"client_instance_id": assessment["client_instance_id"],
|
||||
"instance_id_provenance": LAUNCHER_PROVENANCE_VALUE,
|
||||
"instance_identity_trusted": True,
|
||||
"fleet_run_id": (fleet_run_id or "").strip() or None,
|
||||
"session_id": (session_id or "").strip()
|
||||
or f"launch-{secrets.token_hex(12)}",
|
||||
"namespaces": list(SANCTIONED_NAMESPACES),
|
||||
"namespace_server_names": list(NAMESPACE_SERVER_NAMES),
|
||||
}
|
||||
|
||||
# The launch record describes only the namespaces this launch actually
|
||||
# starts, so downstream attribution never implies an unstarted worker.
|
||||
launch["namespaces"] = list(resolved_namespaces)
|
||||
launch["namespace_server_names"] = [
|
||||
f"gitea-{ns}" for ns in resolved_namespaces
|
||||
]
|
||||
|
||||
# Resolve command/args from the production server entry when not provided.
|
||||
if command is None or args is None:
|
||||
import gitea_config
|
||||
|
||||
cmd, cmd_args = gitea_config.server_command()
|
||||
command = command or cmd
|
||||
args = args if args is not None else list(cmd_args)
|
||||
|
||||
servers: dict[str, Any] = {}
|
||||
shared_id = launch["client_instance_id"]
|
||||
for namespace in resolved_namespaces:
|
||||
server_name = f"gitea-{namespace}"
|
||||
profile = profile_by_namespace[namespace]
|
||||
env = namespace_worker_env(
|
||||
profile_name=profile,
|
||||
client_type=launch["client_type"],
|
||||
client_instance_id=shared_id,
|
||||
config_path=config_path,
|
||||
fleet_run_id=launch.get("fleet_run_id"),
|
||||
session_id=launch.get("session_id"),
|
||||
)
|
||||
servers[server_name] = {
|
||||
"command": command,
|
||||
"args": list(args),
|
||||
"env": env,
|
||||
}
|
||||
|
||||
excluded = [ns for ns in SANCTIONED_NAMESPACES if ns not in resolved_namespaces]
|
||||
return {
|
||||
"mcpServers": servers,
|
||||
"launch": launch,
|
||||
"client_instance_id": shared_id,
|
||||
"client_type": launch["client_type"],
|
||||
"namespaces": list(resolved_namespaces),
|
||||
"excluded_namespaces": excluded,
|
||||
"project_scoped": bool(excluded),
|
||||
"shared_instance_id_across_namespaces": True,
|
||||
"namespace_count": len(resolved_namespaces),
|
||||
}
|
||||
|
||||
|
||||
def launcher_entry_for_profile(
|
||||
profile_name: str,
|
||||
*,
|
||||
client_type: str | None = None,
|
||||
config_path: str | None = None,
|
||||
client_instance_id: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
server_key: str = "gitea-tools",
|
||||
launch_nonce: str | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Thin single-server production launcher entry with trusted instance ID.
|
||||
|
||||
Used when only one namespace is being configured. Still mints (or reuses)
|
||||
a trusted ``client_instance_id`` so production never relies on the legacy
|
||||
placeholder identity for normal launches.
|
||||
"""
|
||||
import gitea_config
|
||||
|
||||
if client_instance_id is None:
|
||||
launch = mint_application_launch(
|
||||
client_type or "unknown",
|
||||
launch_nonce=launch_nonce,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
now=now,
|
||||
)
|
||||
client_instance_id = launch["client_instance_id"]
|
||||
client = launch["client_type"]
|
||||
fleet_run = launch.get("fleet_run_id")
|
||||
session = launch.get("session_id")
|
||||
else:
|
||||
assessment = fleet.assess_instance_identity(client_instance_id)
|
||||
if not assessment["trusted"]:
|
||||
raise ValueError(
|
||||
f"untrusted client_instance_id {client_instance_id!r}"
|
||||
)
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
fleet_run = (fleet_run_id or "").strip() or None
|
||||
session = (session_id or "").strip() or None
|
||||
client_instance_id = assessment["client_instance_id"]
|
||||
|
||||
command, args = gitea_config.server_command()
|
||||
env = namespace_worker_env(
|
||||
profile_name=profile_name,
|
||||
client_type=client,
|
||||
client_instance_id=client_instance_id,
|
||||
config_path=config_path or gitea_config.DEFAULT_CONFIG_PATH,
|
||||
fleet_run_id=fleet_run,
|
||||
session_id=session,
|
||||
)
|
||||
return {
|
||||
server_key: {
|
||||
"command": command,
|
||||
"args": args,
|
||||
"env": env,
|
||||
},
|
||||
"client_instance_id": client_instance_id,
|
||||
"client_type": client,
|
||||
}
|
||||
|
||||
|
||||
def collect_instance_ids_from_mcp_servers(
|
||||
mcp_servers: Mapping[str, Any],
|
||||
*,
|
||||
namespaces: "Sequence[str] | None" = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Inspect a production mcpServers map for shared instance attribution.
|
||||
|
||||
Returns the unique set of client_instance_id values across the gitea-*
|
||||
servers of one launch and whether every worker of that launch shares
|
||||
exactly one trusted ID.
|
||||
|
||||
The inspected set is the launch's own namespaces, not an assumed five
|
||||
(#985): a project-scoped author/reviewer/merger launch must be able to
|
||||
prove shared attribution without a controller or reconciler entry counting
|
||||
as a missing worker. Pass *namespaces* to assert an exact expected subset;
|
||||
omit it to inspect whichever sanctioned namespace servers are present.
|
||||
"""
|
||||
if namespaces is None:
|
||||
expected_servers = [
|
||||
name for name in NAMESPACE_SERVER_NAMES if name in mcp_servers
|
||||
]
|
||||
else:
|
||||
expected_servers = [
|
||||
f"gitea-{ns}" for ns in resolve_launch_namespaces(namespaces)
|
||||
]
|
||||
|
||||
ids: list[str] = []
|
||||
by_server: dict[str, str | None] = {}
|
||||
for name in expected_servers:
|
||||
entry = mcp_servers.get(name) or {}
|
||||
env = entry.get("env") or {}
|
||||
raw = (env.get(CLIENT_INSTANCE_ENV) or "").strip() or None
|
||||
by_server[name] = raw
|
||||
if raw:
|
||||
ids.append(raw)
|
||||
unique = sorted(set(ids))
|
||||
trusted = [
|
||||
i
|
||||
for i in unique
|
||||
if fleet.assess_instance_identity(i)["trusted"]
|
||||
]
|
||||
missing_servers = [n for n in expected_servers if n not in mcp_servers]
|
||||
return {
|
||||
"server_instance_ids": by_server,
|
||||
"unique_instance_ids": unique,
|
||||
"trusted_instance_ids": trusted,
|
||||
"inspected_servers": list(expected_servers),
|
||||
"missing_servers": missing_servers,
|
||||
"shared_single_trusted_id": (
|
||||
bool(expected_servers)
|
||||
and not missing_servers
|
||||
and len(unique) == 1
|
||||
and len(trusted) == 1
|
||||
and all(by_server.get(n) == unique[0] for n in expected_servers)
|
||||
),
|
||||
"namespace_server_count": sum(
|
||||
1 for n in NAMESPACE_SERVER_NAMES if n in mcp_servers
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def inherit_or_refuse_client_instance(
|
||||
env: Mapping[str, str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Resolve instance identity for a worker process at serve time.
|
||||
|
||||
Production workers inherit the launcher-issued ID. They never mint a
|
||||
trusted ID themselves. Missing / legacy / malformed values fail soft into
|
||||
an untrusted assessment so registration can still record a diagnostic row
|
||||
without authorizing multi-instance fleet mutation.
|
||||
"""
|
||||
source = dict(env if env is not None else os.environ)
|
||||
raw = (source.get(CLIENT_INSTANCE_ENV) or "").strip() or None
|
||||
provenance_marker = (source.get(INSTANCE_PROVENANCE_ENV) or "").strip()
|
||||
assessment = dict(fleet.assess_instance_identity(raw))
|
||||
# The ``inst-…`` format is public and reproducible, so format alone cannot
|
||||
# establish trust: anyone able to set one env var could otherwise hand-write
|
||||
# a well-formed ID and be believed. Trust therefore requires the launcher's
|
||||
# own seal as well, and a well-formed but unsealed identity fails closed
|
||||
# rather than degrading to a mere annotation (#985).
|
||||
if assessment["trusted"] and provenance_marker != LAUNCHER_PROVENANCE_VALUE:
|
||||
assessment["launcher_sealed"] = False
|
||||
assessment["trusted"] = False
|
||||
assessment["complete"] = False
|
||||
assessment["provenance"] = fleet.INSTANCE_ID_PROVENANCE_UNSEALED
|
||||
assessment["reasons"] = list(assessment.get("reasons") or []) + [
|
||||
f"{INSTANCE_PROVENANCE_ENV} is not {LAUNCHER_PROVENANCE_VALUE!r}; "
|
||||
"identity format is valid but not sealed by the production "
|
||||
"launcher, so trusted attribution is refused"
|
||||
]
|
||||
else:
|
||||
assessment["launcher_sealed"] = bool(
|
||||
assessment["trusted"]
|
||||
and provenance_marker == LAUNCHER_PROVENANCE_VALUE
|
||||
)
|
||||
assessment["fleet_run_id"] = (source.get(FLEET_RUN_ENV) or "").strip() or None
|
||||
assessment["session_id"] = (
|
||||
(source.get(CLIENT_SESSION_ENV) or "").strip() or None
|
||||
)
|
||||
assessment["client_type"] = mwi.normalize_client_name(
|
||||
(source.get("GITEA_MCP_CLIENT") or "").strip() or None
|
||||
)
|
||||
return assessment
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Runnable launch path (#985)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
#: How each supported LLM client is started with a per-launch MCP config.
|
||||
#:
|
||||
#: The launch interface is deliberately a data-driven registry rather than a
|
||||
#: Claude-specific code path: adding another supported client is one entry, and
|
||||
#: every client necessarily goes through the same mint-once/propagate-to-all
|
||||
#: identity mechanism because the argv builder only ever receives a config file
|
||||
#: this module wrote.
|
||||
CLIENT_LAUNCH_SPECS: dict[str, dict[str, Any]] = {
|
||||
"claude_code": {
|
||||
"command": "claude",
|
||||
"config_args": lambda path: ["--mcp-config", path, "--strict-mcp-config"],
|
||||
"docs": "claude --mcp-config <per-launch.json> --strict-mcp-config",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def supported_launch_clients() -> tuple[str, ...]:
|
||||
"""Client types with a sanctioned runnable launch path."""
|
||||
return tuple(sorted(CLIENT_LAUNCH_SPECS))
|
||||
|
||||
|
||||
def build_client_launch_argv(
|
||||
client_type: str,
|
||||
config_path: str,
|
||||
*,
|
||||
extra_args: "Sequence[str] | None" = None,
|
||||
) -> list[str]:
|
||||
"""Build the argv that starts *client_type* against a per-launch config."""
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
spec = CLIENT_LAUNCH_SPECS.get(client)
|
||||
if spec is None:
|
||||
raise ValueError(
|
||||
f"no sanctioned runnable launch path for client {client!r}; "
|
||||
f"supported clients: {list(supported_launch_clients())}"
|
||||
)
|
||||
argv = [str(spec["command"])] + list(spec["config_args"](str(config_path)))
|
||||
if extra_args:
|
||||
argv.extend(str(a) for a in extra_args)
|
||||
return argv
|
||||
|
||||
|
||||
def write_launch_config(
|
||||
mcp_servers: Mapping[str, Any],
|
||||
*,
|
||||
directory: str | None = None,
|
||||
) -> str:
|
||||
"""Write one launch's ``mcpServers`` config to a fresh per-launch file.
|
||||
|
||||
The file is per-launch and owner-readable only. #985 explicitly rules out
|
||||
persisting an instance ID into a shared, reused ``.mcp.json``: two
|
||||
concurrent sessions reading one static trusted ID is precisely the reuse
|
||||
case the duplicate guard must reject. A distinct file per launch keeps the
|
||||
minted identity bound to the launch that minted it.
|
||||
"""
|
||||
import json
|
||||
import tempfile
|
||||
|
||||
fd, path = tempfile.mkstemp(
|
||||
prefix="gitea-mcp-launch-", suffix=".json", dir=directory
|
||||
)
|
||||
try:
|
||||
os.fchmod(fd, 0o600)
|
||||
with os.fdopen(fd, "w", encoding="utf-8") as handle:
|
||||
json.dump({"mcpServers": dict(mcp_servers)}, handle, indent=2)
|
||||
handle.write("\n")
|
||||
except Exception:
|
||||
try:
|
||||
os.unlink(path)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
return path
|
||||
|
||||
|
||||
def prepare_application_launch(
|
||||
profile_by_namespace: Mapping[str, str],
|
||||
*,
|
||||
client_type: str,
|
||||
namespaces: "Sequence[str] | None" = None,
|
||||
config_path: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
launch_nonce: str | None = None,
|
||||
extra_args: "Sequence[str] | None" = None,
|
||||
config_directory: str | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Mint one launch and produce everything needed to exec the client.
|
||||
|
||||
Returns the built servers, the per-launch config file path, the argv to
|
||||
exec, and the shared trusted identity — without starting anything, so the
|
||||
same preparation is testable and inspectable via ``--dry-run``.
|
||||
"""
|
||||
built = build_application_mcp_servers(
|
||||
profile_by_namespace,
|
||||
client_type=client_type,
|
||||
namespaces=namespaces,
|
||||
config_path=config_path,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
launch_nonce=launch_nonce,
|
||||
now=now,
|
||||
)
|
||||
# Fail before writing anything if this client has no runnable path.
|
||||
argv_probe = build_client_launch_argv(
|
||||
client_type, "<pending>", extra_args=extra_args
|
||||
)
|
||||
launch_config_path = write_launch_config(
|
||||
built["mcpServers"], directory=config_directory
|
||||
)
|
||||
argv = build_client_launch_argv(
|
||||
client_type, launch_config_path, extra_args=extra_args
|
||||
)
|
||||
proof = collect_instance_ids_from_mcp_servers(
|
||||
built["mcpServers"], namespaces=built["namespaces"]
|
||||
)
|
||||
return {
|
||||
"client_type": built["client_type"],
|
||||
"client_instance_id": built["client_instance_id"],
|
||||
"namespaces": built["namespaces"],
|
||||
"excluded_namespaces": built["excluded_namespaces"],
|
||||
"project_scoped": built["project_scoped"],
|
||||
"launch_config_path": launch_config_path,
|
||||
"argv": argv,
|
||||
"argv_template": argv_probe,
|
||||
"mcpServers": built["mcpServers"],
|
||||
"launch": built["launch"],
|
||||
"shared_single_trusted_id": proof["shared_single_trusted_id"],
|
||||
"attribution_proof": proof,
|
||||
}
|
||||
|
||||
|
||||
def _parse_profile_assignments(values: "Sequence[str]") -> dict[str, str]:
|
||||
"""Parse ``namespace=profile`` CLI pairs into a mapping."""
|
||||
profiles: dict[str, str] = {}
|
||||
for raw in values or ():
|
||||
text = str(raw)
|
||||
if "=" not in text:
|
||||
raise ValueError(
|
||||
f"invalid --profile {text!r}; expected namespace=profile-name"
|
||||
)
|
||||
namespace, _, profile = text.partition("=")
|
||||
namespace = namespace.strip().lower()
|
||||
profile = profile.strip()
|
||||
if not namespace or not profile:
|
||||
raise ValueError(
|
||||
f"invalid --profile {text!r}; expected namespace=profile-name"
|
||||
)
|
||||
if namespace in profiles:
|
||||
raise ValueError(f"duplicate --profile entry for namespace {namespace!r}")
|
||||
profiles[namespace] = profile
|
||||
return profiles
|
||||
|
||||
|
||||
def main(argv: "Sequence[str] | None" = None) -> int:
|
||||
"""Runnable entry point: start a supported LLM client for one launch.
|
||||
|
||||
Example (project-scoped, three namespaces)::
|
||||
|
||||
python3 -m mcp_application_launcher \\
|
||||
--client claude_code \\
|
||||
--namespaces author,reviewer,merger \\
|
||||
--profile author=prgs-author \\
|
||||
--profile reviewer=prgs-reviewer \\
|
||||
--profile merger=prgs-merger
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="mcp_application_launcher",
|
||||
description=(
|
||||
"Start a supported LLM client with one trusted, per-launch "
|
||||
"GITEA_MCP_CLIENT_INSTANCE shared by every MCP namespace worker."
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--client",
|
||||
default="claude_code",
|
||||
help=f"client to launch; supported: {list(supported_launch_clients())}",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--namespaces",
|
||||
default=None,
|
||||
help=(
|
||||
"comma-separated sanctioned namespace subset (e.g. "
|
||||
"'author,reviewer,merger'); omit to launch all five"
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--profile",
|
||||
action="append",
|
||||
default=[],
|
||||
metavar="NAMESPACE=PROFILE",
|
||||
help="profile for one namespace; repeat once per launched namespace",
|
||||
)
|
||||
parser.add_argument("--config-path", default=None, help="profiles.json path")
|
||||
parser.add_argument("--fleet-run-id", default=None)
|
||||
parser.add_argument("--session-id", default=None)
|
||||
parser.add_argument(
|
||||
"--dry-run",
|
||||
action="store_true",
|
||||
help="print the launch plan as JSON and exit without starting anything",
|
||||
)
|
||||
parser.add_argument(
|
||||
"client_args",
|
||||
nargs=argparse.REMAINDER,
|
||||
help="arguments forwarded to the client after '--'",
|
||||
)
|
||||
ns = parser.parse_args(list(argv) if argv is not None else None)
|
||||
|
||||
try:
|
||||
profiles = _parse_profile_assignments(ns.profile)
|
||||
requested = (
|
||||
[part.strip() for part in ns.namespaces.split(",") if part.strip()]
|
||||
if ns.namespaces
|
||||
else None
|
||||
)
|
||||
resolved = resolve_launch_namespaces(requested)
|
||||
# Profiles for namespaces that are not being launched are almost always
|
||||
# a mistake (a stale five-namespace invocation), so refuse rather than
|
||||
# silently ignore them.
|
||||
stray = sorted(set(profiles) - set(resolved))
|
||||
if stray:
|
||||
raise ValueError(
|
||||
f"--profile given for namespace(s) {stray} that this launch "
|
||||
f"does not start; launched namespaces are {list(resolved)}"
|
||||
)
|
||||
forwarded = [a for a in (ns.client_args or []) if a != "--"]
|
||||
prepared = prepare_application_launch(
|
||||
profiles,
|
||||
client_type=ns.client,
|
||||
namespaces=requested,
|
||||
config_path=ns.config_path,
|
||||
fleet_run_id=ns.fleet_run_id,
|
||||
session_id=ns.session_id,
|
||||
extra_args=forwarded,
|
||||
)
|
||||
except ValueError as exc:
|
||||
parser.error(str(exc))
|
||||
return 2 # pragma: no cover - argparse.error raises SystemExit
|
||||
|
||||
if ns.dry_run:
|
||||
plan = {
|
||||
key: prepared[key]
|
||||
for key in (
|
||||
"client_type",
|
||||
"client_instance_id",
|
||||
"namespaces",
|
||||
"excluded_namespaces",
|
||||
"project_scoped",
|
||||
"launch_config_path",
|
||||
"argv",
|
||||
"shared_single_trusted_id",
|
||||
)
|
||||
}
|
||||
print(json.dumps(plan, indent=2))
|
||||
return 0
|
||||
|
||||
argv_to_exec = prepared["argv"]
|
||||
os.execvp(argv_to_exec[0], argv_to_exec)
|
||||
return 0 # pragma: no cover - execvp does not return
|
||||
|
||||
|
||||
if __name__ == "__main__": # pragma: no cover - process entry point
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,873 @@
|
||||
"""Instance-level fleet identity and health snapshots (#978).
|
||||
|
||||
#948 established per-worker ownership; #975 made heartbeats keep those rows
|
||||
live. Neither surface could enumerate the fleet at *instance* granularity:
|
||||
which application launch owns which five namespace workers, whether two
|
||||
Codex launches are distinct, or whether a live collision is real rather than
|
||||
a shared client type.
|
||||
|
||||
This module is pure. Callers supply registry rows (and optional enrichments);
|
||||
nothing here opens SQLite, scans process tables, or mutates state. Production
|
||||
evidence for the fleet gate is the snapshot returned by the sanctioned
|
||||
controller/reconciler tool that wraps this assessor.
|
||||
|
||||
Identity hierarchy (highest → lowest):
|
||||
|
||||
* ``client_type`` — application family (``codex``, ``claude_code``, …)
|
||||
* ``client_instance_id`` — one running application launch (trusted launcher)
|
||||
* ``fleet_run_id`` — operator-approved enrollment / canary cohort
|
||||
* ``worker_id`` / ``worker_identity`` — one namespace worker process
|
||||
* ``namespace`` — author | reviewer | merger | controller | reconciler
|
||||
|
||||
Multiple simultaneous instances of the same ``client_type`` are first-class.
|
||||
Sharing only a profile or client type is never a duplicate.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import secrets
|
||||
from collections import defaultdict
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable, Iterable, Mapping
|
||||
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
# --- Classification labels ------------------------------------------------
|
||||
|
||||
CLASS_EXPECTED = "expected_enrolled"
|
||||
CLASS_MISSING = "missing_expected"
|
||||
CLASS_UNMANIFESTED = "unmanifested"
|
||||
CLASS_DUPLICATE_NAMESPACE = "duplicate_namespace_worker"
|
||||
CLASS_INSTANCE_ID_COLLISION = "instance_id_collision"
|
||||
CLASS_WORKER_ID_COLLISION = "worker_identity_collision"
|
||||
CLASS_SESSION_COLLISION = "session_identity_collision"
|
||||
CLASS_GENERATION_COLLISION = "generation_identity_collision"
|
||||
CLASS_PROCESS_COLLISION = "process_identity_collision"
|
||||
CLASS_PID_COLLISION = "pid_collision"
|
||||
CLASS_OWNERSHIP_COLLISION = "ownership_fencing_collision"
|
||||
CLASS_ORPHANED = "orphaned_unowned"
|
||||
CLASS_UNKNOWN_CLIENT = "unknown_client"
|
||||
CLASS_FOREIGN_REPOSITORY = "foreign_repository"
|
||||
CLASS_OLD_REVISION = "old_revision"
|
||||
CLASS_STALE_WORKER = "stale_orphaned_worker"
|
||||
CLASS_LEGACY_INCOMPLETE = "legacy_incomplete_identity"
|
||||
CLASS_HISTORICAL = "historical_dead"
|
||||
CLASS_HEALTHY = "healthy"
|
||||
|
||||
#: Active blockers that make the live fleet unsafe for mutation-gated work.
|
||||
ACTIVE_BLOCKER_CLASSES = frozenset(
|
||||
{
|
||||
CLASS_MISSING,
|
||||
CLASS_UNMANIFESTED,
|
||||
CLASS_DUPLICATE_NAMESPACE,
|
||||
CLASS_INSTANCE_ID_COLLISION,
|
||||
CLASS_WORKER_ID_COLLISION,
|
||||
CLASS_SESSION_COLLISION,
|
||||
CLASS_GENERATION_COLLISION,
|
||||
CLASS_PROCESS_COLLISION,
|
||||
CLASS_PID_COLLISION,
|
||||
CLASS_OWNERSHIP_COLLISION,
|
||||
CLASS_ORPHANED,
|
||||
CLASS_UNKNOWN_CLIENT,
|
||||
CLASS_FOREIGN_REPOSITORY,
|
||||
CLASS_OLD_REVISION,
|
||||
CLASS_STALE_WORKER,
|
||||
CLASS_LEGACY_INCOMPLETE,
|
||||
}
|
||||
)
|
||||
|
||||
SANCTIONED_NAMESPACES = frozenset(
|
||||
{"author", "reviewer", "merger", "controller", "reconciler"}
|
||||
)
|
||||
|
||||
INSTANCE_ID_PROVENANCE_TRUSTED = "trusted_launcher"
|
||||
INSTANCE_ID_PROVENANCE_LEGACY = "legacy_incomplete"
|
||||
INSTANCE_ID_PROVENANCE_MISSING = "missing"
|
||||
#: Well-formed ``inst-…`` identity presented without the launcher's own
|
||||
#: provenance seal (#985). The format alone is public and reproducible, so an
|
||||
#: unsealed value is treated as manually asserted trust and fails closed.
|
||||
INSTANCE_ID_PROVENANCE_UNSEALED = "unsealed_launcher"
|
||||
|
||||
CLIENT_INSTANCE_ENV = "GITEA_MCP_CLIENT_INSTANCE"
|
||||
FLEET_RUN_ENV = "GITEA_MCP_FLEET_RUN_ID"
|
||||
PROCESS_IDENTITY_ENV = "GITEA_MCP_PROCESS_IDENTITY"
|
||||
INSTANCE_PROVENANCE_ENV = "GITEA_MCP_INSTANCE_PROVENANCE"
|
||||
|
||||
_LEGACY_INSTANCE_PREFIXES = ("pid-", "proc-", "legacy-")
|
||||
#: Trusted launcher-issued IDs use the reserved ``inst-`` prefix
|
||||
#: (see :func:`generate_client_instance_id`). Ordinary user-supplied strings
|
||||
#: without that prefix never count as trusted attribution.
|
||||
_TRUSTED_INSTANCE_PREFIX = "inst-"
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _ts(value: datetime) -> str:
|
||||
return value.astimezone(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
|
||||
|
||||
def generate_client_instance_id(
|
||||
client_type: str | None,
|
||||
*,
|
||||
launch_nonce: str | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> str:
|
||||
"""Mint a distinct instance ID for one application launch (#978).
|
||||
|
||||
The trusted launcher (or host that starts all five namespace workers)
|
||||
generates this once per launch and injects it as ``GITEA_MCP_CLIENT_INSTANCE``
|
||||
into every worker environment. Workers never invent their own instance ID
|
||||
from PID proximity or timestamps.
|
||||
"""
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
stamp = (now or _utc_now()).astimezone(timezone.utc).strftime(
|
||||
mwi.IDENTITY_TIMESTAMP_FORMAT
|
||||
)
|
||||
nonce = launch_nonce if launch_nonce is not None else secrets.token_hex(16)
|
||||
digest = hashlib.sha256(
|
||||
f"{client}\x1f{stamp}\x1f{nonce}".encode("utf-8")
|
||||
).hexdigest()[:12]
|
||||
return f"inst-{client}-{stamp}-{digest}"
|
||||
|
||||
|
||||
def assess_instance_identity(
|
||||
raw_instance_id: str | None,
|
||||
*,
|
||||
source: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Classify whether a client_instance_id is trusted enough for mutation.
|
||||
|
||||
Trusted instance IDs are non-empty, match the launcher-minted ``inst-…``
|
||||
format from :func:`generate_client_instance_id`, and are not pre-#978
|
||||
PID/proc/legacy placeholders. Ordinary user-supplied or malformed values
|
||||
fail closed as untrusted so they cannot spoof multi-instance attribution.
|
||||
Incomplete identities remain visible for diagnosis but cannot authorize
|
||||
unsafe mutation.
|
||||
"""
|
||||
text = (raw_instance_id or "").strip()
|
||||
if not text:
|
||||
return {
|
||||
"client_instance_id": None,
|
||||
"complete": False,
|
||||
"trusted": False,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_MISSING,
|
||||
"reasons": [
|
||||
"client_instance_id is missing; the trusted launcher must set "
|
||||
f"{CLIENT_INSTANCE_ENV} once per application launch"
|
||||
],
|
||||
}
|
||||
lowered = text.lower()
|
||||
if lowered.startswith(_LEGACY_INSTANCE_PREFIXES) or source == "pid_fallback":
|
||||
return {
|
||||
"client_instance_id": text,
|
||||
"complete": False,
|
||||
"trusted": False,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_LEGACY,
|
||||
"reasons": [
|
||||
f"client_instance_id {text!r} is a legacy PID/process fallback, "
|
||||
"not a trusted launcher-issued instance identity"
|
||||
],
|
||||
}
|
||||
if not text.startswith(_TRUSTED_INSTANCE_PREFIX) or len(text) <= len(
|
||||
_TRUSTED_INSTANCE_PREFIX
|
||||
):
|
||||
return {
|
||||
"client_instance_id": text,
|
||||
"complete": False,
|
||||
"trusted": False,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_LEGACY,
|
||||
"reasons": [
|
||||
f"client_instance_id {text!r} is malformed or user-supplied and "
|
||||
f"does not use the trusted launcher prefix "
|
||||
f"{_TRUSTED_INSTANCE_PREFIX!r}; refusing trusted attribution"
|
||||
],
|
||||
}
|
||||
return {
|
||||
"client_instance_id": text,
|
||||
"complete": True,
|
||||
"trusted": True,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_TRUSTED,
|
||||
"reasons": [],
|
||||
}
|
||||
|
||||
|
||||
def resolve_client_instance_from_env(
|
||||
env: Mapping[str, str] | None = None,
|
||||
*,
|
||||
pid: int | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Resolve instance identity from launcher env without inventing one.
|
||||
|
||||
When the trusted key is absent, return incomplete evidence rather than a
|
||||
silent ``pid-<n>`` identity. Callers that still need a non-empty registry
|
||||
key may choose a legacy placeholder deliberately; they must not treat it as
|
||||
trusted.
|
||||
"""
|
||||
source = dict(env or {})
|
||||
raw = (source.get(CLIENT_INSTANCE_ENV) or "").strip()
|
||||
assessment = assess_instance_identity(raw or None)
|
||||
assessment["fleet_run_id"] = (source.get(FLEET_RUN_ENV) or "").strip() or None
|
||||
assessment["process_identity"] = (
|
||||
(source.get(PROCESS_IDENTITY_ENV) or "").strip()
|
||||
or (f"pid-{pid}" if pid is not None else None)
|
||||
)
|
||||
return assessment
|
||||
|
||||
|
||||
def _public_worker(record: Mapping[str, Any]) -> dict[str, Any]:
|
||||
return {
|
||||
"worker_id": record.get("worker_identity") or record.get("worker_id"),
|
||||
"worker_identity": record.get("worker_identity") or record.get("worker_id"),
|
||||
"client_type": mwi.normalize_client_name(
|
||||
record.get("client_name") or record.get("client_type")
|
||||
),
|
||||
"client_instance_id": record.get("client_instance_id"),
|
||||
"fleet_run_id": record.get("fleet_run_id"),
|
||||
"namespace": record.get("namespace"),
|
||||
"profile": record.get("profile"),
|
||||
"declared_role": record.get("role") or record.get("declared_role"),
|
||||
"authenticated_account": record.get("authenticated_account"),
|
||||
"session_id": record.get("session_id"),
|
||||
"generation_id": record.get("generation_id"),
|
||||
"process_identity": record.get("process_identity")
|
||||
or (
|
||||
f"pid-{record['pid']}"
|
||||
if record.get("pid") is not None
|
||||
else None
|
||||
),
|
||||
"pid": record.get("pid"),
|
||||
"repository_binding": record.get("repository_binding"),
|
||||
"remote": record.get("remote"),
|
||||
"startup_revision": record.get("startup_revision"),
|
||||
"loaded_revision": record.get("loaded_revision"),
|
||||
"parity_revision": record.get("parity_revision"),
|
||||
"live_revision": record.get("live_revision"),
|
||||
"runtime_provenance": record.get("runtime_provenance")
|
||||
or record.get("transport"),
|
||||
"transport": record.get("transport"),
|
||||
"started_at": record.get("started_at"),
|
||||
"last_heartbeat_at": record.get("last_heartbeat_at"),
|
||||
"heartbeat_ttl_seconds": record.get("heartbeat_ttl_seconds"),
|
||||
"fencing_epoch": record.get("fencing_epoch"),
|
||||
"status": record.get("status"),
|
||||
"instance_id_provenance": record.get("instance_id_provenance"),
|
||||
}
|
||||
|
||||
|
||||
def _liveness(
|
||||
record: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime | None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None,
|
||||
) -> dict[str, Any]:
|
||||
pid_alive = None
|
||||
if pid_alive_probe is not None and record.get("pid") is not None:
|
||||
try:
|
||||
pid_alive = pid_alive_probe(record.get("pid"))
|
||||
except Exception:
|
||||
pid_alive = None
|
||||
return mwi.WorkerRegistry.is_live(record, now=now, pid_alive=pid_alive)
|
||||
|
||||
|
||||
def _consistency_token(rows: Iterable[Mapping[str, Any]], snapshot_at: str) -> str:
|
||||
material = [snapshot_at]
|
||||
for row in sorted(
|
||||
rows,
|
||||
key=lambda r: (
|
||||
str(r.get("worker_identity") or ""),
|
||||
str(r.get("last_heartbeat_at") or ""),
|
||||
str(r.get("fencing_epoch") or ""),
|
||||
),
|
||||
):
|
||||
material.append(
|
||||
"|".join(
|
||||
[
|
||||
str(row.get("worker_identity") or ""),
|
||||
str(row.get("client_instance_id") or ""),
|
||||
str(row.get("status") or ""),
|
||||
str(row.get("last_heartbeat_at") or ""),
|
||||
str(row.get("fencing_epoch") or ""),
|
||||
str(row.get("generation_id") or ""),
|
||||
]
|
||||
)
|
||||
)
|
||||
digest = hashlib.sha256("\n".join(material).encode("utf-8")).hexdigest()[:16]
|
||||
return f"fleetrev-{digest}"
|
||||
|
||||
|
||||
def build_worker_snapshot_row(
|
||||
record: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None = None,
|
||||
canonical_repository: str | None = None,
|
||||
expected_live_revision: str | None = None,
|
||||
heartbeat_supervised: bool | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""One point-in-time worker row for the fleet snapshot."""
|
||||
stamp = now or _utc_now()
|
||||
base = _public_worker(record)
|
||||
identity = assess_instance_identity(
|
||||
base.get("client_instance_id"),
|
||||
source=record.get("instance_id_source"),
|
||||
)
|
||||
liveness = _liveness(record, now=stamp, pid_alive_probe=pid_alive_probe)
|
||||
is_historical = str(record.get("status") or "") != mwi.STATUS_ACTIVE
|
||||
live = bool(liveness.get("live")) and not is_historical
|
||||
|
||||
repo = (base.get("repository_binding") or "").strip() or None
|
||||
foreign_repo = bool(
|
||||
canonical_repository
|
||||
and repo
|
||||
and repo.rstrip("/") != str(canonical_repository).rstrip("/")
|
||||
)
|
||||
old_revision = False
|
||||
if expected_live_revision:
|
||||
for key in ("startup_revision", "loaded_revision", "parity_revision", "live_revision"):
|
||||
rev = (base.get(key) or "").strip()
|
||||
if rev and rev != expected_live_revision:
|
||||
old_revision = True
|
||||
break
|
||||
|
||||
ownership_state = "historical" if is_historical else (
|
||||
"live" if live else "stale"
|
||||
)
|
||||
if live and not identity["trusted"]:
|
||||
ownership_state = "live_untrusted_identity"
|
||||
if live and not base.get("session_id"):
|
||||
ownership_state = "orphaned"
|
||||
|
||||
mutation_safe = bool(
|
||||
live
|
||||
and identity["trusted"]
|
||||
and not foreign_repo
|
||||
and not old_revision
|
||||
and ownership_state == "live"
|
||||
and base.get("client_type") != mwi.UNKNOWN_CLIENT
|
||||
)
|
||||
|
||||
restart_required = bool(
|
||||
old_revision
|
||||
or (live and not liveness.get("heartbeat_fresh", True))
|
||||
)
|
||||
|
||||
return {
|
||||
**base,
|
||||
"instance_identity": identity,
|
||||
"client_instance_id": identity["client_instance_id"] or base.get("client_instance_id"),
|
||||
"instance_id_provenance": identity["provenance"],
|
||||
"instance_identity_trusted": identity["trusted"],
|
||||
"live": live,
|
||||
"historical": is_historical,
|
||||
"liveness": liveness,
|
||||
"heartbeat": {
|
||||
"registered": bool(base.get("last_heartbeat_at")),
|
||||
"supervised": heartbeat_supervised,
|
||||
"age_seconds": liveness.get("heartbeat_age_seconds"),
|
||||
"ttl_seconds": liveness.get("heartbeat_ttl_seconds"),
|
||||
"fresh": liveness.get("heartbeat_fresh"),
|
||||
"last_heartbeat_at": base.get("last_heartbeat_at"),
|
||||
},
|
||||
"fencing": {
|
||||
"fencing_epoch": base.get("fencing_epoch"),
|
||||
"generation_id": base.get("generation_id"),
|
||||
},
|
||||
"ownership_state": ownership_state,
|
||||
"foreign_repository": foreign_repo,
|
||||
"old_revision": old_revision,
|
||||
"stale": not live and not is_historical,
|
||||
"restart_required": restart_required,
|
||||
"mutation_safe": mutation_safe,
|
||||
"conflicting_live_sessions": [],
|
||||
}
|
||||
|
||||
|
||||
def _collision_groups(
|
||||
live_rows: list[dict[str, Any]],
|
||||
key_fn,
|
||||
) -> dict[str, list[dict[str, Any]]]:
|
||||
groups: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||
for row in live_rows:
|
||||
key = key_fn(row)
|
||||
if key is None or key == "" or key == "None":
|
||||
continue
|
||||
groups[str(key)].append(row)
|
||||
return {k: v for k, v in groups.items() if len(v) > 1}
|
||||
|
||||
|
||||
def snapshot_instance_fleet(
|
||||
records: Iterable[Mapping[str, Any]],
|
||||
*,
|
||||
expected_manifest: list[Mapping[str, Any]] | None = None,
|
||||
now: datetime | None = None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None = None,
|
||||
canonical_repository: str | None = None,
|
||||
expected_live_revision: str | None = None,
|
||||
registry_revision: str | None = None,
|
||||
known_client_types: Iterable[str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Authoritative point-in-time fleet snapshot with classification (#978).
|
||||
|
||||
Historical dead rows are reported separately and never automatically make
|
||||
the live fleet unsafe.
|
||||
"""
|
||||
stamp = now or _utc_now()
|
||||
snapshot_at = _ts(stamp)
|
||||
known = {
|
||||
mwi.normalize_client_name(c)
|
||||
for c in (known_client_types or mwi.CLIENT_ALIASES.values())
|
||||
}
|
||||
known.discard(mwi.UNKNOWN_CLIENT)
|
||||
|
||||
all_rows: list[dict[str, Any]] = []
|
||||
for record in records:
|
||||
all_rows.append(
|
||||
build_worker_snapshot_row(
|
||||
record,
|
||||
now=stamp,
|
||||
pid_alive_probe=pid_alive_probe,
|
||||
canonical_repository=canonical_repository,
|
||||
expected_live_revision=expected_live_revision,
|
||||
)
|
||||
)
|
||||
|
||||
live_rows = [r for r in all_rows if r["live"]]
|
||||
historical_rows = [r for r in all_rows if r["historical"]]
|
||||
stale_rows = [r for r in all_rows if r["stale"]]
|
||||
|
||||
# --- identity collisions among live workers ---
|
||||
findings: list[dict[str, Any]] = []
|
||||
|
||||
def _finding(
|
||||
classification: str,
|
||||
*,
|
||||
severity: str,
|
||||
workers: list[dict[str, Any]] | None = None,
|
||||
instance_ids: list[str] | None = None,
|
||||
detail: str,
|
||||
active_blocker: bool,
|
||||
) -> None:
|
||||
findings.append(
|
||||
{
|
||||
"classification": classification,
|
||||
"severity": severity,
|
||||
"active_blocker": active_blocker,
|
||||
"detail": detail,
|
||||
"client_instance_ids": instance_ids or sorted(
|
||||
{
|
||||
str(w.get("client_instance_id"))
|
||||
for w in (workers or [])
|
||||
if w.get("client_instance_id")
|
||||
}
|
||||
),
|
||||
"worker_identities": [
|
||||
w.get("worker_identity") for w in (workers or [])
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
# Duplicate worker identity (should not happen with PK, still detect)
|
||||
for wid, group in _collision_groups(
|
||||
live_rows, lambda r: r.get("worker_identity")
|
||||
).items():
|
||||
_finding(
|
||||
CLASS_WORKER_ID_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"worker identity {wid!r} is claimed by {len(group)} live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Reused session identity across live workers
|
||||
for sid, group in _collision_groups(live_rows, lambda r: r.get("session_id")).items():
|
||||
# Same session may appear once; collision only when multiple workers share it
|
||||
# across different worker identities (always true for group size > 1).
|
||||
_finding(
|
||||
CLASS_SESSION_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"session identity {sid!r} is reused by {len(group)} live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Generation claimed by multiple live sessions/workers is a conflict when
|
||||
# the workers are not the five sanctioned namespaces of one instance.
|
||||
for gen, group in _collision_groups(
|
||||
live_rows, lambda r: r.get("generation_id")
|
||||
).items():
|
||||
namespaces = {g.get("namespace") for g in group if g.get("namespace")}
|
||||
instance_ids = {g.get("client_instance_id") for g in group}
|
||||
# Multiple workers under one generation is only valid if they share one
|
||||
# instance and distinct namespaces. Same generation + same namespace = bad.
|
||||
by_ns: dict[str, list] = defaultdict(list)
|
||||
for g in group:
|
||||
by_ns[str(g.get("namespace") or "")].append(g)
|
||||
ns_dups = {ns: rows for ns, rows in by_ns.items() if ns and len(rows) > 1}
|
||||
if ns_dups or len(instance_ids) > 1:
|
||||
_finding(
|
||||
CLASS_GENERATION_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=(
|
||||
f"generation {gen!r} is contested across namespaces/instances "
|
||||
f"(namespaces={sorted(namespaces)}, "
|
||||
f"instances={sorted(str(i) for i in instance_ids if i)})"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Process identity / PID collisions across distinct workers
|
||||
for proc, group in _collision_groups(
|
||||
live_rows, lambda r: r.get("process_identity")
|
||||
).items():
|
||||
if len({r.get("worker_identity") for r in group}) > 1:
|
||||
_finding(
|
||||
CLASS_PROCESS_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"process identity {proc!r} is shared by distinct live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
for pid, group in _collision_groups(live_rows, lambda r: r.get("pid")).items():
|
||||
if len({r.get("worker_identity") for r in group}) > 1:
|
||||
_finding(
|
||||
CLASS_PID_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"PID {pid} is shared by distinct live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Fencing/ownership: same fencing epoch on different workers of different instances
|
||||
for epoch, group in _collision_groups(
|
||||
live_rows,
|
||||
lambda r: (
|
||||
f"{r.get('generation_id')}:{r.get('fencing_epoch')}"
|
||||
if r.get("generation_id") is not None and r.get("fencing_epoch") is not None
|
||||
else None
|
||||
),
|
||||
).items():
|
||||
if len({r.get("client_instance_id") for r in group}) > 1:
|
||||
_finding(
|
||||
CLASS_OWNERSHIP_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=(
|
||||
f"fencing token {epoch!r} spans more than one client_instance_id"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Per-instance grouping
|
||||
by_instance: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||
unkeyed_live: list[dict[str, Any]] = []
|
||||
for row in live_rows:
|
||||
iid = row.get("client_instance_id")
|
||||
if not iid:
|
||||
unkeyed_live.append(row)
|
||||
continue
|
||||
by_instance[str(iid)].append(row)
|
||||
|
||||
instances: list[dict[str, Any]] = []
|
||||
for iid, workers in sorted(by_instance.items()):
|
||||
client_types = sorted({w.get("client_type") for w in workers if w.get("client_type")})
|
||||
trusted = all(w.get("instance_identity_trusted") for w in workers)
|
||||
namespaces = [w.get("namespace") for w in workers]
|
||||
ns_counts: dict[str, int] = defaultdict(int)
|
||||
for ns in namespaces:
|
||||
if ns:
|
||||
ns_counts[str(ns)] += 1
|
||||
dup_ns = sorted(ns for ns, n in ns_counts.items() if n > 1)
|
||||
if dup_ns:
|
||||
_finding(
|
||||
CLASS_DUPLICATE_NAMESPACE,
|
||||
severity="blocker",
|
||||
workers=[w for w in workers if w.get("namespace") in dup_ns],
|
||||
instance_ids=[iid],
|
||||
detail=(
|
||||
f"instance {iid!r} has more than one live worker for "
|
||||
f"namespace(s) {dup_ns}"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Live reuse of one instance ID with incompatible client types
|
||||
if len(client_types) > 1:
|
||||
_finding(
|
||||
CLASS_INSTANCE_ID_COLLISION,
|
||||
severity="blocker",
|
||||
workers=workers,
|
||||
instance_ids=[iid],
|
||||
detail=(
|
||||
f"client_instance_id {iid!r} is live under multiple client "
|
||||
f"types {client_types}"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
if not trusted:
|
||||
_finding(
|
||||
CLASS_LEGACY_INCOMPLETE,
|
||||
severity="blocker",
|
||||
workers=workers,
|
||||
instance_ids=[iid],
|
||||
detail=(
|
||||
f"instance {iid!r} lacks trusted launcher-issued instance "
|
||||
"identity; diagnostic reads remain available"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
unknown = [w for w in workers if w.get("client_type") == mwi.UNKNOWN_CLIENT]
|
||||
if unknown:
|
||||
_finding(
|
||||
CLASS_UNKNOWN_CLIENT,
|
||||
severity="blocker",
|
||||
workers=unknown,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has worker(s) with unknown client_type",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
foreign = [w for w in workers if w.get("foreign_repository")]
|
||||
if foreign:
|
||||
_finding(
|
||||
CLASS_FOREIGN_REPOSITORY,
|
||||
severity="blocker",
|
||||
workers=foreign,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has foreign-repository workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
old = [w for w in workers if w.get("old_revision")]
|
||||
if old:
|
||||
_finding(
|
||||
CLASS_OLD_REVISION,
|
||||
severity="blocker",
|
||||
workers=old,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has old-revision workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
orphans = [w for w in workers if w.get("ownership_state") == "orphaned"]
|
||||
if orphans:
|
||||
_finding(
|
||||
CLASS_ORPHANED,
|
||||
severity="blocker",
|
||||
workers=orphans,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has orphaned/unowned workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
instances.append(
|
||||
{
|
||||
"client_instance_id": iid,
|
||||
"client_types": client_types,
|
||||
"client_type": client_types[0] if len(client_types) == 1 else None,
|
||||
"fleet_run_ids": sorted(
|
||||
{w.get("fleet_run_id") for w in workers if w.get("fleet_run_id")}
|
||||
),
|
||||
"worker_count": len(workers),
|
||||
"namespaces": sorted({n for n in namespaces if n}),
|
||||
"namespace_counts": dict(ns_counts),
|
||||
"duplicate_namespaces": dup_ns,
|
||||
"trusted_instance_identity": trusted,
|
||||
"workers": workers,
|
||||
"mutation_safe": all(w.get("mutation_safe") for w in workers)
|
||||
and not dup_ns
|
||||
and trusted,
|
||||
}
|
||||
)
|
||||
|
||||
for row in unkeyed_live:
|
||||
_finding(
|
||||
CLASS_LEGACY_INCOMPLETE,
|
||||
severity="blocker",
|
||||
workers=[row],
|
||||
detail="live worker has no client_instance_id",
|
||||
active_blocker=True,
|
||||
)
|
||||
if row.get("client_type") == mwi.UNKNOWN_CLIENT:
|
||||
_finding(
|
||||
CLASS_UNKNOWN_CLIENT,
|
||||
severity="blocker",
|
||||
workers=[row],
|
||||
detail="live worker has unknown client_type and no instance id",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
for row in stale_rows:
|
||||
_finding(
|
||||
CLASS_STALE_WORKER,
|
||||
severity="warning",
|
||||
workers=[row],
|
||||
detail=(
|
||||
f"worker {row.get('worker_identity')!r} is active in the registry "
|
||||
"but not live (stale heartbeat or dead pid)"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
for row in historical_rows:
|
||||
_finding(
|
||||
CLASS_HISTORICAL,
|
||||
severity="info",
|
||||
workers=[row],
|
||||
detail=(
|
||||
f"historical registration {row.get('worker_identity')!r} "
|
||||
f"(status={row.get('status')!r}) is not an active blocker"
|
||||
),
|
||||
active_blocker=False,
|
||||
)
|
||||
|
||||
# Manifest comparison
|
||||
expected = list(expected_manifest or [])
|
||||
expected_ids = {
|
||||
str(item.get("client_instance_id")).strip()
|
||||
for item in expected
|
||||
if (item.get("client_instance_id") or "").strip()
|
||||
}
|
||||
live_ids = set(by_instance.keys())
|
||||
missing_ids = sorted(expected_ids - live_ids)
|
||||
unmanifested_ids = sorted(live_ids - expected_ids) if expected_ids else []
|
||||
|
||||
for iid in missing_ids:
|
||||
_finding(
|
||||
CLASS_MISSING,
|
||||
severity="blocker",
|
||||
instance_ids=[iid],
|
||||
detail=f"expected enrolled instance {iid!r} is missing from the live fleet",
|
||||
active_blocker=True,
|
||||
)
|
||||
for iid in unmanifested_ids:
|
||||
_finding(
|
||||
CLASS_UNMANIFESTED,
|
||||
severity="blocker",
|
||||
instance_ids=[iid],
|
||||
workers=by_instance.get(iid, []),
|
||||
detail=(
|
||||
f"live instance {iid!r} is not on the approved fleet manifest "
|
||||
"(unmanifested)"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Same client_type multi-instance is healthy when each has distinct instance IDs
|
||||
by_type: dict[str, list[str]] = defaultdict(list)
|
||||
for inst in instances:
|
||||
for ct in inst.get("client_types") or []:
|
||||
by_type[str(ct)].append(inst["client_instance_id"])
|
||||
multi_instance_same_type = {
|
||||
ct: ids for ct, ids in by_type.items() if len(ids) > 1
|
||||
}
|
||||
|
||||
active_blockers = [f for f in findings if f.get("active_blocker")]
|
||||
historical_only = [f for f in findings if f.get("classification") == CLASS_HISTORICAL]
|
||||
live_safe = not active_blockers
|
||||
|
||||
consistency = registry_revision or _consistency_token(all_rows, snapshot_at)
|
||||
|
||||
return {
|
||||
"success": True,
|
||||
"read_only": True,
|
||||
"snapshot_at": snapshot_at,
|
||||
"consistency_token": consistency,
|
||||
"registry_revision": consistency,
|
||||
"live_worker_count": len(live_rows),
|
||||
"historical_worker_count": len(historical_rows),
|
||||
"stale_worker_count": len(stale_rows),
|
||||
"instance_count": len(instances),
|
||||
"workers": all_rows,
|
||||
"live_workers": live_rows,
|
||||
"historical_workers": historical_rows,
|
||||
"stale_workers": stale_rows,
|
||||
"instances": instances,
|
||||
"multi_instance_same_client_type": multi_instance_same_type,
|
||||
"same_client_type_not_duplicate": True,
|
||||
"expected_manifest": [
|
||||
{
|
||||
"client_instance_id": item.get("client_instance_id"),
|
||||
"client_type": item.get("client_type"),
|
||||
"fleet_run_id": item.get("fleet_run_id"),
|
||||
"namespaces": item.get("namespaces"),
|
||||
}
|
||||
for item in expected
|
||||
],
|
||||
"missing_expected_instance_ids": missing_ids,
|
||||
"unmanifested_instance_ids": unmanifested_ids,
|
||||
"findings": findings,
|
||||
"active_blockers": active_blockers,
|
||||
"historical_findings": historical_only,
|
||||
"live_fleet_safe": live_safe,
|
||||
"mutation_safe": live_safe and all(
|
||||
inst.get("mutation_safe") for inst in instances
|
||||
)
|
||||
if instances
|
||||
else live_safe,
|
||||
"classification_model": {
|
||||
"exactly_one_process_per_profile": False,
|
||||
"exactly_one_instance_per_client_type": False,
|
||||
"multiple_instances_per_client_type": True,
|
||||
"duplicate_requires": [
|
||||
"live client_instance_id collision",
|
||||
"duplicate namespace worker within one instance",
|
||||
"reused worker/session/generation/process/pid/fencing identity",
|
||||
"cross-instance ownership collision",
|
||||
"unmanifested instance when a manifest is required",
|
||||
],
|
||||
},
|
||||
"reasons": [f["detail"] for f in active_blockers],
|
||||
}
|
||||
|
||||
|
||||
def compare_snapshot_heartbeats(
|
||||
earlier: Mapping[str, Any],
|
||||
later: Mapping[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
"""Prove heartbeat continuity and stable ownership across two snapshots."""
|
||||
earlier_live = {
|
||||
w.get("worker_identity"): w for w in earlier.get("live_workers") or []
|
||||
}
|
||||
later_live = {
|
||||
w.get("worker_identity"): w for w in later.get("live_workers") or []
|
||||
}
|
||||
shared = sorted(set(earlier_live) & set(later_live))
|
||||
continuity: list[dict[str, Any]] = []
|
||||
stable_ownership = True
|
||||
for wid in shared:
|
||||
a = earlier_live[wid]
|
||||
b = later_live[wid]
|
||||
same_instance = a.get("client_instance_id") == b.get("client_instance_id")
|
||||
same_session = a.get("session_id") == b.get("session_id")
|
||||
same_generation = a.get("generation_id") == b.get("generation_id")
|
||||
hb_advanced_or_equal = True
|
||||
if a.get("last_heartbeat_at") and b.get("last_heartbeat_at"):
|
||||
hb_advanced_or_equal = b["last_heartbeat_at"] >= a["last_heartbeat_at"]
|
||||
if not (same_instance and same_session and same_generation):
|
||||
stable_ownership = False
|
||||
continuity.append(
|
||||
{
|
||||
"worker_identity": wid,
|
||||
"same_client_instance_id": same_instance,
|
||||
"same_session_id": same_session,
|
||||
"same_generation_id": same_generation,
|
||||
"heartbeat_non_decreasing": hb_advanced_or_equal,
|
||||
"earlier_heartbeat": a.get("last_heartbeat_at"),
|
||||
"later_heartbeat": b.get("last_heartbeat_at"),
|
||||
}
|
||||
)
|
||||
return {
|
||||
"shared_live_workers": shared,
|
||||
"continuity": continuity,
|
||||
"stable_ownership": stable_ownership
|
||||
and all(c["heartbeat_non_decreasing"] for c in continuity),
|
||||
"dropped_workers": sorted(set(earlier_live) - set(later_live)),
|
||||
"new_workers": sorted(set(later_live) - set(earlier_live)),
|
||||
}
|
||||
+483
-2
@@ -29,6 +29,7 @@ implements:
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import atexit
|
||||
import hashlib
|
||||
import os
|
||||
import re
|
||||
@@ -123,6 +124,103 @@ DEFAULT_REGISTRY_PATH = os.path.expanduser(
|
||||
#: worker within a single operator coffee break.
|
||||
DEFAULT_HEARTBEAT_TTL_SECONDS = 900.0
|
||||
|
||||
#: Optional operator override for the beat interval, in seconds. Clamped by
|
||||
#: :func:`heartbeat_interval_for` so it can never be set at or above the TTL.
|
||||
HEARTBEAT_INTERVAL_ENV = "GITEA_WORKER_HEARTBEAT_INTERVAL_SECONDS"
|
||||
|
||||
#: Never beat faster than this, so a misconfigured interval cannot turn the
|
||||
#: supervisor into a busy sqlite writer.
|
||||
MIN_HEARTBEAT_INTERVAL_SECONDS = 1.0
|
||||
|
||||
|
||||
def heartbeat_interval_for(
|
||||
ttl_seconds: float = DEFAULT_HEARTBEAT_TTL_SECONDS,
|
||||
override: str | float | None = None,
|
||||
) -> float:
|
||||
"""Beat interval for *ttl_seconds*: one third of the TTL, capped at a half.
|
||||
|
||||
#975. A third means two consecutive beats can be lost before the row is
|
||||
allowed to look stale, and the half-TTL cap is a hard ceiling so no
|
||||
override can produce an interval that expires the registration it is
|
||||
supposed to renew. That is the whole liveness contract: a healthy worker
|
||||
stays owned, and a worker that has genuinely stopped beating still expires
|
||||
on the configured TTL — this function never touches the TTL itself.
|
||||
"""
|
||||
try:
|
||||
ttl = float(ttl_seconds)
|
||||
except (TypeError, ValueError):
|
||||
ttl = DEFAULT_HEARTBEAT_TTL_SECONDS
|
||||
if not ttl > 0:
|
||||
ttl = DEFAULT_HEARTBEAT_TTL_SECONDS
|
||||
|
||||
interval = ttl / 3.0
|
||||
if override is not None:
|
||||
candidate = str(override).strip()
|
||||
if candidate:
|
||||
try:
|
||||
parsed = float(candidate)
|
||||
except (TypeError, ValueError):
|
||||
parsed = None
|
||||
if parsed is not None and parsed > 0:
|
||||
interval = parsed
|
||||
|
||||
ceiling = ttl / 2.0
|
||||
if interval > ceiling:
|
||||
interval = ceiling
|
||||
if interval < MIN_HEARTBEAT_INTERVAL_SECONDS:
|
||||
interval = min(MIN_HEARTBEAT_INTERVAL_SECONDS, ceiling)
|
||||
return interval
|
||||
|
||||
|
||||
#: Recorded-vs-presented pairs compared by :func:`_heartbeat_expectation_drift`.
|
||||
_HEARTBEAT_EXPECTATION_FIELDS = (
|
||||
"session_id",
|
||||
"generation_id",
|
||||
"client_name",
|
||||
"pid",
|
||||
)
|
||||
|
||||
|
||||
def _heartbeat_expectation_drift(
|
||||
record: dict[str, Any],
|
||||
*,
|
||||
expected_session_id: str | None = None,
|
||||
expected_generation_id: str | None = None,
|
||||
expected_client_name: str | None = None,
|
||||
expected_pid: int | None = None,
|
||||
) -> list[tuple[str, Any, Any]]:
|
||||
"""Return ``(field, recorded, presented)`` for every mismatched expectation.
|
||||
|
||||
Only supplied expectations are compared, so a caller that presents nothing
|
||||
gets the pre-#975 behaviour. ``client_name`` is compared through
|
||||
:func:`normalize_client_name` so one application's namespaces stay one
|
||||
client while genuinely different clients stay distinct.
|
||||
"""
|
||||
presented = {
|
||||
"session_id": expected_session_id,
|
||||
"generation_id": expected_generation_id,
|
||||
"client_name": (
|
||||
normalize_client_name(expected_client_name)
|
||||
if expected_client_name is not None
|
||||
else None
|
||||
),
|
||||
"pid": expected_pid,
|
||||
}
|
||||
drift: list[tuple[str, Any, Any]] = []
|
||||
for field in _HEARTBEAT_EXPECTATION_FIELDS:
|
||||
want = presented[field]
|
||||
if want is None:
|
||||
continue
|
||||
got = record.get(field)
|
||||
if field == "pid":
|
||||
match = got is not None and int(got) == int(want)
|
||||
else:
|
||||
match = got == want
|
||||
if not match:
|
||||
drift.append((field, got, want))
|
||||
return drift
|
||||
|
||||
|
||||
STATUS_ACTIVE = "active"
|
||||
STATUS_SUPERSEDED = "superseded"
|
||||
STATUS_RELEASED = "released"
|
||||
@@ -203,8 +301,22 @@ CREATE INDEX IF NOT EXISTS idx_worker_session
|
||||
ON worker_registrations(session_id, status);
|
||||
CREATE INDEX IF NOT EXISTS idx_worker_profile
|
||||
ON worker_registrations(profile, status);
|
||||
CREATE INDEX IF NOT EXISTS idx_worker_instance
|
||||
ON worker_registrations(client_instance_id, status);
|
||||
"""
|
||||
|
||||
#: #978 optional columns added without rewriting historical rows.
|
||||
_SCHEMA_OPTIONAL_COLUMNS: tuple[tuple[str, str], ...] = (
|
||||
("fleet_run_id", "TEXT"),
|
||||
("authenticated_account", "TEXT"),
|
||||
("process_identity", "TEXT"),
|
||||
("startup_revision", "TEXT"),
|
||||
("loaded_revision", "TEXT"),
|
||||
("parity_revision", "TEXT"),
|
||||
("live_revision", "TEXT"),
|
||||
("instance_id_provenance", "TEXT"),
|
||||
)
|
||||
|
||||
|
||||
class WorkerRegistryError(RuntimeError):
|
||||
"""Raised for registry misuse that is a programming error, not a refusal."""
|
||||
@@ -536,6 +648,17 @@ class WorkerRegistry:
|
||||
conn = self._connect()
|
||||
try:
|
||||
conn.executescript(_SCHEMA_SQL)
|
||||
existing = {
|
||||
row[1]
|
||||
for row in conn.execute(
|
||||
"PRAGMA table_info(worker_registrations)"
|
||||
).fetchall()
|
||||
}
|
||||
for name, decl in _SCHEMA_OPTIONAL_COLUMNS:
|
||||
if name not in existing:
|
||||
conn.execute(
|
||||
f"ALTER TABLE worker_registrations ADD COLUMN {name} {decl}"
|
||||
)
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
@@ -660,6 +783,14 @@ class WorkerRegistry:
|
||||
heartbeat_ttl_seconds: float = DEFAULT_HEARTBEAT_TTL_SECONDS,
|
||||
now: datetime | None = None,
|
||||
pid_alive_probe=None,
|
||||
fleet_run_id: str | None = None,
|
||||
authenticated_account: str | None = None,
|
||||
process_identity: str | None = None,
|
||||
startup_revision: str | None = None,
|
||||
loaded_revision: str | None = None,
|
||||
parity_revision: str | None = None,
|
||||
live_revision: str | None = None,
|
||||
instance_id_provenance: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Atomically register one worker identity.
|
||||
|
||||
@@ -668,6 +799,10 @@ class WorkerRegistry:
|
||||
to mint a different identity and register that instead (AC32). The
|
||||
existing registration is returned untouched so the caller can see what
|
||||
it collided with.
|
||||
|
||||
#978 also fails closed when a *live* registration already holds the same
|
||||
``client_instance_id`` for the same namespace: two workers of one
|
||||
namespace cannot share one instance.
|
||||
"""
|
||||
parsed = parse_worker_identity(worker_identity)
|
||||
if not parsed["valid"]:
|
||||
@@ -685,6 +820,9 @@ class WorkerRegistry:
|
||||
}
|
||||
|
||||
stamp = _ts(now or _utc_now())
|
||||
proc_id = process_identity or (
|
||||
f"pid-{int(pid)}" if pid is not None else None
|
||||
)
|
||||
with self._tx() as conn:
|
||||
existing = conn.execute(
|
||||
"SELECT * FROM worker_registrations WHERE worker_identity = ?",
|
||||
@@ -721,6 +859,41 @@ class WorkerRegistry:
|
||||
),
|
||||
}
|
||||
|
||||
# #978: refuse a second live worker for the same (instance, namespace).
|
||||
if namespace and client_instance_id:
|
||||
peers = conn.execute(
|
||||
"SELECT * FROM worker_registrations "
|
||||
"WHERE client_instance_id = ? AND namespace = ? AND status = ?",
|
||||
(client_instance_id, namespace, STATUS_ACTIVE),
|
||||
).fetchall()
|
||||
for peer_row in peers:
|
||||
peer = self._row_to_record(peer_row)
|
||||
pid_alive = (
|
||||
pid_alive_probe(peer.get("pid"))
|
||||
if pid_alive_probe is not None
|
||||
else None
|
||||
)
|
||||
if self.is_live(peer, now=now, pid_alive=pid_alive)["live"]:
|
||||
return {
|
||||
"success": False,
|
||||
"registered": False,
|
||||
"mutation_performed": False,
|
||||
"blocker_kind": BLOCKER_CONFLICTING_SESSIONS,
|
||||
"collision": True,
|
||||
"collision_kind": "duplicate_namespace_worker",
|
||||
"existing_registration": _public_record(peer),
|
||||
"reasons": [
|
||||
f"client_instance_id {client_instance_id!r} already "
|
||||
f"has a live worker for namespace {namespace!r} "
|
||||
f"({peer.get('worker_identity')!r}); #978 fails closed "
|
||||
"rather than registering a second worker"
|
||||
],
|
||||
"exact_next_action": (
|
||||
"Stop the extra namespace worker, or use a distinct "
|
||||
"client_instance_id for a separate application launch."
|
||||
),
|
||||
}
|
||||
|
||||
epoch = self._next_epoch(conn, generation_id)
|
||||
conn.execute(
|
||||
"""
|
||||
@@ -729,8 +902,11 @@ class WorkerRegistry:
|
||||
generation_id, role, profile, namespace, remote,
|
||||
repository_binding, pid, transport, token_fingerprint,
|
||||
started_at, last_heartbeat_at, heartbeat_ttl_seconds,
|
||||
fencing_epoch, status
|
||||
) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)
|
||||
fencing_epoch, status,
|
||||
fleet_run_id, authenticated_account, process_identity,
|
||||
startup_revision, loaded_revision, parity_revision,
|
||||
live_revision, instance_id_provenance
|
||||
) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)
|
||||
""",
|
||||
(
|
||||
worker_identity,
|
||||
@@ -751,6 +927,14 @@ class WorkerRegistry:
|
||||
float(heartbeat_ttl_seconds),
|
||||
epoch,
|
||||
STATUS_ACTIVE,
|
||||
(fleet_run_id or "").strip() or None,
|
||||
(authenticated_account or "").strip() or None,
|
||||
proc_id,
|
||||
(startup_revision or "").strip() or None,
|
||||
(loaded_revision or "").strip() or None,
|
||||
(parity_revision or "").strip() or None,
|
||||
(live_revision or "").strip() or None,
|
||||
(instance_id_provenance or "").strip() or None,
|
||||
),
|
||||
)
|
||||
row = conn.execute(
|
||||
@@ -786,11 +970,25 @@ class WorkerRegistry:
|
||||
worker_identity: str,
|
||||
fencing_epoch: int,
|
||||
now: datetime | None = None,
|
||||
expected_session_id: str | None = None,
|
||||
expected_generation_id: str | None = None,
|
||||
expected_client_name: str | None = None,
|
||||
expected_pid: int | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Renew only the owning registration (#948 AC11).
|
||||
|
||||
A stale epoch is refused rather than silently renewed, so a superseded
|
||||
session that resumes cannot heartbeat its way back into ownership.
|
||||
|
||||
#975 adds optional keyword-only *expectations*. Every one supplied must
|
||||
match the recorded row or the renewal is refused. They exist because
|
||||
identity plus epoch cannot express "renew the row I registered, and only
|
||||
that row": a recycled PID, or a second session of the same client, would
|
||||
otherwise be renewable by the wrong beater. They are fencing tokens,
|
||||
never assertions — supplying one can only cause a refusal, never grant
|
||||
anything, and omitting them preserves the pre-#975 behaviour exactly.
|
||||
Drift reports the existing ``BLOCKER_FENCED`` rather than a new
|
||||
``blocker_kind``, because consumers switch on that value.
|
||||
"""
|
||||
stamp = _ts(now or _utc_now())
|
||||
with self._tx() as conn:
|
||||
@@ -807,6 +1005,30 @@ class WorkerRegistry:
|
||||
"reasons": [f"no registration for {worker_identity!r}"],
|
||||
}
|
||||
record = self._row_to_record(row)
|
||||
drift = _heartbeat_expectation_drift(
|
||||
record,
|
||||
expected_session_id=expected_session_id,
|
||||
expected_generation_id=expected_generation_id,
|
||||
expected_client_name=expected_client_name,
|
||||
expected_pid=expected_pid,
|
||||
)
|
||||
if drift:
|
||||
return {
|
||||
"success": False,
|
||||
"renewed": False,
|
||||
"mutation_performed": False,
|
||||
"blocker_kind": BLOCKER_FENCED,
|
||||
"expectation_drift": drift,
|
||||
"reasons": [
|
||||
"presented worker expectations do not match the recorded "
|
||||
"registration, so this beater does not own the row: "
|
||||
+ "; ".join(
|
||||
f"{field} recorded {recorded!r}, presented {presented!r}"
|
||||
for field, recorded, presented in drift
|
||||
)
|
||||
+ " (#975)"
|
||||
],
|
||||
}
|
||||
if record["status"] != STATUS_ACTIVE:
|
||||
return {
|
||||
"success": False,
|
||||
@@ -1013,8 +1235,17 @@ def _public_record(record: dict[str, Any]) -> dict[str, Any]:
|
||||
"transport": record.get("transport"),
|
||||
"started_at": record.get("started_at"),
|
||||
"last_heartbeat_at": record.get("last_heartbeat_at"),
|
||||
"heartbeat_ttl_seconds": record.get("heartbeat_ttl_seconds"),
|
||||
"fencing_epoch": record.get("fencing_epoch"),
|
||||
"status": record.get("status"),
|
||||
"fleet_run_id": record.get("fleet_run_id"),
|
||||
"authenticated_account": record.get("authenticated_account"),
|
||||
"process_identity": record.get("process_identity"),
|
||||
"startup_revision": record.get("startup_revision"),
|
||||
"loaded_revision": record.get("loaded_revision"),
|
||||
"parity_revision": record.get("parity_revision"),
|
||||
"live_revision": record.get("live_revision"),
|
||||
"instance_id_provenance": record.get("instance_id_provenance"),
|
||||
}
|
||||
|
||||
|
||||
@@ -1430,3 +1661,253 @@ def resolve_bound_remote(
|
||||
"explicitly to avoid host drift (#948)."
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
# --- Production heartbeat lifecycle (#975) --------------------------------
|
||||
#
|
||||
# #948 delivered ``WorkerRegistry.heartbeat()`` and nothing ever called it, so
|
||||
# ``last_heartbeat_at`` stayed pinned to ``started_at`` for every production
|
||||
# worker and ``heartbeat_ttl_seconds`` stopped being a liveness window at all —
|
||||
# it became a hard cap on how long any client could stay attached. This
|
||||
# supervisor is the missing caller.
|
||||
#
|
||||
# It is a daemon thread rather than an asyncio task or a per-request refresh
|
||||
# because the renewal has to survive an *idle* session: a client waiting on a
|
||||
# lease with no tool call in flight must not lose ownership, which rules out
|
||||
# request-driven refresh. Registration itself happens lazily inside a tool call
|
||||
# on the server's event loop, and the registry performs blocking
|
||||
# ``BEGIN IMMEDIATE`` sqlite writes, which must not run on that loop.
|
||||
# ``daemon=True`` is deliberate: a hard kill takes the thread down with the
|
||||
# process, so a dead worker still goes stale on the normal TTL.
|
||||
|
||||
#: Refusals that mean this worker no longer owns its row. Beating again could
|
||||
#: only ever be an attempt to renew ownership it has already lost, so the
|
||||
#: supervisor stops permanently instead of retrying.
|
||||
TERMINAL_HEARTBEAT_BLOCKERS = frozenset({BLOCKER_FENCED, BLOCKER_NO_ATTACHMENT})
|
||||
|
||||
|
||||
class WorkerHeartbeatSupervisor:
|
||||
"""Periodically renew exactly one worker registration.
|
||||
|
||||
Contract:
|
||||
|
||||
* It never registers. A supervisor exists only for an already-registered
|
||||
identity, so it cannot create a second registration or a second identity
|
||||
system.
|
||||
* Every beat presents the full expectation set, so it can renew only the row
|
||||
matching this exact client name, worker identity, session, generation and
|
||||
pid.
|
||||
* A terminal refusal (fenced or missing) stops it permanently and records
|
||||
why. A fenced session must never beat its way back into ownership.
|
||||
* A transient failure (a locked database, say) is counted and the loop
|
||||
continues, so one contended write does not silently end the heartbeat.
|
||||
* Nothing here raises into a caller. ``beat_once`` returns its outcome and
|
||||
the thread body swallows everything, because a heartbeat failure must
|
||||
degrade to "not renewed" and never crash a tool call.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
registry: WorkerRegistry,
|
||||
*,
|
||||
worker_identity: str,
|
||||
fencing_epoch: int,
|
||||
session_id: str | None = None,
|
||||
generation_id: str | None = None,
|
||||
client_name: str | None = None,
|
||||
pid: int | None = None,
|
||||
ttl_seconds: float = DEFAULT_HEARTBEAT_TTL_SECONDS,
|
||||
interval_seconds: float | None = None,
|
||||
clock=None,
|
||||
) -> None:
|
||||
self._registry = registry
|
||||
self.worker_identity = worker_identity
|
||||
self.fencing_epoch = int(fencing_epoch)
|
||||
self.session_id = session_id
|
||||
self.generation_id = generation_id
|
||||
self.client_name = (
|
||||
normalize_client_name(client_name) if client_name is not None else None
|
||||
)
|
||||
self.pid = int(pid) if pid is not None else None
|
||||
self.ttl_seconds = float(ttl_seconds)
|
||||
self.interval_seconds = (
|
||||
heartbeat_interval_for(self.ttl_seconds)
|
||||
if interval_seconds is None
|
||||
else heartbeat_interval_for(self.ttl_seconds, interval_seconds)
|
||||
)
|
||||
#: Injectable so every TTL test uses controlled time and no test waits
|
||||
#: for a real interval or a real TTL to elapse.
|
||||
self._clock = clock or _utc_now
|
||||
|
||||
self._stop_event = threading.Event()
|
||||
self._thread: threading.Thread | None = None
|
||||
self._lock = threading.Lock()
|
||||
self._atexit_registered = False
|
||||
|
||||
self.started = False
|
||||
self.stopped_reason: str | None = None
|
||||
self.beats_attempted = 0
|
||||
self.beats_renewed = 0
|
||||
self.transient_failures = 0
|
||||
self.last_beat_at: str | None = None
|
||||
self.last_result: dict[str, Any] | None = None
|
||||
|
||||
# -- one beat --
|
||||
|
||||
def beat_once(self, now: datetime | None = None) -> dict[str, Any]:
|
||||
"""Renew once. Never raises; returns the registry outcome or a failure."""
|
||||
if self.stopped_reason is not None:
|
||||
return {
|
||||
"success": False,
|
||||
"renewed": False,
|
||||
"beat_attempted": False,
|
||||
"reasons": [
|
||||
f"supervisor already stopped: {self.stopped_reason}"
|
||||
],
|
||||
}
|
||||
|
||||
stamp = now or self._clock()
|
||||
self.beats_attempted += 1
|
||||
try:
|
||||
result = self._registry.heartbeat(
|
||||
worker_identity=self.worker_identity,
|
||||
fencing_epoch=self.fencing_epoch,
|
||||
now=stamp,
|
||||
expected_session_id=self.session_id,
|
||||
expected_generation_id=self.generation_id,
|
||||
expected_client_name=self.client_name,
|
||||
expected_pid=self.pid,
|
||||
)
|
||||
except Exception as exc:
|
||||
# Transient by assumption: an unexpected error is not proof this
|
||||
# worker lost ownership, so it must not silently end the heartbeat.
|
||||
self.transient_failures += 1
|
||||
result = {
|
||||
"success": False,
|
||||
"renewed": False,
|
||||
"transient": True,
|
||||
"blocker_kind": None,
|
||||
"reasons": [f"heartbeat raised {type(exc).__name__}: {exc}"],
|
||||
}
|
||||
self.last_result = result
|
||||
return result
|
||||
|
||||
self.last_result = result
|
||||
if result.get("renewed"):
|
||||
self.beats_renewed += 1
|
||||
self.last_beat_at = result.get("last_heartbeat_at") or _ts(stamp)
|
||||
return result
|
||||
|
||||
blocker = result.get("blocker_kind")
|
||||
if blocker in TERMINAL_HEARTBEAT_BLOCKERS:
|
||||
self._stop_internal(
|
||||
reason=(
|
||||
f"refused with blocker_kind={blocker!r}: "
|
||||
+ "; ".join(result.get("reasons") or [])
|
||||
)
|
||||
)
|
||||
else:
|
||||
self.transient_failures += 1
|
||||
return result
|
||||
|
||||
# -- lifecycle --
|
||||
|
||||
def start(self) -> dict[str, Any]:
|
||||
"""Start the beat thread. Idempotent; safe to call from a tool call."""
|
||||
with self._lock:
|
||||
if self.stopped_reason is not None:
|
||||
return {
|
||||
"started": False,
|
||||
"reasons": [f"supervisor stopped: {self.stopped_reason}"],
|
||||
}
|
||||
if self._thread is not None and self._thread.is_alive():
|
||||
return {"started": True, "already_running": True, "reasons": []}
|
||||
|
||||
self._stop_event.clear()
|
||||
thread = threading.Thread(
|
||||
target=self._run,
|
||||
name=f"gitea-worker-heartbeat-{self.worker_identity}",
|
||||
daemon=True,
|
||||
)
|
||||
self._thread = thread
|
||||
self.started = True
|
||||
if not self._atexit_registered:
|
||||
# Orderly shutdown stops the heartbeat. A hard kill does not
|
||||
# run this, which is correct: the row must then go stale.
|
||||
atexit.register(self._atexit_stop)
|
||||
self._atexit_registered = True
|
||||
thread.start()
|
||||
return {"started": True, "already_running": False, "reasons": []}
|
||||
|
||||
def _run(self) -> None:
|
||||
while not self._stop_event.is_set():
|
||||
# Wait first: registration already stamped a fresh heartbeat, so an
|
||||
# immediate beat would be a redundant write on every server launch.
|
||||
if self._stop_event.wait(self.interval_seconds):
|
||||
return
|
||||
try:
|
||||
self.beat_once()
|
||||
except Exception:
|
||||
# beat_once is already total; this is the last-resort guard that
|
||||
# keeps a supervisor thread from dying silently.
|
||||
self.transient_failures += 1
|
||||
if self.stopped_reason is not None:
|
||||
return
|
||||
|
||||
def stop(self, reason: str = "stopped") -> dict[str, Any]:
|
||||
"""Stop beating. Idempotent, and prompt because the loop waits on an Event."""
|
||||
self._stop_internal(reason=reason)
|
||||
thread = self._thread
|
||||
if thread is not None and thread.is_alive():
|
||||
thread.join(timeout=max(1.0, min(5.0, self.interval_seconds)))
|
||||
return {
|
||||
"stopped": True,
|
||||
"reason": self.stopped_reason,
|
||||
"thread_alive": bool(thread is not None and thread.is_alive()),
|
||||
}
|
||||
|
||||
def _stop_internal(self, *, reason: str) -> None:
|
||||
if self.stopped_reason is None:
|
||||
self.stopped_reason = reason
|
||||
self._stop_event.set()
|
||||
if self._atexit_registered:
|
||||
try:
|
||||
atexit.unregister(self._atexit_stop)
|
||||
except Exception:
|
||||
pass
|
||||
self._atexit_registered = False
|
||||
|
||||
def _atexit_stop(self) -> None:
|
||||
try:
|
||||
self.stop(reason="process exit")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# -- observability --
|
||||
|
||||
def status(self) -> dict[str, Any]:
|
||||
"""Read-only observability payload; safe to embed in a tool result."""
|
||||
thread = self._thread
|
||||
return {
|
||||
"supervised": True,
|
||||
"worker_identity": self.worker_identity,
|
||||
"fencing_epoch": self.fencing_epoch,
|
||||
"session_id": self.session_id,
|
||||
"generation_id": self.generation_id,
|
||||
"client_name": self.client_name,
|
||||
"pid": self.pid,
|
||||
"heartbeat_ttl_seconds": self.ttl_seconds,
|
||||
"heartbeat_interval_seconds": self.interval_seconds,
|
||||
"started": self.started,
|
||||
"running": bool(
|
||||
thread is not None
|
||||
and thread.is_alive()
|
||||
and self.stopped_reason is None
|
||||
),
|
||||
"stopped_reason": self.stopped_reason,
|
||||
"beats_attempted": self.beats_attempted,
|
||||
"beats_renewed": self.beats_renewed,
|
||||
"transient_failures": self.transient_failures,
|
||||
"last_heartbeat_at": self.last_beat_at,
|
||||
"last_blocker_kind": (self.last_result or {}).get("blocker_kind"),
|
||||
}
|
||||
|
||||
+146
-9
@@ -1,9 +1,16 @@
|
||||
"""Root checkout guard (#475).
|
||||
|
||||
The project root checkout is the stable control checkout on master/prgs/master.
|
||||
Author/reviewer/merge flows must fail closed when the control checkout is
|
||||
contaminated (wrong branch, detached HEAD, dirty, or HEAD behind/ahead of
|
||||
prgs/master). Isolated ``branches/...`` worktrees remain allowed.
|
||||
The project root checkout is the stable control checkout on its integration
|
||||
branch. Author/reviewer/merge flows must fail closed when the control checkout
|
||||
is contaminated (wrong branch, detached HEAD, dirty, or HEAD behind/ahead of the
|
||||
tracking integration ref). Isolated ``branches/...`` worktrees remain allowed.
|
||||
|
||||
The tracking ref is *derived per repository* (#983) rather than assumed to be
|
||||
``prgs/master``: a namespace bound to another repository — say remote ``MDCPS``
|
||||
on branch ``dev`` — is gated against ``refs/remotes/MDCPS/dev``. Derivation is
|
||||
delegated to :mod:`canonical_repository_root`, the authoritative
|
||||
repository/context resolver, so gating and parity reporting share one resolved
|
||||
target instead of maintaining two disagreeing hardcoded defaults.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -11,6 +18,7 @@ from __future__ import annotations
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
import canonical_repository_root
|
||||
from author_mutation_worktree import is_path_under_branches
|
||||
from reviewer_worktree import parse_dirty_tracked_files
|
||||
|
||||
@@ -20,19 +28,65 @@ REMEDIATION = (
|
||||
)
|
||||
|
||||
BASE_BRANCHES = frozenset({"master", "main", "dev"})
|
||||
|
||||
# Legacy PRGS-specific probe order. Retained only for callers that pass an
|
||||
# explicit ``remote_refs`` override; it is no longer the silent default, because
|
||||
# inheriting it in a non-PRGS checkout compared that checkout against a ref it
|
||||
# can never have (#983).
|
||||
REMOTE_MASTER_REFS = ("prgs/master", "refs/remotes/prgs/master")
|
||||
|
||||
|
||||
def _derive_probe_refs(root: str, explicit_remote: str | None) -> dict:
|
||||
"""Derive the ordered tracking refs to probe for *root*.
|
||||
|
||||
Returns the derivation payload from :mod:`canonical_repository_root` plus a
|
||||
``refs`` tuple, which is empty when the target is not provable.
|
||||
"""
|
||||
derived = canonical_repository_root.resolve_target_base_ref(
|
||||
root, explicit_remote=explicit_remote
|
||||
)
|
||||
return {
|
||||
"refs": tuple(derived.get("tracking_refs") or ()),
|
||||
"remote": derived.get("remote"),
|
||||
"branch": derived.get("branch"),
|
||||
"source": derived.get("source"),
|
||||
"proven": bool(derived.get("proven")),
|
||||
"reason_code": derived.get("reason_code"),
|
||||
"reasons": list(derived.get("reasons") or []),
|
||||
"cached_remote_head_branch": derived.get("cached_remote_head_branch"),
|
||||
"cached_remote_head_conflicts": bool(derived.get("cached_remote_head_conflicts")),
|
||||
}
|
||||
|
||||
|
||||
def resolve_remote_master_sha(
|
||||
canonical_repo_root: str,
|
||||
*,
|
||||
remote_refs: tuple[str, ...] | None = None,
|
||||
explicit_remote: str | None = None,
|
||||
) -> str | None:
|
||||
"""Return the commit SHA for the tracking master ref when available."""
|
||||
"""Return the commit SHA for the tracking integration ref when available.
|
||||
|
||||
This remains the single place that turns a ref into a SHA. Returns None when
|
||||
the target cannot be derived or resolved — exactly what this function already
|
||||
returned when ``rev-parse`` failed. Callers that must fail closed on missing
|
||||
evidence (the #749/#757 bootstrap path) surface that None as *missing
|
||||
evidence*, never as "no constraint".
|
||||
|
||||
*explicit_remote* is a caller-supplied disambiguation and is named that way
|
||||
deliberately: an internally inferred remote handed back in would suppress the
|
||||
resolver's ambiguity gate (#983 B2). No production caller supplies it.
|
||||
"""
|
||||
root = (canonical_repo_root or "").strip()
|
||||
if not root:
|
||||
return None
|
||||
for ref in remote_refs or REMOTE_MASTER_REFS:
|
||||
if remote_refs:
|
||||
probe: tuple[str, ...] = tuple(remote_refs)
|
||||
else:
|
||||
derived = _derive_probe_refs(root, explicit_remote)
|
||||
if not derived["proven"]:
|
||||
return None
|
||||
probe = derived["refs"]
|
||||
for ref in probe:
|
||||
res = subprocess.run(
|
||||
["git", "-C", root, "rev-parse", "--verify", ref],
|
||||
capture_output=True,
|
||||
@@ -46,6 +100,84 @@ def resolve_remote_master_sha(
|
||||
return None
|
||||
|
||||
|
||||
def resolve_remote_master_ref_state(
|
||||
canonical_repo_root: str,
|
||||
*,
|
||||
remote_refs: tuple[str, ...] | None = None,
|
||||
explicit_remote: str | None = None,
|
||||
) -> dict:
|
||||
"""Resolve the tracking integration ref together with its commit SHA.
|
||||
|
||||
Returns ``sha``, the ``ref`` it came from, the derived ``remote`` /
|
||||
``branch``, a machine-checkable ``reason_code``, and ``reasons``. ``sha`` is
|
||||
None whenever the target cannot be resolved — never a fallback to some other
|
||||
repository's commit.
|
||||
|
||||
The SHA itself is obtained through :func:`resolve_remote_master_sha` rather
|
||||
than by re-probing here, so one public function stays authoritative for
|
||||
ref-to-SHA resolution. An explicit *remote_refs* keeps the historical
|
||||
behaviour exactly: those refs are probed in order and no derivation happens.
|
||||
"""
|
||||
root = (canonical_repo_root or "").strip()
|
||||
state: dict = {
|
||||
"sha": None,
|
||||
"ref": None,
|
||||
"remote": None,
|
||||
"branch": None,
|
||||
"source": None,
|
||||
"reason_code": None,
|
||||
"reasons": [],
|
||||
"cached_remote_head_branch": None,
|
||||
"cached_remote_head_conflicts": False,
|
||||
}
|
||||
if not root:
|
||||
state["reasons"].append("no canonical repository root supplied (fail closed)")
|
||||
return state
|
||||
|
||||
if remote_refs:
|
||||
probe: tuple[str, ...] = tuple(remote_refs)
|
||||
state["source"] = "explicit_remote_refs"
|
||||
else:
|
||||
derived = _derive_probe_refs(root, explicit_remote)
|
||||
state["remote"] = derived["remote"]
|
||||
state["branch"] = derived["branch"]
|
||||
state["source"] = derived["source"]
|
||||
state["cached_remote_head_branch"] = derived["cached_remote_head_branch"]
|
||||
state["cached_remote_head_conflicts"] = derived["cached_remote_head_conflicts"]
|
||||
if not derived["proven"]:
|
||||
state["reason_code"] = derived["reason_code"]
|
||||
state["reasons"] = derived["reasons"]
|
||||
return state
|
||||
probe = derived["refs"]
|
||||
|
||||
sha = resolve_remote_master_sha(root, remote_refs=probe, explicit_remote=explicit_remote)
|
||||
if not sha:
|
||||
state["reasons"].append(
|
||||
"tracking integration ref "
|
||||
f"{' / '.join(probe) if probe else '(none derived)'} does not resolve in "
|
||||
f"'{root}' (fail closed)"
|
||||
)
|
||||
return state
|
||||
|
||||
state["sha"] = sha
|
||||
# Name the ref that actually carries this commit. When the SHA comes from a
|
||||
# test double no probe will match, so fall back to the first derived ref,
|
||||
# which is the one the guard is conceptually comparing against.
|
||||
for ref in probe:
|
||||
res = subprocess.run(
|
||||
["git", "-C", root, "rev-parse", "--verify", ref],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
if res.returncode == 0 and (res.stdout or "").strip() == sha:
|
||||
state["ref"] = ref
|
||||
break
|
||||
else:
|
||||
state["ref"] = probe[0] if probe else None
|
||||
return state
|
||||
|
||||
|
||||
resolve_tracking_master_sha = resolve_remote_master_sha
|
||||
|
||||
|
||||
@@ -59,8 +191,9 @@ def assess_root_checkout_guard(
|
||||
remote_master_sha: str | None,
|
||||
resolved_role: str | None = None,
|
||||
actual_role: str | None = None,
|
||||
remote_master_ref: str | None = None,
|
||||
) -> dict:
|
||||
"""Fail closed when the control checkout is not clean master/prgs/master.
|
||||
"""Fail closed when the control checkout is not clean on its integration ref.
|
||||
|
||||
``resolved_role`` is the preflight-resolved *task* role and ``actual_role``
|
||||
is the *active profile* role (#540). The reconciler exemption honours either
|
||||
@@ -102,9 +235,13 @@ def assess_root_checkout_guard(
|
||||
)
|
||||
|
||||
if remote_master_sha and head_sha and head_sha != remote_master_sha:
|
||||
# #983: name the ref that was actually compared. Reporting a literal
|
||||
# 'prgs/master' in a checkout gated against refs/remotes/MDCPS/dev sends
|
||||
# the operator to inspect a ref that repository does not have.
|
||||
ref_label = (remote_master_ref or "").strip() or "the tracking integration ref"
|
||||
reasons.append(
|
||||
"control checkout HEAD does not match prgs/master "
|
||||
f"(HEAD {head_sha[:12]}, prgs/master {remote_master_sha[:12]})"
|
||||
f"control checkout HEAD does not match {ref_label} "
|
||||
f"(HEAD {head_sha[:12]}, {ref_label} {remote_master_sha[:12]})"
|
||||
)
|
||||
|
||||
proven = not reasons
|
||||
|
||||
@@ -163,6 +163,18 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
# #978: instance-level fleet identity/health snapshot. gitea.read is the
|
||||
# operation gate; the tool additionally restricts role_kind to
|
||||
# controller|reconciler so author/reviewer/merger cannot use it as a
|
||||
# mutation surface and no unrelated write permission is introduced.
|
||||
"snapshot_instance_fleet": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
"gitea_snapshot_instance_fleet": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
# #644: Phase 2 Web Console recovery tasks.
|
||||
"clear_stale_binding": {
|
||||
"permission": "gitea.read",
|
||||
|
||||
@@ -175,8 +175,21 @@ class TestLauncherSnippets(unittest.TestCase):
|
||||
def test_only_safe_keys_no_secrets(self):
|
||||
entry = gitea_config.launcher_entry("prgs", "/cfg/profiles.json")["gitea-tools"]
|
||||
self.assertEqual(set(entry), {"command", "args", "env"})
|
||||
self.assertEqual(set(entry["env"]), {"GITEA_MCP_CONFIG", "GITEA_MCP_PROFILE", "GITEA_CLIENT_MANAGED"})
|
||||
# #978 B1: production launcher also injects trusted client identity.
|
||||
required = {
|
||||
"GITEA_MCP_CONFIG",
|
||||
"GITEA_MCP_PROFILE",
|
||||
"GITEA_CLIENT_MANAGED",
|
||||
"GITEA_MCP_CLIENT",
|
||||
"GITEA_MCP_CLIENT_INSTANCE",
|
||||
"GITEA_MCP_INSTANCE_PROVENANCE",
|
||||
}
|
||||
self.assertTrue(required.issubset(set(entry["env"])), entry["env"])
|
||||
self.assertEqual(entry["env"]["GITEA_MCP_PROFILE"], "prgs")
|
||||
self.assertTrue(
|
||||
entry["env"]["GITEA_MCP_CLIENT_INSTANCE"].startswith("inst-"),
|
||||
entry["env"]["GITEA_MCP_CLIENT_INSTANCE"],
|
||||
)
|
||||
blob = json.dumps(entry).lower()
|
||||
for word in ("token", "password", "secret"):
|
||||
self.assertNotIn(word, blob)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,692 @@
|
||||
"""Regression tests for Issue #983: derived target base ref for cross-repository checkouts.
|
||||
|
||||
The mutation guard previously assumed ``prgs/master`` and the parity report
|
||||
assumed ``origin/master``. Any repository using neither — for example remote
|
||||
``MDCPS`` on integration branch ``dev`` — could not prove base equivalence, so
|
||||
every gated mutation failed closed with no reachable remedy.
|
||||
|
||||
Two further defects were found by review at head ``2d5d5c9d`` and are covered
|
||||
here:
|
||||
|
||||
* **B1** — the first fix derived the integration branch from
|
||||
``refs/remotes/<remote>/HEAD``. That symref is a *local cache* written at clone
|
||||
time and never refreshed by fetch, so on the real Weekly Briefings target it
|
||||
still named ``main`` while the checkout tracked and sat exactly on ``dev``. The
|
||||
authoritative signal is the checkout's own configured upstream. The fixtures
|
||||
below therefore reproduce the **disagreement**: the cache says ``main``, the
|
||||
configured upstream says ``dev``, and ``dev`` must win.
|
||||
* **B2** — the parity report passed its internally inferred identity remote back
|
||||
into the resolver, which reads a caller-supplied remote as "the caller already
|
||||
disambiguated" and skips its ambiguity gate. Gating and reporting could then
|
||||
evaluate different repositories. An inferred remote is now never laundered into
|
||||
explicit caller intent, and ambiguity fails closed on both sides.
|
||||
|
||||
These tests build hermetic git repositories on disk (no network, no fetch) and
|
||||
assert the derived target end to end: identity remote, integration branch,
|
||||
tracking ref, fail-closed refusals, and agreement between the mutation guard and
|
||||
the parity report.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import inspect
|
||||
import os
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
|
||||
import anti_stomp_preflight
|
||||
import canonical_repository_root as crr
|
||||
import master_parity_gate
|
||||
import root_checkout_guard
|
||||
|
||||
MDCPS_URL = "https://gitea.example.net/MDCPS/WeeklyBriefings-Meta.git"
|
||||
MDCPS_SLUG = "MDCPS/WeeklyBriefings-Meta"
|
||||
PRGS_URL = "https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools.git"
|
||||
PRGS_SLUG = "Scaled-Tech-Consulting/Gitea-Tools"
|
||||
|
||||
|
||||
def _git(root: str, *args: str) -> str:
|
||||
res = subprocess.run(
|
||||
["git", "-C", root, *args],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
)
|
||||
return (res.stdout or "").strip()
|
||||
|
||||
|
||||
def _make_repo(root: str, *, remote: str | None, url: str | None) -> str:
|
||||
"""Initialise a repository with one commit and an optional named remote."""
|
||||
os.makedirs(root, exist_ok=True)
|
||||
_git(root, "init", "--quiet")
|
||||
_git(root, "config", "user.email", "[email protected]")
|
||||
_git(root, "config", "user.name", "Issue983 Test")
|
||||
_git(root, "config", "commit.gpgsign", "false")
|
||||
with open(os.path.join(root, "seed.txt"), "w", encoding="utf-8") as fh:
|
||||
fh.write("seed\n")
|
||||
_git(root, "add", "seed.txt")
|
||||
_git(root, "commit", "--quiet", "-m", "seed")
|
||||
if remote and url:
|
||||
_git(root, "remote", "add", remote, url)
|
||||
return _git(root, "rev-parse", "HEAD")
|
||||
|
||||
|
||||
def _set_remote_branch(root: str, remote: str, branch: str, sha: str) -> None:
|
||||
"""Create refs/remotes/<remote>/<branch> without contacting a network."""
|
||||
_git(root, "update-ref", f"refs/remotes/{remote}/{branch}", sha)
|
||||
|
||||
|
||||
def _set_remote_head(root: str, remote: str, branch: str) -> None:
|
||||
"""Write the *cached* refs/remotes/<remote>/HEAD symref.
|
||||
|
||||
This is the signal B1 proved untrustworthy. Fixtures use it to reproduce a
|
||||
stale cache, never to manufacture the answer under test.
|
||||
"""
|
||||
_git(
|
||||
root,
|
||||
"symbolic-ref",
|
||||
f"refs/remotes/{remote}/HEAD",
|
||||
f"refs/remotes/{remote}/{branch}",
|
||||
)
|
||||
|
||||
|
||||
def _set_upstream(root: str, remote: str, branch: str) -> None:
|
||||
"""Configure the current branch's upstream, exactly as git tracking does.
|
||||
|
||||
Writes ``branch.<current>.remote`` / ``branch.<current>.merge`` directly
|
||||
rather than via ``--set-upstream-to`` so no ref is required to pre-exist and
|
||||
no network is touched.
|
||||
"""
|
||||
current = _git(root, "symbolic-ref", "--short", "HEAD")
|
||||
_git(root, "config", f"branch.{current}.remote", remote)
|
||||
_git(root, "config", f"branch.{current}.merge", f"refs/heads/{branch}")
|
||||
|
||||
|
||||
def _checkout_new_branch(root: str, branch: str) -> None:
|
||||
_git(root, "checkout", "--quiet", "-b", branch)
|
||||
|
||||
|
||||
def _advance(root: str, message: str) -> str:
|
||||
with open(os.path.join(root, "seed.txt"), "a", encoding="utf-8") as fh:
|
||||
fh.write(message + "\n")
|
||||
_git(root, "add", "seed.txt")
|
||||
_git(root, "commit", "--quiet", "-m", message)
|
||||
return _git(root, "rev-parse", "HEAD")
|
||||
|
||||
|
||||
class _RepoCase(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.root = os.path.join(self._tmp.name, "repo")
|
||||
|
||||
def _weekly_briefings_shape(self) -> str:
|
||||
"""The real Weekly Briefings target, including its stale cache.
|
||||
|
||||
Remote ``MDCPS``; checked out on ``dev``; upstream configured to
|
||||
``MDCPS/dev``; both ``dev`` and ``main`` present as tracking refs; and
|
||||
``refs/remotes/MDCPS/HEAD`` still cached at ``main`` from clone time.
|
||||
"""
|
||||
head = _make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
_checkout_new_branch(self.root, "dev")
|
||||
_set_remote_branch(self.root, "MDCPS", "dev", head)
|
||||
stale = _advance(self.root, "main diverged long ago")
|
||||
_set_remote_branch(self.root, "MDCPS", "main", stale)
|
||||
_git(self.root, "reset", "--hard", "--quiet", head)
|
||||
_set_remote_head(self.root, "MDCPS", "main") # stale clone-time cache
|
||||
_set_upstream(self.root, "MDCPS", "dev") # authoritative
|
||||
return head
|
||||
|
||||
|
||||
class TestPrgsBehaviourPreserved(_RepoCase):
|
||||
"""Required coverage 1: PRGS prgs/master compatibility."""
|
||||
|
||||
def _prgs(self) -> str:
|
||||
head = _make_repo(self.root, remote="prgs", url=PRGS_URL)
|
||||
_set_remote_branch(self.root, "prgs", "master", head)
|
||||
_set_remote_head(self.root, "prgs", "master")
|
||||
_set_upstream(self.root, "prgs", "master")
|
||||
return head
|
||||
|
||||
def test_prgs_master_resolves_unchanged(self):
|
||||
head = self._prgs()
|
||||
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertTrue(got["proven"], got["reasons"])
|
||||
self.assertEqual(got["remote"], "prgs")
|
||||
self.assertEqual(got["branch"], "master")
|
||||
self.assertEqual(got["tracking_ref"], "refs/remotes/prgs/master")
|
||||
self.assertEqual(got["repository_slug"], PRGS_SLUG)
|
||||
self.assertEqual(got["source"], crr.BASE_REF_SOURCE_CONFIGURED_UPSTREAM)
|
||||
# Cache and upstream agree here, which is the ordinary PRGS state.
|
||||
self.assertFalse(got["cached_remote_head_conflicts"])
|
||||
self.assertEqual(root_checkout_guard.resolve_remote_master_sha(self.root), head)
|
||||
|
||||
def test_prgs_resolves_without_a_configured_upstream(self):
|
||||
"""A PRGS checkout with no tracking config still resolves master."""
|
||||
head = _make_repo(self.root, remote="prgs", url=PRGS_URL)
|
||||
_set_remote_branch(self.root, "prgs", "master", head)
|
||||
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertTrue(got["proven"], got["reasons"])
|
||||
self.assertEqual(got["tracking_ref"], "refs/remotes/prgs/master")
|
||||
self.assertEqual(got["source"], crr.BASE_REF_SOURCE_UNIQUE_CANDIDATE)
|
||||
|
||||
def test_legacy_explicit_remote_refs_path_is_untouched(self):
|
||||
"""An explicit remote_refs override still short-circuits derivation."""
|
||||
head = _make_repo(self.root, remote="prgs", url=PRGS_URL)
|
||||
_set_remote_branch(self.root, "prgs", "master", head)
|
||||
|
||||
state = root_checkout_guard.resolve_remote_master_ref_state(
|
||||
self.root, remote_refs=root_checkout_guard.REMOTE_MASTER_REFS
|
||||
)
|
||||
self.assertEqual(state["sha"], head)
|
||||
self.assertEqual(state["source"], "explicit_remote_refs")
|
||||
|
||||
|
||||
class TestStaleCachedRemoteHead(_RepoCase):
|
||||
"""Required coverage 3: configured upstream MDCPS/dev vs stale cache -> main.
|
||||
|
||||
This is B1. The fixture deliberately does **not** point
|
||||
``refs/remotes/MDCPS/HEAD`` at ``dev``; it reproduces the disagreement that
|
||||
was live on ``/Users/jasonwalker/Development/weekly-briefings``.
|
||||
"""
|
||||
|
||||
def test_configured_upstream_beats_stale_cached_remote_head(self):
|
||||
head = self._weekly_briefings_shape()
|
||||
|
||||
# Precondition: the fixture really is in the defective state.
|
||||
self.assertEqual(
|
||||
_git(self.root, "symbolic-ref", "refs/remotes/MDCPS/HEAD"),
|
||||
"refs/remotes/MDCPS/main",
|
||||
)
|
||||
self.assertEqual(_git(self.root, "config", "--get", "branch.dev.merge"), "refs/heads/dev")
|
||||
self.assertNotEqual(
|
||||
_git(self.root, "rev-parse", "refs/remotes/MDCPS/main"),
|
||||
_git(self.root, "rev-parse", "refs/remotes/MDCPS/dev"),
|
||||
)
|
||||
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertTrue(got["proven"], got["reasons"])
|
||||
self.assertEqual(got["branch"], "dev")
|
||||
self.assertEqual(got["tracking_ref"], "refs/remotes/MDCPS/dev")
|
||||
self.assertEqual(got["source"], crr.BASE_REF_SOURCE_CONFIGURED_UPSTREAM)
|
||||
# The stale cache is reported, never obeyed.
|
||||
self.assertEqual(got["cached_remote_head_branch"], "main")
|
||||
self.assertTrue(got["cached_remote_head_conflicts"])
|
||||
self.assertEqual(root_checkout_guard.resolve_remote_master_sha(self.root), head)
|
||||
|
||||
def test_checkout_on_its_integration_tip_is_not_blocked(self):
|
||||
"""The live symptom: a checkout exactly on its tip was reported stale."""
|
||||
head = self._weekly_briefings_shape()
|
||||
|
||||
state = root_checkout_guard.resolve_remote_master_ref_state(self.root)
|
||||
self.assertEqual(state["sha"], head)
|
||||
self.assertTrue(state["cached_remote_head_conflicts"])
|
||||
|
||||
assessment = root_checkout_guard.assess_root_checkout_guard(
|
||||
workspace_path=self.root,
|
||||
canonical_repo_root=self.root,
|
||||
current_branch="dev",
|
||||
head_sha=head,
|
||||
porcelain_status="",
|
||||
remote_master_sha=state["sha"],
|
||||
remote_master_ref=state["ref"],
|
||||
)
|
||||
self.assertTrue(assessment["proven"], assessment["reasons"])
|
||||
|
||||
report = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
self.assertFalse(report["stale"])
|
||||
self.assertEqual(report["base_branch"], "dev")
|
||||
self.assertEqual(report["reasons"], [])
|
||||
|
||||
def test_cached_remote_head_alone_never_proves_a_target(self):
|
||||
"""With no configured upstream, the cache cannot supply the branch."""
|
||||
head = _make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
# Only a non-candidate branch exists, and only the cache names it.
|
||||
_set_remote_branch(self.root, "MDCPS", "trunk", head)
|
||||
_set_remote_head(self.root, "MDCPS", "trunk")
|
||||
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertFalse(got["proven"])
|
||||
self.assertEqual(got["reason_code"], crr.DENY_NO_BASE_BRANCH)
|
||||
self.assertEqual(got["tracking_refs"], ())
|
||||
self.assertEqual(got["cached_remote_head_branch"], "trunk")
|
||||
# The refusal names the misleading signal so an operator is not sent
|
||||
# chasing a ref that looks authoritative.
|
||||
self.assertIn("trunk", " ".join(got["reasons"]))
|
||||
|
||||
def test_cached_remote_head_never_breaks_a_tie(self):
|
||||
"""Two candidates, no upstream: the cache must not decide."""
|
||||
head = _make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
_set_remote_branch(self.root, "MDCPS", "dev", head)
|
||||
_set_remote_branch(self.root, "MDCPS", "main", head)
|
||||
_set_remote_head(self.root, "MDCPS", "dev")
|
||||
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertFalse(got["proven"])
|
||||
self.assertEqual(got["reason_code"], crr.DENY_AMBIGUOUS_BASE_BRANCH)
|
||||
self.assertIsNone(root_checkout_guard.resolve_remote_master_sha(self.root))
|
||||
|
||||
def test_no_proven_source_is_the_remote_head_cache(self):
|
||||
"""Structural guard: the cache is not in the set of proving sources."""
|
||||
sources = {
|
||||
crr.BASE_REF_SOURCE_CONFIGURED_UPSTREAM,
|
||||
crr.BASE_REF_SOURCE_UNIQUE_CANDIDATE,
|
||||
}
|
||||
self.assertNotIn("remote_head_symref", sources)
|
||||
self.assertFalse(hasattr(crr, "BASE_REF_SOURCE_REMOTE_HEAD"))
|
||||
|
||||
|
||||
class TestCrossRepositoryTarget(_RepoCase):
|
||||
"""Required coverage 2/4/5: MDCPS/dev, no origin remote, exact case."""
|
||||
|
||||
def test_mdcps_dev_resolves(self):
|
||||
head = self._weekly_briefings_shape()
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertTrue(got["proven"], got["reasons"])
|
||||
self.assertEqual(got["remote"], "MDCPS")
|
||||
self.assertEqual(got["branch"], "dev")
|
||||
self.assertEqual(got["tracking_ref"], "refs/remotes/MDCPS/dev")
|
||||
self.assertEqual(got["repository_slug"], MDCPS_SLUG)
|
||||
self.assertEqual(root_checkout_guard.resolve_remote_master_sha(self.root), head)
|
||||
|
||||
def test_no_remote_named_origin(self):
|
||||
self._weekly_briefings_shape()
|
||||
self.assertEqual(_git(self.root, "remote"), "MDCPS")
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertTrue(got["proven"], got["reasons"])
|
||||
self.assertNotIn("origin", got["tracking_ref"])
|
||||
|
||||
def test_remote_name_case_is_preserved_exactly(self):
|
||||
self._weekly_briefings_shape()
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertEqual(got["remote"], "MDCPS")
|
||||
self.assertNotEqual(got["remote"], "mdcps")
|
||||
# The tracking ref must address the real ref, which is case-sensitive.
|
||||
self.assertEqual(got["tracking_ref"], "refs/remotes/MDCPS/dev")
|
||||
self.assertTrue(
|
||||
_git(self.root, "rev-parse", "--verify", got["tracking_ref"]),
|
||||
"case-preserved tracking ref must resolve",
|
||||
)
|
||||
|
||||
def test_lowercase_candidate_never_supplies_the_remote_name(self):
|
||||
"""Guards against silently case-folding MDCPS to the candidate 'mdcps'.
|
||||
|
||||
``_IDENTITY_REMOTE_CANDIDATES`` contains a lowercase ``mdcps`` entry and
|
||||
is probed *before* the repository's own remote listing. Git remote names
|
||||
live in case-sensitive config subsections on every platform, so the
|
||||
lowercase probe cannot resolve and the exact-case name must arrive from
|
||||
``git remote``. Asserted through config rather than ref lookup because a
|
||||
case-insensitive filesystem (macOS) resolves loose refs either way, which
|
||||
would make a ref-based assertion test the filesystem instead of the code.
|
||||
"""
|
||||
self._weekly_briefings_shape()
|
||||
res = subprocess.run(
|
||||
["git", "-C", self.root, "remote", "get-url", "mdcps"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
self.assertNotEqual(res.returncode, 0, "git remote names are case-sensitive")
|
||||
|
||||
# A caller naming the wrong case cannot disambiguate, but the target is
|
||||
# unambiguous anyway, so the exact-case name is still resolved.
|
||||
got = crr.resolve_target_base_ref(self.root, explicit_remote="mdcps")
|
||||
self.assertTrue(got["proven"], got["reasons"])
|
||||
self.assertEqual(got["remote"], "MDCPS")
|
||||
self.assertEqual(got["tracking_ref"], "refs/remotes/MDCPS/dev")
|
||||
self.assertFalse(got["identity_explicit"], "a non-matching name is not explicit intent")
|
||||
|
||||
name, slug = crr.resolve_identity_remote(self.root)
|
||||
self.assertEqual(name, "MDCPS")
|
||||
self.assertEqual(slug, MDCPS_SLUG)
|
||||
|
||||
|
||||
class TestTargetStaleness(_RepoCase):
|
||||
"""Required coverage 6/7: matching tip, and behind or divergent checkout."""
|
||||
|
||||
def test_local_equal_to_resolved_tip_is_not_stale(self):
|
||||
self._weekly_briefings_shape()
|
||||
got = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
self.assertTrue(got["determinable"])
|
||||
self.assertFalse(got["stale"])
|
||||
self.assertEqual(got["tracking_ref"], "refs/remotes/MDCPS/dev")
|
||||
self.assertEqual(got["base_remote"], "MDCPS")
|
||||
self.assertEqual(got["base_branch"], "dev")
|
||||
self.assertIsNone(got["reason_code"])
|
||||
|
||||
def test_local_behind_resolved_tip_is_stale(self):
|
||||
head = self._weekly_briefings_shape()
|
||||
advanced = _advance(self.root, "remote moved on")
|
||||
_set_remote_branch(self.root, "MDCPS", "dev", advanced)
|
||||
_git(self.root, "reset", "--hard", "--quiet", head)
|
||||
|
||||
got = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
self.assertTrue(got["determinable"])
|
||||
self.assertTrue(got["stale"])
|
||||
self.assertEqual(got["checkout_head"], head)
|
||||
self.assertEqual(got["remote_tracking_head"], advanced)
|
||||
|
||||
def test_local_divergent_from_resolved_tip_is_stale(self):
|
||||
head = self._weekly_briefings_shape()
|
||||
remote_side = _advance(self.root, "remote side")
|
||||
_set_remote_branch(self.root, "MDCPS", "dev", remote_side)
|
||||
_git(self.root, "reset", "--hard", "--quiet", head)
|
||||
local_side = _advance(self.root, "local side")
|
||||
|
||||
got = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
self.assertTrue(got["stale"])
|
||||
self.assertEqual(got["checkout_head"], local_side)
|
||||
self.assertNotEqual(local_side, remote_side)
|
||||
|
||||
|
||||
class TestFailClosed(_RepoCase):
|
||||
"""Required coverage 8/9: missing remote or ref, and ambiguous targets."""
|
||||
|
||||
def test_missing_remote_fails_closed(self):
|
||||
_make_repo(self.root, remote=None, url=None)
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertFalse(got["proven"])
|
||||
self.assertEqual(got["reason_code"], crr.DENY_NO_IDENTITY_REMOTE)
|
||||
self.assertEqual(got["tracking_refs"], ())
|
||||
self.assertIsNone(root_checkout_guard.resolve_remote_master_sha(self.root))
|
||||
|
||||
def test_missing_tracking_ref_fails_closed(self):
|
||||
_make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
# Remote configured, but nothing has ever been fetched.
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertFalse(got["proven"])
|
||||
self.assertEqual(got["reason_code"], crr.DENY_NO_BASE_BRANCH)
|
||||
self.assertIsNone(root_checkout_guard.resolve_remote_master_sha(self.root))
|
||||
|
||||
def test_configured_upstream_without_a_tracking_ref_fails_closed(self):
|
||||
"""A proven target always names a ref that resolves."""
|
||||
_make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
_set_upstream(self.root, "MDCPS", "dev") # declared, never fetched
|
||||
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertFalse(got["proven"])
|
||||
self.assertEqual(got["reason_code"], crr.DENY_NO_BASE_BRANCH)
|
||||
self.assertIsNone(got["tracking_ref"])
|
||||
|
||||
def test_every_proven_target_resolves(self):
|
||||
"""No 'proven' result may name an unresolvable tracking ref."""
|
||||
head = self._weekly_briefings_shape()
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertTrue(got["proven"])
|
||||
self.assertEqual(_git(self.root, "rev-parse", got["tracking_ref"]), head)
|
||||
|
||||
def test_ambiguous_integration_branch_fails_closed(self):
|
||||
head = _make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
# Two candidate integration branches and no configured upstream.
|
||||
_set_remote_branch(self.root, "MDCPS", "dev", head)
|
||||
_set_remote_branch(self.root, "MDCPS", "main", head)
|
||||
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertFalse(got["proven"])
|
||||
self.assertEqual(got["reason_code"], crr.DENY_AMBIGUOUS_BASE_BRANCH)
|
||||
self.assertIsNone(root_checkout_guard.resolve_remote_master_sha(self.root))
|
||||
|
||||
def test_ambiguous_identity_remote_fails_closed(self):
|
||||
head = _make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
_git(self.root, "remote", "add", "prgs", PRGS_URL)
|
||||
_set_remote_branch(self.root, "MDCPS", "dev", head)
|
||||
_set_remote_branch(self.root, "prgs", "master", head)
|
||||
|
||||
got = crr.resolve_target_base_ref(self.root)
|
||||
self.assertFalse(got["proven"])
|
||||
self.assertEqual(got["reason_code"], crr.DENY_AMBIGUOUS_REMOTE)
|
||||
self.assertEqual(got["tracking_refs"], ())
|
||||
|
||||
def test_orphan_tracking_ref_from_removed_remote_is_ignored(self):
|
||||
"""The live Gitea-Tools symptom: refs/remotes/origin/* outlives its remote."""
|
||||
head = _make_repo(self.root, remote="prgs", url=PRGS_URL)
|
||||
_set_remote_branch(self.root, "prgs", "master", head)
|
||||
_set_upstream(self.root, "prgs", "master")
|
||||
# An abandoned ref left behind by a remote that no longer exists.
|
||||
_set_remote_branch(self.root, "origin", "master", head)
|
||||
_advance(self.root, "orphan must not be consulted")
|
||||
|
||||
got = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
self.assertEqual(got["tracking_ref"], "refs/remotes/prgs/master")
|
||||
self.assertEqual(got["repository_slug"], PRGS_SLUG)
|
||||
self.assertNotIn(
|
||||
"target repository identity could not be derived from its git remote",
|
||||
got["reasons"],
|
||||
)
|
||||
|
||||
|
||||
class TestExplicitVersusInferredRemote(_RepoCase):
|
||||
"""Required coverage 11: explicit disambiguation stays distinct from inference.
|
||||
|
||||
This is B2. A remote the module inferred while probing must never re-enter
|
||||
the resolver as though an operator had named it.
|
||||
"""
|
||||
|
||||
def _two_remotes(self) -> str:
|
||||
head = _make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
_git(self.root, "remote", "add", "prgs", PRGS_URL)
|
||||
_set_remote_branch(self.root, "MDCPS", "dev", head)
|
||||
_set_remote_branch(self.root, "prgs", "master", head)
|
||||
return head
|
||||
|
||||
def test_explicit_remote_disambiguates(self):
|
||||
self._two_remotes()
|
||||
got = crr.resolve_target_base_ref(self.root, explicit_remote="MDCPS")
|
||||
self.assertTrue(got["proven"], got["reasons"])
|
||||
self.assertEqual(got["remote"], "MDCPS")
|
||||
self.assertEqual(got["branch"], "dev")
|
||||
self.assertTrue(got["identity_explicit"])
|
||||
|
||||
def test_inferred_remote_does_not_disambiguate(self):
|
||||
"""Feeding the inferred remote back in must not unlock the target."""
|
||||
self._two_remotes()
|
||||
inferred, _ = crr.resolve_identity_remote(self.root)
|
||||
self.assertIsNotNone(inferred, "the first-wins probe still returns a name")
|
||||
|
||||
# The report infers internally and must still refuse.
|
||||
report = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
self.assertEqual(report["reason_code"], crr.DENY_AMBIGUOUS_REMOTE)
|
||||
self.assertIsNone(report["repository_slug"])
|
||||
self.assertIsNone(report["tracking_ref"])
|
||||
self.assertFalse(report["stale"])
|
||||
self.assertTrue(report["reasons"])
|
||||
|
||||
def test_report_never_passes_a_remote_into_the_resolver(self):
|
||||
"""Structural guard against the exact B2 regression."""
|
||||
src = inspect.getsource(master_parity_gate.assess_target_repository_parity)
|
||||
self.assertIn("resolve_target_base_ref(canonical_root)", src)
|
||||
self.assertNotIn("resolve_target_base_ref(canonical_root, remote=", src)
|
||||
self.assertNotIn("explicit_remote=identity", src)
|
||||
# Reporting must use the ambiguity-aware identity resolver.
|
||||
self.assertIn("assess_identity_remote(canonical_root)", src)
|
||||
|
||||
def test_explicit_parameter_is_named_for_its_meaning(self):
|
||||
for fn in (
|
||||
crr.resolve_target_base_ref,
|
||||
root_checkout_guard.resolve_remote_master_sha,
|
||||
root_checkout_guard.resolve_remote_master_ref_state,
|
||||
):
|
||||
params = inspect.signature(fn).parameters
|
||||
self.assertIn("explicit_remote", params, fn.__name__)
|
||||
self.assertNotIn("remote", params, fn.__name__)
|
||||
|
||||
def test_unmatched_explicit_remote_cannot_unlock_an_ambiguous_target(self):
|
||||
self._two_remotes()
|
||||
got = crr.resolve_target_base_ref(self.root, explicit_remote="nonexistent")
|
||||
self.assertFalse(got["proven"])
|
||||
self.assertEqual(got["reason_code"], crr.DENY_AMBIGUOUS_REMOTE)
|
||||
|
||||
|
||||
class TestGatingAndReportingAgree(_RepoCase):
|
||||
"""Required coverage 10: guard and parity report make identical decisions."""
|
||||
|
||||
def test_same_resolved_target_for_gate_and_report(self):
|
||||
self._weekly_briefings_shape()
|
||||
|
||||
gate = root_checkout_guard.resolve_remote_master_ref_state(self.root)
|
||||
report = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
|
||||
self.assertEqual(gate["remote"], report["base_remote"])
|
||||
self.assertEqual(gate["branch"], report["base_branch"])
|
||||
self.assertEqual(gate["sha"], report["remote_tracking_head"])
|
||||
self.assertIn(
|
||||
gate["ref"],
|
||||
(report["tracking_ref"], f"{gate['remote']}/{gate['branch']}"),
|
||||
)
|
||||
|
||||
def test_both_sides_refuse_the_same_unresolvable_target(self):
|
||||
_make_repo(self.root, remote=None, url=None)
|
||||
|
||||
gate = root_checkout_guard.resolve_remote_master_ref_state(self.root)
|
||||
report = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
self.assertIsNone(gate["sha"])
|
||||
self.assertIsNone(report["remote_tracking_head"])
|
||||
self.assertFalse(report["stale"])
|
||||
self.assertTrue(report["reasons"])
|
||||
self.assertEqual(gate["reason_code"], report["reason_code"])
|
||||
|
||||
def test_both_sides_refuse_the_same_ambiguous_target(self):
|
||||
head = _make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
_git(self.root, "remote", "add", "prgs", PRGS_URL)
|
||||
_set_remote_branch(self.root, "MDCPS", "dev", head)
|
||||
_set_remote_branch(self.root, "prgs", "master", head)
|
||||
|
||||
gate = root_checkout_guard.resolve_remote_master_ref_state(self.root)
|
||||
report = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
|
||||
self.assertIsNone(gate["sha"])
|
||||
self.assertEqual(gate["reason_code"], crr.DENY_AMBIGUOUS_REMOTE)
|
||||
# The report must not name a repository the gate refuses to act on.
|
||||
self.assertEqual(report["reason_code"], gate["reason_code"])
|
||||
self.assertIsNone(report["repository_slug"])
|
||||
self.assertIsNone(report["base_remote"])
|
||||
self.assertIsNone(report["base_branch"])
|
||||
self.assertFalse(report["stale"])
|
||||
|
||||
def test_both_sides_refuse_the_same_ambiguous_branch(self):
|
||||
head = _make_repo(self.root, remote="MDCPS", url=MDCPS_URL)
|
||||
_set_remote_branch(self.root, "MDCPS", "dev", head)
|
||||
_set_remote_branch(self.root, "MDCPS", "main", head)
|
||||
|
||||
gate = root_checkout_guard.resolve_remote_master_ref_state(self.root)
|
||||
report = master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
self.assertEqual(gate["reason_code"], crr.DENY_AMBIGUOUS_BASE_BRANCH)
|
||||
self.assertEqual(report["reason_code"], gate["reason_code"])
|
||||
self.assertIsNone(report["tracking_ref"])
|
||||
|
||||
|
||||
class TestProductionCallers(unittest.TestCase):
|
||||
"""Required coverage 13: every production caller consumes the same target."""
|
||||
|
||||
def test_guard_reports_the_ref_it_actually_compared(self):
|
||||
assessment = root_checkout_guard.assess_root_checkout_guard(
|
||||
workspace_path="/tmp/nonexistent-workspace-983",
|
||||
canonical_repo_root="/tmp/nonexistent-root-983",
|
||||
current_branch="dev",
|
||||
head_sha="a" * 40,
|
||||
porcelain_status="",
|
||||
remote_master_sha="b" * 40,
|
||||
remote_master_ref="refs/remotes/MDCPS/dev",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
joined = " ".join(assessment["reasons"])
|
||||
self.assertIn("refs/remotes/MDCPS/dev", joined)
|
||||
self.assertNotIn("prgs/master", joined)
|
||||
|
||||
def test_guard_message_without_a_ref_stays_generic(self):
|
||||
assessment = root_checkout_guard.assess_root_checkout_guard(
|
||||
workspace_path="/tmp/nonexistent-workspace-983",
|
||||
canonical_repo_root="/tmp/nonexistent-root-983",
|
||||
current_branch="master",
|
||||
head_sha="a" * 40,
|
||||
porcelain_status="",
|
||||
remote_master_sha="b" * 40,
|
||||
)
|
||||
joined = " ".join(assessment["reasons"])
|
||||
self.assertIn("the tracking integration ref", joined)
|
||||
self.assertNotIn("prgs/master", joined)
|
||||
|
||||
def test_anti_stomp_preflight_forwards_the_resolved_ref(self):
|
||||
sig = inspect.signature(anti_stomp_preflight.assess_anti_stomp_preflight)
|
||||
self.assertIn("remote_master_ref", sig.parameters)
|
||||
src = inspect.getsource(anti_stomp_preflight.assess_anti_stomp_preflight)
|
||||
self.assertIn("remote_master_ref=remote_master_ref", src)
|
||||
|
||||
def test_no_production_caller_inherits_the_prgs_default(self):
|
||||
"""Every resolve site must derive, or pass remote_refs explicitly."""
|
||||
import gitea_mcp_server
|
||||
|
||||
src = inspect.getsource(gitea_mcp_server)
|
||||
# The four historical call sites now consume the resolved-target state.
|
||||
self.assertGreaterEqual(src.count("resolve_remote_master_ref_state("), 4)
|
||||
self.assertNotIn("resolve_remote_master_sha(canonical_root)", src)
|
||||
|
||||
def test_no_production_caller_supplies_an_explicit_remote(self):
|
||||
"""Nothing in production may suppress the ambiguity gate (#983 B2)."""
|
||||
import gitea_mcp_server
|
||||
|
||||
for module in (gitea_mcp_server, anti_stomp_preflight, master_parity_gate):
|
||||
src = inspect.getsource(module)
|
||||
self.assertNotIn("explicit_remote=", src, module.__name__)
|
||||
|
||||
|
||||
class TestRepositoryStructureUntouched(_RepoCase):
|
||||
"""Required coverage 12: resolution mutates no ref, remote, config, or checkout."""
|
||||
|
||||
def test_resolution_creates_no_refs_or_branches(self):
|
||||
self._weekly_briefings_shape()
|
||||
|
||||
fmt = "--format=%(refname) %(objectname)"
|
||||
before_refs = _git(self.root, "for-each-ref", fmt)
|
||||
before_remotes = _git(self.root, "remote")
|
||||
before_config = _git(self.root, "config", "--local", "--list")
|
||||
before_head = _git(self.root, "rev-parse", "HEAD")
|
||||
before_branch = _git(self.root, "symbolic-ref", "--short", "HEAD")
|
||||
before_status = _git(self.root, "status", "--porcelain", "--untracked-files=all")
|
||||
before_symref = _git(self.root, "symbolic-ref", "refs/remotes/MDCPS/HEAD")
|
||||
|
||||
crr.resolve_target_base_ref(self.root)
|
||||
crr.assess_identity_remote(self.root)
|
||||
root_checkout_guard.resolve_remote_master_sha(self.root)
|
||||
root_checkout_guard.resolve_remote_master_ref_state(self.root)
|
||||
master_parity_gate.assess_target_repository_parity(
|
||||
canonical_root=self.root, source="test"
|
||||
)
|
||||
|
||||
self.assertEqual(_git(self.root, "for-each-ref", fmt), before_refs)
|
||||
self.assertEqual(_git(self.root, "remote"), before_remotes)
|
||||
self.assertEqual(_git(self.root, "config", "--local", "--list"), before_config)
|
||||
self.assertEqual(_git(self.root, "rev-parse", "HEAD"), before_head)
|
||||
self.assertEqual(_git(self.root, "symbolic-ref", "--short", "HEAD"), before_branch)
|
||||
self.assertEqual(
|
||||
_git(self.root, "status", "--porcelain", "--untracked-files=all"), before_status
|
||||
)
|
||||
# The stale cache is specifically NOT repaired: that would be a mutation.
|
||||
self.assertEqual(_git(self.root, "symbolic-ref", "refs/remotes/MDCPS/HEAD"), before_symref)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,664 @@
|
||||
"""#985 — trusted client-instance identity for project-scoped LLM launches.
|
||||
|
||||
Covers the acceptance criteria of issue #985:
|
||||
|
||||
* a three-namespace (author/reviewer/merger) launch succeeds with no
|
||||
controller or reconciler profile,
|
||||
* every worker of one launch shares exactly one trusted instance ID,
|
||||
* two concurrent launches receive distinct IDs and stay distinguishable,
|
||||
* reuse, missing, legacy, malformed, unsanctioned, and unsealed identities all
|
||||
fail closed,
|
||||
* the existing five-namespace behaviour is unchanged,
|
||||
* Claude Code has a documented, runnable launch command.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import datetime as _dt
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
|
||||
import mcp_application_launcher as launcher # noqa: E402
|
||||
import mcp_fleet_snapshot as fleet # noqa: E402
|
||||
|
||||
NOW = _dt.datetime(2026, 7, 31, 12, 0, 0, tzinfo=_dt.timezone.utc)
|
||||
|
||||
PROJECT_NAMESPACES = ("author", "reviewer", "merger")
|
||||
ALL_NAMESPACES = ("author", "reviewer", "merger", "controller", "reconciler")
|
||||
|
||||
#: A project-scoped configuration intentionally has no controller/reconciler
|
||||
#: profile — that absence is the condition under test, not an omission.
|
||||
PROJECT_PROFILES = {ns: f"prgs-{ns}" for ns in PROJECT_NAMESPACES}
|
||||
FULL_PROFILES = {ns: f"prgs-{ns}" for ns in ALL_NAMESPACES}
|
||||
|
||||
|
||||
def _instance_ids(servers):
|
||||
return {
|
||||
name: entry["env"][launcher.CLIENT_INSTANCE_ENV]
|
||||
for name, entry in servers.items()
|
||||
}
|
||||
|
||||
|
||||
class ResolveLaunchNamespacesTests(unittest.TestCase):
|
||||
"""The validated subset allow-list."""
|
||||
|
||||
def test_none_resolves_to_all_five_in_canonical_order(self):
|
||||
self.assertEqual(
|
||||
launcher.resolve_launch_namespaces(None), tuple(ALL_NAMESPACES)
|
||||
)
|
||||
|
||||
def test_project_subset_resolves_in_canonical_order(self):
|
||||
# Deliberately out of order on input; canonical order on output.
|
||||
self.assertEqual(
|
||||
launcher.resolve_launch_namespaces(["merger", "author", "reviewer"]),
|
||||
("author", "reviewer", "merger"),
|
||||
)
|
||||
|
||||
def test_case_and_whitespace_normalised(self):
|
||||
self.assertEqual(
|
||||
launcher.resolve_launch_namespaces([" Author ", "REVIEWER"]),
|
||||
("author", "reviewer"),
|
||||
)
|
||||
|
||||
def test_unsanctioned_namespace_refused(self):
|
||||
with self.assertRaises(ValueError) as ctx:
|
||||
launcher.resolve_launch_namespaces(["author", "admin"])
|
||||
self.assertIn("admin", str(ctx.exception))
|
||||
|
||||
def test_typo_is_refused_not_silently_dropped(self):
|
||||
with self.assertRaises(ValueError):
|
||||
launcher.resolve_launch_namespaces(["author", "reviewr"])
|
||||
|
||||
def test_empty_subset_refused(self):
|
||||
with self.assertRaises(ValueError):
|
||||
launcher.resolve_launch_namespaces([])
|
||||
with self.assertRaises(ValueError):
|
||||
launcher.resolve_launch_namespaces([" "])
|
||||
|
||||
def test_duplicate_namespace_refused(self):
|
||||
with self.assertRaises(ValueError):
|
||||
launcher.resolve_launch_namespaces(["author", "author"])
|
||||
|
||||
|
||||
class ProjectScopedLaunchTests(unittest.TestCase):
|
||||
"""AC: three-namespace launch without controller/reconciler profiles."""
|
||||
|
||||
def _build(self, **kwargs):
|
||||
kwargs.setdefault("client_type", "claude_code")
|
||||
kwargs.setdefault("namespaces", PROJECT_NAMESPACES)
|
||||
kwargs.setdefault("command", "python3")
|
||||
kwargs.setdefault("args", ["mcp_server.py"])
|
||||
profiles = kwargs.pop("profiles", PROJECT_PROFILES)
|
||||
return launcher.build_application_mcp_servers(profiles, **kwargs)
|
||||
|
||||
def test_three_namespace_launch_succeeds_without_controller_reconciler(self):
|
||||
built = self._build(launch_nonce="proj-1", now=NOW)
|
||||
servers = built["mcpServers"]
|
||||
self.assertEqual(
|
||||
sorted(servers), ["gitea-author", "gitea-merger", "gitea-reviewer"]
|
||||
)
|
||||
self.assertEqual(built["namespace_count"], 3)
|
||||
self.assertEqual(list(built["namespaces"]), list(PROJECT_NAMESPACES))
|
||||
|
||||
def test_excluded_namespaces_are_not_started(self):
|
||||
built = self._build(launch_nonce="proj-2", now=NOW)
|
||||
self.assertNotIn("gitea-controller", built["mcpServers"])
|
||||
self.assertNotIn("gitea-reconciler", built["mcpServers"])
|
||||
self.assertEqual(
|
||||
sorted(built["excluded_namespaces"]), ["controller", "reconciler"]
|
||||
)
|
||||
self.assertTrue(built["project_scoped"])
|
||||
|
||||
def test_all_workers_of_one_launch_share_one_trusted_id(self):
|
||||
built = self._build(launch_nonce="proj-3", now=NOW)
|
||||
ids = set(_instance_ids(built["mcpServers"]).values())
|
||||
self.assertEqual(len(ids), 1, ids)
|
||||
shared = next(iter(ids))
|
||||
self.assertTrue(fleet.assess_instance_identity(shared)["trusted"])
|
||||
self.assertEqual(built["client_instance_id"], shared)
|
||||
|
||||
def test_every_worker_env_is_launcher_sealed(self):
|
||||
built = self._build(launch_nonce="proj-4", now=NOW)
|
||||
for name, entry in built["mcpServers"].items():
|
||||
env = entry["env"]
|
||||
self.assertEqual(
|
||||
env[launcher.INSTANCE_PROVENANCE_ENV],
|
||||
fleet.INSTANCE_ID_PROVENANCE_TRUSTED,
|
||||
name,
|
||||
)
|
||||
self.assertEqual(env["GITEA_CLIENT_MANAGED"], "1", name)
|
||||
self.assertFalse(
|
||||
env[launcher.CLIENT_INSTANCE_ENV].startswith("legacy-"), name
|
||||
)
|
||||
|
||||
def test_attribution_proof_holds_for_three_namespace_launch(self):
|
||||
"""A subset launch must not read as 'missing two workers'."""
|
||||
built = self._build(launch_nonce="proj-5", now=NOW)
|
||||
proof = launcher.collect_instance_ids_from_mcp_servers(
|
||||
built["mcpServers"], namespaces=PROJECT_NAMESPACES
|
||||
)
|
||||
self.assertTrue(proof["shared_single_trusted_id"], proof)
|
||||
self.assertEqual(proof["missing_servers"], [])
|
||||
# And without an explicit expectation it inspects what is present.
|
||||
inferred = launcher.collect_instance_ids_from_mcp_servers(
|
||||
built["mcpServers"]
|
||||
)
|
||||
self.assertTrue(inferred["shared_single_trusted_id"], inferred)
|
||||
|
||||
def test_missing_profile_for_requested_namespace_refused(self):
|
||||
with self.assertRaises(ValueError) as ctx:
|
||||
self._build(profiles={"author": "prgs-author"})
|
||||
self.assertIn("merger", str(ctx.exception))
|
||||
|
||||
def test_profile_for_unlaunched_namespace_is_ignored_not_started(self):
|
||||
"""Extra profiles never widen the launch beyond the validated subset."""
|
||||
built = self._build(profiles=FULL_PROFILES, launch_nonce="proj-6", now=NOW)
|
||||
self.assertEqual(len(built["mcpServers"]), 3)
|
||||
self.assertNotIn("gitea-controller", built["mcpServers"])
|
||||
|
||||
def test_unsanctioned_namespace_refused_at_build(self):
|
||||
with self.assertRaises(ValueError):
|
||||
self._build(
|
||||
profiles={**PROJECT_PROFILES, "admin": "prgs-admin"},
|
||||
namespaces=["author", "admin"],
|
||||
)
|
||||
|
||||
|
||||
class ConcurrentLaunchDistinctionTests(unittest.TestCase):
|
||||
"""AC: concurrent launches are distinct and distinguishable."""
|
||||
|
||||
def _build(self, nonce, namespaces=PROJECT_NAMESPACES, profiles=None):
|
||||
return launcher.build_application_mcp_servers(
|
||||
profiles or PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=namespaces,
|
||||
launch_nonce=nonce,
|
||||
now=NOW,
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
|
||||
def test_two_concurrent_launches_receive_different_ids(self):
|
||||
a = self._build("concurrent-a")
|
||||
b = self._build("concurrent-b")
|
||||
self.assertNotEqual(a["client_instance_id"], b["client_instance_id"])
|
||||
|
||||
def test_same_profile_across_launches_stays_distinguishable(self):
|
||||
"""Same profile, two launches: distinguishable, so not a false duplicate."""
|
||||
a = self._build("dist-a")
|
||||
b = self._build("dist-b")
|
||||
a_author = a["mcpServers"]["gitea-author"]["env"]
|
||||
b_author = b["mcpServers"]["gitea-author"]["env"]
|
||||
# Same profile on purpose — that alone must never imply duplication.
|
||||
self.assertEqual(a_author["GITEA_MCP_PROFILE"], b_author["GITEA_MCP_PROFILE"])
|
||||
self.assertNotEqual(
|
||||
a_author[launcher.CLIENT_INSTANCE_ENV],
|
||||
b_author[launcher.CLIENT_INSTANCE_ENV],
|
||||
)
|
||||
for env in (a_author, b_author):
|
||||
assessed = launcher.inherit_or_refuse_client_instance(env)
|
||||
self.assertTrue(assessed["trusted"], assessed)
|
||||
self.assertTrue(assessed["launcher_sealed"], assessed)
|
||||
|
||||
def test_no_nonce_still_yields_distinct_ids(self):
|
||||
a = self._build(None)
|
||||
b = self._build(None)
|
||||
self.assertNotEqual(a["client_instance_id"], b["client_instance_id"])
|
||||
|
||||
def test_ids_are_never_persisted_between_launches(self):
|
||||
"""No static per-project ID: two runs of one config differ."""
|
||||
first = self._build("static-check-1")
|
||||
second = self._build("static-check-2")
|
||||
self.assertNotEqual(
|
||||
first["mcpServers"]["gitea-author"]["env"][launcher.CLIENT_INSTANCE_ENV],
|
||||
second["mcpServers"]["gitea-author"]["env"][launcher.CLIENT_INSTANCE_ENV],
|
||||
)
|
||||
|
||||
|
||||
class FailClosedIdentityTests(unittest.TestCase):
|
||||
"""AC: missing / legacy / reused / conflicting / unsealed all fail closed."""
|
||||
|
||||
def test_missing_identity_fails_closed(self):
|
||||
assessed = launcher.inherit_or_refuse_client_instance({})
|
||||
self.assertFalse(assessed["trusted"])
|
||||
self.assertFalse(assessed["launcher_sealed"])
|
||||
self.assertEqual(
|
||||
assessed["provenance"], fleet.INSTANCE_ID_PROVENANCE_MISSING
|
||||
)
|
||||
|
||||
def test_legacy_pid_identity_fails_closed(self):
|
||||
for legacy in ("legacy-pid-40578", "pid-1234", "proc-99"):
|
||||
with self.subTest(legacy=legacy):
|
||||
assessed = launcher.inherit_or_refuse_client_instance(
|
||||
{
|
||||
launcher.CLIENT_INSTANCE_ENV: legacy,
|
||||
launcher.INSTANCE_PROVENANCE_ENV: (
|
||||
fleet.INSTANCE_ID_PROVENANCE_TRUSTED
|
||||
),
|
||||
}
|
||||
)
|
||||
self.assertFalse(assessed["trusted"], legacy)
|
||||
self.assertFalse(assessed["launcher_sealed"], legacy)
|
||||
|
||||
def test_wellformed_but_unsealed_identity_fails_closed(self):
|
||||
"""The decisive case: format alone must not manufacture trust."""
|
||||
forged = fleet.generate_client_instance_id(
|
||||
"claude_code", launch_nonce="forged", now=NOW
|
||||
)
|
||||
# A hand-set env var with a valid-looking ID but no launcher seal.
|
||||
assessed = launcher.inherit_or_refuse_client_instance(
|
||||
{launcher.CLIENT_INSTANCE_ENV: forged}
|
||||
)
|
||||
self.assertFalse(assessed["trusted"], assessed)
|
||||
self.assertFalse(assessed["launcher_sealed"], assessed)
|
||||
self.assertEqual(
|
||||
assessed["provenance"], fleet.INSTANCE_ID_PROVENANCE_UNSEALED
|
||||
)
|
||||
# The value is still reported for diagnosis, not silently dropped.
|
||||
self.assertEqual(assessed["client_instance_id"], forged)
|
||||
|
||||
def test_wrong_provenance_marker_fails_closed(self):
|
||||
forged = fleet.generate_client_instance_id(
|
||||
"claude_code", launch_nonce="wrong-marker", now=NOW
|
||||
)
|
||||
assessed = launcher.inherit_or_refuse_client_instance(
|
||||
{
|
||||
launcher.CLIENT_INSTANCE_ENV: forged,
|
||||
launcher.INSTANCE_PROVENANCE_ENV: "totally_trusted",
|
||||
}
|
||||
)
|
||||
self.assertFalse(assessed["trusted"], assessed)
|
||||
self.assertEqual(
|
||||
assessed["provenance"], fleet.INSTANCE_ID_PROVENANCE_UNSEALED
|
||||
)
|
||||
|
||||
def test_sealed_launcher_env_is_trusted(self):
|
||||
env = launcher.namespace_worker_env(
|
||||
profile_name="prgs-author",
|
||||
client_type="claude_code",
|
||||
client_instance_id=fleet.generate_client_instance_id(
|
||||
"claude_code", launch_nonce="sealed", now=NOW
|
||||
),
|
||||
)
|
||||
assessed = launcher.inherit_or_refuse_client_instance(env)
|
||||
self.assertTrue(assessed["trusted"], assessed)
|
||||
self.assertTrue(assessed["launcher_sealed"], assessed)
|
||||
|
||||
def test_untrusted_id_refused_on_build_and_worker_env(self):
|
||||
with self.assertRaises(ValueError):
|
||||
launcher.build_application_mcp_servers(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
client_instance_id="legacy-pid-1",
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
with self.assertRaises(ValueError):
|
||||
launcher.namespace_worker_env(
|
||||
profile_name="prgs-author",
|
||||
client_type="claude_code",
|
||||
client_instance_id="hand-written",
|
||||
)
|
||||
|
||||
def test_extra_env_cannot_forge_seal_or_identity(self):
|
||||
trusted = fleet.generate_client_instance_id(
|
||||
"claude_code", launch_nonce="seal-guard", now=NOW
|
||||
)
|
||||
env = launcher.namespace_worker_env(
|
||||
profile_name="prgs-author",
|
||||
client_type="claude_code",
|
||||
client_instance_id=trusted,
|
||||
extra_env={
|
||||
launcher.CLIENT_INSTANCE_ENV: "inst-spoof-20260101T000000Z-abcdef",
|
||||
launcher.INSTANCE_PROVENANCE_ENV: "trusted_launcher",
|
||||
"GITEA_CLIENT_MANAGED": "0",
|
||||
},
|
||||
)
|
||||
self.assertEqual(env[launcher.CLIENT_INSTANCE_ENV], trusted)
|
||||
self.assertEqual(env["GITEA_CLIENT_MANAGED"], "1")
|
||||
|
||||
def test_conflicting_ids_across_workers_break_shared_attribution(self):
|
||||
"""One launch whose workers disagree must not read as shared."""
|
||||
built = launcher.build_application_mcp_servers(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
launch_nonce="conflict",
|
||||
now=NOW,
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
servers = built["mcpServers"]
|
||||
servers["gitea-merger"]["env"][launcher.CLIENT_INSTANCE_ENV] = (
|
||||
fleet.generate_client_instance_id(
|
||||
"claude_code", launch_nonce="other", now=NOW
|
||||
)
|
||||
)
|
||||
proof = launcher.collect_instance_ids_from_mcp_servers(
|
||||
servers, namespaces=PROJECT_NAMESPACES
|
||||
)
|
||||
self.assertFalse(proof["shared_single_trusted_id"], proof)
|
||||
self.assertEqual(len(proof["unique_instance_ids"]), 2)
|
||||
|
||||
def test_absent_expected_worker_breaks_shared_attribution(self):
|
||||
built = launcher.build_application_mcp_servers(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
launch_nonce="absent",
|
||||
now=NOW,
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
servers = dict(built["mcpServers"])
|
||||
servers.pop("gitea-merger")
|
||||
proof = launcher.collect_instance_ids_from_mcp_servers(
|
||||
servers, namespaces=PROJECT_NAMESPACES
|
||||
)
|
||||
self.assertFalse(proof["shared_single_trusted_id"], proof)
|
||||
self.assertEqual(proof["missing_servers"], ["gitea-merger"])
|
||||
|
||||
def test_reused_trusted_id_is_reuse_not_a_fresh_launch(self):
|
||||
"""Resume reuses by design; two *live* launches sharing it is the hazard."""
|
||||
first = launcher.build_application_mcp_servers(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
launch_nonce="reuse-src",
|
||||
now=NOW,
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
resumed = launcher.build_application_mcp_servers(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
client_instance_id=first["client_instance_id"],
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
self.assertEqual(
|
||||
first["client_instance_id"], resumed["client_instance_id"]
|
||||
)
|
||||
# Two independent launches must never collide by default.
|
||||
independent = launcher.build_application_mcp_servers(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
self.assertNotEqual(
|
||||
first["client_instance_id"], independent["client_instance_id"]
|
||||
)
|
||||
|
||||
|
||||
class BackwardCompatibilityTests(unittest.TestCase):
|
||||
"""AC: existing full five-namespace behaviour remains compatible."""
|
||||
|
||||
def test_omitting_namespaces_launches_all_five(self):
|
||||
built = launcher.build_application_mcp_servers(
|
||||
FULL_PROFILES,
|
||||
client_type="codex",
|
||||
launch_nonce="compat",
|
||||
now=NOW,
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
self.assertEqual(len(built["mcpServers"]), 5)
|
||||
self.assertEqual(built["namespace_count"], 5)
|
||||
self.assertEqual(built["excluded_namespaces"], [])
|
||||
self.assertFalse(built["project_scoped"])
|
||||
ids = set(_instance_ids(built["mcpServers"]).values())
|
||||
self.assertEqual(len(ids), 1)
|
||||
|
||||
def test_five_namespace_missing_profile_still_refused(self):
|
||||
incomplete = {ns: f"prgs-{ns}" for ns in PROJECT_NAMESPACES}
|
||||
with self.assertRaises(ValueError) as ctx:
|
||||
launcher.build_application_mcp_servers(
|
||||
incomplete,
|
||||
client_type="codex",
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
message = str(ctx.exception)
|
||||
self.assertIn("controller", message)
|
||||
self.assertIn("reconciler", message)
|
||||
|
||||
def test_legacy_collect_call_without_namespaces_still_works(self):
|
||||
built = launcher.build_application_mcp_servers(
|
||||
FULL_PROFILES,
|
||||
client_type="codex",
|
||||
launch_nonce="compat-proof",
|
||||
now=NOW,
|
||||
command="python3",
|
||||
args=["mcp_server.py"],
|
||||
)
|
||||
proof = launcher.collect_instance_ids_from_mcp_servers(built["mcpServers"])
|
||||
self.assertTrue(proof["shared_single_trusted_id"], proof)
|
||||
self.assertEqual(proof["namespace_server_count"], 5)
|
||||
|
||||
def test_gitea_config_wrapper_supports_subset(self):
|
||||
import gitea_config
|
||||
|
||||
built = gitea_config.multi_namespace_launcher_entries(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
config_path="/cfg/profiles.json",
|
||||
launch_nonce="cfg-subset",
|
||||
)
|
||||
self.assertEqual(built["namespace_count"], 3)
|
||||
self.assertEqual(
|
||||
sorted(built["excluded_namespaces"]), ["controller", "reconciler"]
|
||||
)
|
||||
|
||||
def test_gitea_config_wrapper_default_is_five(self):
|
||||
import gitea_config
|
||||
|
||||
built = gitea_config.multi_namespace_launcher_entries(
|
||||
FULL_PROFILES,
|
||||
client_type="claude_code",
|
||||
config_path="/cfg/profiles.json",
|
||||
launch_nonce="cfg-full",
|
||||
)
|
||||
self.assertEqual(built["namespace_count"], 5)
|
||||
|
||||
|
||||
class RunnableLaunchPathTests(unittest.TestCase):
|
||||
"""AC: Claude Code has a documented supported launch command."""
|
||||
|
||||
def test_claude_code_is_a_supported_client(self):
|
||||
self.assertIn("claude_code", launcher.supported_launch_clients())
|
||||
|
||||
def test_claude_code_argv_uses_per_launch_config(self):
|
||||
argv = launcher.build_client_launch_argv("claude_code", "/tmp/launch.json")
|
||||
self.assertEqual(argv[0], "claude")
|
||||
self.assertIn("--mcp-config", argv)
|
||||
self.assertIn("/tmp/launch.json", argv)
|
||||
self.assertIn("--strict-mcp-config", argv)
|
||||
|
||||
def test_extra_args_are_forwarded(self):
|
||||
argv = launcher.build_client_launch_argv(
|
||||
"claude_code", "/tmp/launch.json", extra_args=["--resume"]
|
||||
)
|
||||
self.assertEqual(argv[-1], "--resume")
|
||||
|
||||
def test_unsupported_client_refused(self):
|
||||
with self.assertRaises(ValueError):
|
||||
launcher.build_client_launch_argv("not_a_client", "/tmp/x.json")
|
||||
|
||||
def test_prepare_launch_writes_private_per_launch_config(self):
|
||||
prepared = launcher.prepare_application_launch(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
launch_nonce="prep-1",
|
||||
now=NOW,
|
||||
)
|
||||
path = prepared["launch_config_path"]
|
||||
self.addCleanup(lambda: os.path.exists(path) and os.unlink(path))
|
||||
self.assertTrue(os.path.exists(path))
|
||||
self.assertEqual(os.stat(path).st_mode & 0o777, 0o600)
|
||||
with open(path, encoding="utf-8") as handle:
|
||||
payload = json.load(handle)
|
||||
self.assertEqual(
|
||||
sorted(payload["mcpServers"]),
|
||||
["gitea-author", "gitea-merger", "gitea-reviewer"],
|
||||
)
|
||||
self.assertTrue(prepared["shared_single_trusted_id"], prepared)
|
||||
self.assertIn(path, prepared["argv"])
|
||||
|
||||
def test_two_prepared_launches_use_distinct_files_and_ids(self):
|
||||
first = launcher.prepare_application_launch(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
now=NOW,
|
||||
)
|
||||
second = launcher.prepare_application_launch(
|
||||
PROJECT_PROFILES,
|
||||
client_type="claude_code",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
now=NOW,
|
||||
)
|
||||
for prepared in (first, second):
|
||||
path = prepared["launch_config_path"]
|
||||
self.addCleanup(lambda p=path: os.path.exists(p) and os.unlink(p))
|
||||
self.assertNotEqual(
|
||||
first["launch_config_path"], second["launch_config_path"]
|
||||
)
|
||||
self.assertNotEqual(
|
||||
first["client_instance_id"], second["client_instance_id"]
|
||||
)
|
||||
|
||||
def test_unsupported_client_writes_no_config_file(self):
|
||||
"""Fail before creating a file, so a bad invocation leaves no residue."""
|
||||
with self.assertRaises(ValueError):
|
||||
launcher.prepare_application_launch(
|
||||
PROJECT_PROFILES,
|
||||
client_type="not_a_client",
|
||||
namespaces=PROJECT_NAMESPACES,
|
||||
now=NOW,
|
||||
)
|
||||
|
||||
|
||||
class CliTests(unittest.TestCase):
|
||||
"""The runnable entry point itself."""
|
||||
|
||||
def _run_dry(self, argv):
|
||||
import contextlib
|
||||
import io
|
||||
|
||||
buffer = io.StringIO()
|
||||
with contextlib.redirect_stdout(buffer):
|
||||
code = launcher.main(argv)
|
||||
return code, buffer.getvalue()
|
||||
|
||||
def test_dry_run_project_scoped_plan(self):
|
||||
code, out = self._run_dry(
|
||||
[
|
||||
"--client",
|
||||
"claude_code",
|
||||
"--namespaces",
|
||||
"author,reviewer,merger",
|
||||
"--profile",
|
||||
"author=prgs-author",
|
||||
"--profile",
|
||||
"reviewer=prgs-reviewer",
|
||||
"--profile",
|
||||
"merger=prgs-merger",
|
||||
"--dry-run",
|
||||
]
|
||||
)
|
||||
self.assertEqual(code, 0)
|
||||
plan = json.loads(out)
|
||||
path = plan["launch_config_path"]
|
||||
self.addCleanup(lambda: os.path.exists(path) and os.unlink(path))
|
||||
self.assertEqual(plan["namespaces"], ["author", "reviewer", "merger"])
|
||||
self.assertEqual(
|
||||
sorted(plan["excluded_namespaces"]), ["controller", "reconciler"]
|
||||
)
|
||||
self.assertTrue(plan["shared_single_trusted_id"])
|
||||
self.assertTrue(
|
||||
fleet.assess_instance_identity(plan["client_instance_id"])["trusted"]
|
||||
)
|
||||
self.assertEqual(plan["argv"][0], "claude")
|
||||
|
||||
def test_cli_refuses_profile_for_unlaunched_namespace(self):
|
||||
with self.assertRaises(SystemExit):
|
||||
self._run_dry(
|
||||
[
|
||||
"--namespaces",
|
||||
"author,reviewer,merger",
|
||||
"--profile",
|
||||
"author=prgs-author",
|
||||
"--profile",
|
||||
"reviewer=prgs-reviewer",
|
||||
"--profile",
|
||||
"merger=prgs-merger",
|
||||
"--profile",
|
||||
"controller=prgs-controller",
|
||||
"--dry-run",
|
||||
]
|
||||
)
|
||||
|
||||
def test_cli_refuses_unsanctioned_namespace(self):
|
||||
with self.assertRaises(SystemExit):
|
||||
self._run_dry(
|
||||
[
|
||||
"--namespaces",
|
||||
"author,admin",
|
||||
"--profile",
|
||||
"author=prgs-author",
|
||||
"--dry-run",
|
||||
]
|
||||
)
|
||||
|
||||
def test_cli_refuses_missing_profile(self):
|
||||
with self.assertRaises(SystemExit):
|
||||
self._run_dry(
|
||||
[
|
||||
"--namespaces",
|
||||
"author,reviewer",
|
||||
"--profile",
|
||||
"author=prgs-author",
|
||||
"--dry-run",
|
||||
]
|
||||
)
|
||||
|
||||
def test_cli_refuses_malformed_profile_pair(self):
|
||||
with self.assertRaises(SystemExit):
|
||||
self._run_dry(
|
||||
["--namespaces", "author", "--profile", "prgs-author", "--dry-run"]
|
||||
)
|
||||
|
||||
def test_two_cli_runs_receive_distinct_ids(self):
|
||||
args = [
|
||||
"--namespaces",
|
||||
"author,reviewer,merger",
|
||||
"--profile",
|
||||
"author=prgs-author",
|
||||
"--profile",
|
||||
"reviewer=prgs-reviewer",
|
||||
"--profile",
|
||||
"merger=prgs-merger",
|
||||
"--dry-run",
|
||||
]
|
||||
_, first = self._run_dry(args)
|
||||
_, second = self._run_dry(args)
|
||||
a = json.loads(first)
|
||||
b = json.loads(second)
|
||||
for plan in (a, b):
|
||||
path = plan["launch_config_path"]
|
||||
self.addCleanup(lambda p=path: os.path.exists(p) and os.unlink(p))
|
||||
self.assertNotEqual(a["client_instance_id"], b["client_instance_id"])
|
||||
|
||||
|
||||
if __name__ == "__main__": # pragma: no cover
|
||||
unittest.main()
|
||||
@@ -87,13 +87,27 @@ class TestAssessRootCheckoutGuard(unittest.TestCase):
|
||||
self.assertTrue(result["block"])
|
||||
self.assertIn("tracked local edits", result["reasons"][0])
|
||||
|
||||
def test_head_behind_prgs_master_blocked(self):
|
||||
def test_head_behind_tracking_base_ref_blocked(self):
|
||||
"""#983: the base ref is derived, so the message no longer hardcodes PRGS."""
|
||||
result = self._assess(
|
||||
head_sha=OTHER_SHA,
|
||||
remote_master_sha=MASTER_SHA,
|
||||
)
|
||||
self.assertTrue(result["block"])
|
||||
self.assertIn("does not match prgs/master", result["reasons"][0])
|
||||
self.assertIn(
|
||||
"does not match the tracking integration ref", result["reasons"][0]
|
||||
)
|
||||
|
||||
def test_head_behind_named_base_ref_reports_that_ref(self):
|
||||
"""The resolved ref is named, so a non-PRGS target is reported accurately."""
|
||||
result = self._assess(
|
||||
head_sha=OTHER_SHA,
|
||||
remote_master_sha=MASTER_SHA,
|
||||
remote_master_ref="refs/remotes/MDCPS/dev",
|
||||
)
|
||||
self.assertTrue(result["block"])
|
||||
self.assertIn("does not match refs/remotes/MDCPS/dev", result["reasons"][0])
|
||||
self.assertNotIn("prgs/master", result["reasons"][0])
|
||||
|
||||
def test_merger_requires_clean_control_checkout(self):
|
||||
result = self._assess(
|
||||
|
||||
Reference in New Issue
Block a user