diff --git a/author_issue_bootstrap.py b/author_issue_bootstrap.py new file mode 100644 index 0000000..c7784fd --- /dev/null +++ b/author_issue_bootstrap.py @@ -0,0 +1,1106 @@ +"""Sanctioned author issue worktree bootstrap for allocated issues (#850). + +Bootstraps an allocated author branch, canonical worktree under ``branches/``, +worktree registration, issue lock, and lease/assignment binding without +caller-side Git, Bash, or helper scripts. + +Features: +1. Durable phase journal with read-after-write evidence for every phase. +2. Idempotent replay handling via idempotency keys. +3. Authoritative expected-base / concurrency-pin validation. +4. Typed stale-pin refusal without silent rebasing or repointing. +5. Canonical branches-root enforcement (path MUST be inside branches/). +6. Preexisting work preservation (dirty tracked/untracked check, foreign ownership refusal). +7. Compensating recovery limited strictly to artifacts created by this transition. +8. Satisfiable exact_next_action for MCP scheduled workers. +""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +from typing import Any, Mapping + +import author_mutation_worktree +import control_plane_db +import issue_lock_store +import issue_lock_worktree +import lease_lifecycle +from reviewer_worktree import parse_dirty_tracked_files +import task_capability_map + +BOOTSTRAP_TASKS = frozenset( + { + "bootstrap_author_issue_worktree", + "gitea_bootstrap_author_issue_worktree", + } +) + +PHASE_1_REQUEST_ACCEPTED = "1_request_accepted" +PHASE_2_BRANCH_CONFIRMED = "2_branch_confirmed" +PHASE_3_PATH_RESERVED = "3_path_reserved" +PHASE_4_WORKTREE_CONFIRMED = "4_worktree_confirmed" +PHASE_5_REGISTRATION_VERIFIED = "5_registration_verified" +PHASE_6_STATE_ESTABLISHED = "6_state_established" +PHASE_7_TRANSITION_COMPLETED = "7_transition_completed" +PHASE_COMPENSATING_RECOVERY = "compensating_recovery" + +JOURNAL_DIR_NAME = "bootstrap-journals" + + +def is_author_issue_bootstrap_task(task: str | None) -> bool: + """True when *task* is the author issue worktree bootstrap task.""" + return (task or "").strip() in BOOTSTRAP_TASKS + + +def get_journal_dir(override: str | None = None) -> str: + """Return the root directory for durable bootstrap phase journals.""" + if override: + path = override + elif os.environ.get("GITEA_BOOTSTRAP_JOURNAL_DIR"): + path = os.environ["GITEA_BOOTSTRAP_JOURNAL_DIR"] + else: + cache_dir = os.path.expanduser("~/.cache/gitea-tools") + path = os.path.join(cache_dir, JOURNAL_DIR_NAME) + os.makedirs(path, exist_ok=True) + return path + + +def _journal_file_path(idempotency_key: str, journal_dir: str | None = None) -> str: + safe_key = "".join( + c if c.isalnum() or c in ("-", "_", ".") else "_" + for c in idempotency_key + ) + return os.path.join(get_journal_dir(journal_dir), f"{safe_key}.json") + + +def load_phase_journal( + idempotency_key: str, journal_dir: str | None = None +) -> dict[str, Any] | None: + """Load a durable phase journal if it exists.""" + path = _journal_file_path(idempotency_key, journal_dir=journal_dir) + if not os.path.isfile(path): + return None + try: + with open(path, "r", encoding="utf-8") as f: + return json.load(f) + except Exception: + return None + + +def save_phase_journal( + journal: dict[str, Any], journal_dir: str | None = None +) -> None: + """Persist a durable phase journal with write-through file sync.""" + key = journal["idempotency_key"] + path = _journal_file_path(key, journal_dir=journal_dir) + tmp_path = f"{path}.tmp.{os.getpid()}" + with open(tmp_path, "w", encoding="utf-8") as f: + json.dump(journal, f, indent=2, sort_keys=True) + os.replace(tmp_path, path) + + +def derive_default_idempotency_key( + remote: str, + org: str | None, + repo: str | None, + issue_number: int, + assignment_id: str | None = None, + lease_id: str | None = None, +) -> str: + parts = [ + "bootstrap", + (remote or "prgs").strip(), + (org or "Scaled-Tech-Consulting").strip(), + (repo or "Gitea-Tools").strip(), + f"issue-{issue_number}", + ] + if assignment_id: + parts.append(assignment_id.strip()) + if lease_id: + parts.append(lease_id.strip()) + return ":".join(parts) + + +def _verify_assignment_and_lease_ids( + *, + assignment_id: str | None, + lease_id: str | None, + issue_number: int, + owner_session: str, + remote: str, + org: str | None, + repo: str | None, + db: control_plane_db.ControlPlaneDB | None = None, +) -> dict[str, Any] | None: + """Fail closed when caller-supplied assignment/lease IDs are unverified (#531 F4). + + Both IDs are optional together. When either is supplied, both must be + present and must resolve to the same live control-plane lease/assignment + bound to this issue and owner session. Fabricated identifiers never pass. + """ + asn = (assignment_id or "").strip() + lid = (lease_id or "").strip() + if not asn and not lid: + return None + if not asn or not lid: + return { + "success": False, + "reason_code": "incomplete_assignment_lease_ids", + "message": ( + "assignment_id and lease_id must be supplied together when " + "either is provided (fail closed)." + ), + "exact_next_action": ( + "Pass both identifiers from allocate_next_work / control-plane " + "assignment proof, or omit both." + ), + } + + try: + store = db if db is not None else control_plane_db.ControlPlaneDB() + state = store.get_lease_workflow_state(lid) + except Exception as exc: + return { + "success": False, + "reason_code": "assignment_lease_lookup_failed", + "message": f"Could not verify assignment/lease against control plane: {exc}", + "exact_next_action": ( + "Ensure the control-plane DB is available and retry with live IDs." + ), + } + if not state or not state.get("lease"): + return { + "success": False, + "reason_code": "unknown_lease_id", + "message": f"lease_id '{lid}' is not present in the control plane (fail closed).", + "exact_next_action": "Pass a live lease_id from control-plane assignment.", + } + lease = state["lease"] + assignment = state.get("assignment") or {} + work = state.get("work_item") or {} + recorded_asn = str(assignment.get("assignment_id") or "").strip() + if recorded_asn and recorded_asn != asn: + return { + "success": False, + "reason_code": "assignment_lease_mismatch", + "message": ( + f"assignment_id '{asn}' does not match lease '{lid}' " + f"(recorded assignment '{recorded_asn}') (fail closed)." + ), + "exact_next_action": "Re-read allocate_next_work proof and pass matching IDs.", + } + if not recorded_asn: + # Some lease rows may not yet have an assignment join; still require + # the lease itself to exist and bind to the claimed session/issue. + pass + lease_session = str(lease.get("session_id") or "").strip() + if lease_session and lease_session != owner_session: + return { + "success": False, + "reason_code": "lease_session_mismatch", + "message": ( + f"lease_id '{lid}' is owned by session '{lease_session}', not " + f"'{owner_session}' (fail closed)." + ), + "exact_next_action": "Use the session that holds the lease, or re-allocate.", + } + # Bind issue number when the work item records one. + work_number = work.get("number") or work.get("issue_number") or assignment.get("issue_number") + try: + if work_number is not None and int(work_number) != int(issue_number): + return { + "success": False, + "reason_code": "lease_issue_mismatch", + "message": ( + f"lease_id '{lid}' is bound to issue #{work_number}, not " + f"#{issue_number} (fail closed)." + ), + "exact_next_action": "Pass the lease issued for this issue number.", + } + except (TypeError, ValueError): + pass + _ = (remote, org, repo) # reserved for host-scoped DBs + return None + + +def run_compensating_recovery( + journal: dict[str, Any], + canonical_repo_root: str, + journal_dir: str | None = None, + *, + db: control_plane_db.ControlPlaneDB | None = None, +) -> dict[str, Any]: + """Execute compensating recovery for artifacts created by this transition only.""" + artifacts = journal.get("artifacts_created") or {} + pending = journal.get("pending_creations") or {} + rolled_back: list[str] = [] + worktree_path = journal.get("worktree_path") + branch_name = journal.get("branch_name") + + # Roll back workflow lease/assignment when this transition bound one (#531 F5). + lease_id = str(journal.get("lease_id") or "").strip() + session_id = str(journal.get("owner_session") or "").strip() + if lease_id and session_id: + try: + store = db if db is not None else control_plane_db.ControlPlaneDB() + lease_lifecycle.release_lease( + store, lease_id=lease_id, session_id=session_id + ) + rolled_back.append(f"lease:{lease_id}") + except Exception as exc: + rolled_back.append(f"lease_release_failed:{lease_id}:{type(exc).__name__}") + + # Roll back issue lock if created + if artifacts.get("lock_created") or pending.get("lock"): + issue_num = journal.get("issue_number") + if issue_num and session_id: + try: + issue_lock_store.release_session_lock( + issue_number=issue_num, + session=session_id, + lock_dir=journal_dir, + ) + rolled_back.append(f"lock:issue-{issue_num}") + except Exception: + pass + artifacts["lock_created"] = False + + worktree_created = ( + artifacts.get("worktree_registered") + or artifacts.get("worktree_dir_created") + or (pending.get("worktree_path") == worktree_path and worktree_path) + ) + + if worktree_created and worktree_path: + if os.path.exists(worktree_path): + # Re-verify cleanliness before destructive removal (F-4) + porc_res = subprocess.run( + ["git", "-C", worktree_path, "status", "--porcelain"], + capture_output=True, + text=True, + check=False, + ) + is_dirty = porc_res.returncode == 0 and bool(porc_res.stdout.strip()) + if is_dirty: + rolled_back.append(f"worktree_path_preserved_dirty:{worktree_path}") + else: + try: + subprocess.run( + [ + "git", + "-C", + canonical_repo_root, + "worktree", + "remove", + "--force", + worktree_path, + ], + capture_output=True, + text=True, + check=False, + ) + except Exception: + pass + if os.path.exists(worktree_path): + shutil.rmtree(worktree_path, ignore_errors=True) + try: + subprocess.run( + ["git", "-C", canonical_repo_root, "worktree", "prune"], + capture_output=True, + text=True, + check=False, + ) + except Exception: + pass + rolled_back.append(f"worktree_path:{worktree_path}") + else: + rolled_back.append(f"worktree_path:{worktree_path}") + + branch_created = ( + artifacts.get("branch_created") + or (pending.get("branch_name") == branch_name and branch_name) + ) + + if branch_created and branch_name: + try: + res = subprocess.run( + [ + "git", + "-C", + canonical_repo_root, + "rev-parse", + "--verify", + branch_name, + ], + capture_output=True, + text=True, + check=False, + ) + if res.returncode == 0: + # Check for author commits on branch before branch deletion (F-4) + resolved_base = journal.get("resolved_base_sha") or "master" + rev_list_res = subprocess.run( + [ + "git", + "-C", + canonical_repo_root, + "rev-list", + f"{resolved_base}..{branch_name}", + ], + capture_output=True, + text=True, + check=False, + ) + has_commits = rev_list_res.returncode == 0 and bool(rev_list_res.stdout.strip()) + if has_commits: + rolled_back.append(f"branch_preserved_commits:{branch_name}") + else: + subprocess.run( + [ + "git", + "-C", + canonical_repo_root, + "branch", + "-D", + branch_name, + ], + capture_output=True, + text=True, + check=False, + ) + rolled_back.append(f"branch:{branch_name}") + except Exception: + pass + + recovery_info = { + "executed": True, + "rolled_back": rolled_back, + "reason": journal.get("failure_reason"), + } + journal["compensating_recovery"] = recovery_info + journal["current_phase"] = PHASE_COMPENSATING_RECOVERY + save_phase_journal(journal, journal_dir=journal_dir) + return recovery_info + + +def assess_author_issue_bootstrap( + *, + workspace_path: str, + canonical_repo_root: str, + current_branch: str | None = None, + head_sha: str | None = None, + porcelain_status: str = "", + remote_master_sha: str | None = None, + remote_master_sha_error: str | None = None, + task: str | None = None, +) -> dict[str, Any]: + """Assess whether author issue worktree bootstrap may proceed from control or worktree root.""" + root = os.path.realpath(canonical_repo_root or "") + workspace = os.path.realpath(workspace_path or root or ".") + branch = (current_branch or "").strip() + dirty = parse_dirty_tracked_files(porcelain_status or "") + under_branches = ( + author_mutation_worktree.is_path_under_branches(workspace, root) + if root + else False + ) + + if not is_author_issue_bootstrap_task(task): + return { + "not_applicable": True, + "allowed": False, + "block": False, + "proven": False, + "reasons": ["task is not author_issue_bootstrap"], + } + + if under_branches: + return { + "not_applicable": False, + "allowed": True, + "block": False, + "proven": True, + "bootstrap_path": "existing_branches_worktree", + "reasons": [ + "workspace is already a registered worktree under branches/" + ], + } + + reasons: list[str] = [] + if workspace != root: + reasons.append( + "bootstrap requires workspace to be canonical control checkout or branches/ worktree" + ) + if branch not in author_mutation_worktree.BASE_BRANCHES: + reasons.append( + f"control checkout branch '{branch}' is not an accepted base branch " + f"({', '.join(sorted(author_mutation_worktree.BASE_BRANCHES))})" + ) + if dirty: + reasons.append( + f"control checkout has tracked local edits: {', '.join(dirty[:5])}" + ) + + if remote_master_sha_error: + reasons.append( + f"could not verify live master tip: {remote_master_sha_error}" + ) + elif remote_master_sha and head_sha: + h = head_sha.strip().lower() + rm = remote_master_sha.strip().lower() + if h != rm: + reasons.append( + f"control checkout HEAD ({h[:12]}) != live master tip ({rm[:12]})" + ) + + if reasons: + return { + "not_applicable": False, + "allowed": False, + "block": True, + "proven": False, + "reasons": reasons, + } + + return { + "not_applicable": False, + "allowed": True, + "block": False, + "proven": True, + "bootstrap_path": "clean_canonical_control_checkout", + "reasons": [ + "control checkout is clean on accepted base branch matching live master" + ], + } + + +import fcntl +import stat + + +class BootstrapTransitionLock: + """Inter-process file lock scoped to the transition identity / idempotency key.""" + + def __init__(self, idempotency_key: str, journal_dir: str | None = None): + safe_key = "".join( + c if c.isalnum() or c in ("-", "_", ".") else "_" + for c in idempotency_key + ) + lock_dir = os.path.realpath(get_journal_dir(journal_dir)) + if not os.path.isdir(lock_dir): + raise RuntimeError(f"Lock directory '{lock_dir}' does not exist or is not a directory") + + raw_lock_path = os.path.abspath(os.path.join(lock_dir, f"{safe_key}.lock")) + try: + common = os.path.commonpath([lock_dir, os.path.dirname(raw_lock_path)]) + except Exception: + common = None + if common != lock_dir: + raise RuntimeError(f"Lock path '{raw_lock_path}' escapes canonical lock directory '{lock_dir}'") + + self.lock_path = raw_lock_path + self.fd = None + + def __enter__(self): + if os.path.islink(self.lock_path): + raise RuntimeError(f"Refusing lock acquisition: lock path '{self.lock_path}' is a symlink") + + flags = os.O_RDWR | os.O_CREAT + if hasattr(os, "O_NOFOLLOW"): + flags |= os.O_NOFOLLOW + if hasattr(os, "O_CLOEXEC"): + flags |= os.O_CLOEXEC + + try: + fd = os.open(self.lock_path, flags, 0o600) + except OSError as exc: + raise RuntimeError(f"Failed to open lock file safely '{self.lock_path}': {exc}") from exc + + st = os.fstat(fd) + if not stat.S_ISREG(st.st_mode): + os.close(fd) + raise RuntimeError(f"Lock target '{self.lock_path}' is not a regular file") + + self.fd = fd + fcntl.flock(self.fd, fcntl.LOCK_EX) + return self + + def __exit__(self, exc_type, exc_val, exc_tb): + if self.fd is not None: + try: + fcntl.flock(self.fd, fcntl.LOCK_UN) + except Exception: + pass + try: + os.close(self.fd) + except Exception: + pass + self.fd = None + + +def bootstrap_author_issue_worktree( + *, + issue_number: int, + canonical_repo_root: str, + assignment_id: str | None = None, + lease_id: str | None = None, + expected_base_sha: str | None = None, + branch_name: str | None = None, + worktree_path: str | None = None, + idempotency_key: str | None = None, + remote: str = "prgs", + host: str | None = None, + org: str | None = "Scaled-Tech-Consulting", + repo: str | None = "Gitea-Tools", + active_identity: str | None = "jcwalker3", + active_profile: str | None = "prgs-author", + owner_session: str | None = None, + lock_dir: str | None = None, + dry_run: bool = False, +) -> dict[str, Any]: + """Execute the sanctioned author issue worktree bootstrap transition.""" + root = os.path.realpath(canonical_repo_root) + + session = (owner_session or "").strip() + if not session: + return { + "success": False, + "reason_code": "missing_owner_session", + "message": "Missing required owner_session parameter (fail closed). Session identifier cannot be fabricated or defaulted.", + "exact_next_action": ( + "Pass explicit owner_session resolved from gitea_whoami or session context." + ), + } + + if active_identity is None or not str(active_identity).strip(): + return { + "success": False, + "reason_code": "missing_active_identity", + "message": "Missing required active_identity parameter (fail closed). Identity cannot be fabricated or defaulted.", + "exact_next_action": "Pass explicit active_identity resolved from gitea_whoami.", + } + identity = str(active_identity).strip() + + if active_profile is None or not str(active_profile).strip(): + return { + "success": False, + "reason_code": "missing_active_profile", + "message": "Missing required active_profile parameter (fail closed). Profile cannot be fabricated or defaulted.", + "exact_next_action": "Pass explicit active_profile resolved from gitea_whoami.", + } + profile = str(active_profile).strip() + + # Derive standard inputs + expected_pattern = f"issue-{issue_number}" + target_branch = (branch_name or "").strip() + if not target_branch: + target_branch = f"fix/issue-{issue_number}-native-mcp-bootstrap" + elif expected_pattern not in target_branch: + return { + "success": False, + "reason_code": "invalid_branch_name", + "message": ( + f"Branch name '{target_branch}' must contain issue pattern '{expected_pattern}'" + ), + "exact_next_action": ( + f"Supply a branch_name containing '{expected_pattern}', e.g., 'fix/issue-{issue_number}-...'" + ), + } + + worktree_name = target_branch.replace("/", "-") + target_worktree = (worktree_path or "").strip() + if not target_worktree: + target_worktree = os.path.join(root, "branches", worktree_name) + target_worktree = os.path.realpath(os.path.abspath(target_worktree)) + + key = (idempotency_key or "").strip() + if not key: + key = derive_default_idempotency_key( + remote=remote, + org=org, + repo=repo, + issue_number=issue_number, + assignment_id=assignment_id, + lease_id=lease_id, + ) + + # Review #531 Finding 4: never embed unverified caller-supplied IDs. + id_block = _verify_assignment_and_lease_ids( + assignment_id=assignment_id, + lease_id=lease_id, + issue_number=issue_number, + owner_session=session, + remote=remote, + org=org, + repo=repo, + ) + if id_block is not None: + return id_block + + # Acquire cross-process file lock scoped to the idempotency key / transition identity + with BootstrapTransitionLock(key, journal_dir=lock_dir): + # Idempotency check + existing = load_phase_journal(key, journal_dir=lock_dir) + if existing and existing.get("completed"): + if ( + existing.get("issue_number") == issue_number + and existing.get("branch_name") == target_branch + and os.path.realpath(existing.get("worktree_path", "")) + == target_worktree + ): + return { + "success": True, + "replayed": True, + "message": ( + f"Idempotent replay: worktree for issue #{issue_number} already bootstrapped at {target_worktree}" + ), + "issue_number": issue_number, + "branch_name": target_branch, + "worktree_path": target_worktree, + "base_sha": existing.get("resolved_base_sha"), + "lease_id": existing.get("lease_id"), + "assignment_id": existing.get("assignment_id"), + "idempotency_key": key, + "phase_journal": existing, + "exact_next_action": ( + "Call gitea_whoami, then gitea_resolve_task_capability(task='work_issue') " + "and proceed with author implementation in the bootstrapped worktree." + ), + } + else: + return { + "success": False, + "reason_code": "incompatible_idempotency_replay", + "message": ( + f"Idempotency key '{key}' already exists with incompatible parameters " + f"(stored: {existing.get('branch_name')}, {existing.get('worktree_path')}; " + f"requested: {target_branch}, {target_worktree})" + ), + "exact_next_action": ( + "Supply a unique idempotency_key or pass compatible parameters." + ), + } + + # Initialize or resume Phase Journal + if existing: + journal = existing + artifacts = journal.setdefault("artifacts_created", {}) + artifacts.setdefault("branch_created", False) + artifacts.setdefault("worktree_dir_created", False) + artifacts.setdefault("worktree_registered", False) + artifacts.setdefault("lock_created", False) + else: + journal = { + "idempotency_key": key, + "issue_number": issue_number, + "assignment_id": assignment_id, + "lease_id": lease_id, + "expected_base_sha": expected_base_sha, + "resolved_base_sha": None, + "branch_name": target_branch, + "worktree_path": target_worktree, + "active_identity": identity, + "active_profile": profile, + "owner_session": session, + "remote": remote, + "org": org, + "repo": repo, + "phases": {}, + "artifacts_created": { + "branch_created": False, + "worktree_dir_created": False, + "worktree_registered": False, + "lock_created": False, + }, + "current_phase": PHASE_1_REQUEST_ACCEPTED, + "completed": False, + } + + # Fetch current live master SHA + try: + rev_res = subprocess.run( + ["git", "-C", root, "rev-parse", "HEAD"], + capture_output=True, + text=True, + check=True, + ) + live_master_sha = rev_res.stdout.strip() + except Exception as exc: + return { + "success": False, + "reason_code": "git_rev_parse_failed", + "message": f"Could not determine repository HEAD: {exc}", + "exact_next_action": "Verify repository git state and retry.", + } + + # Phase 1: REQUEST_ACCEPTED & Concurrency Pin Check + if expected_base_sha: + exp_norm = expected_base_sha.strip().lower() + live_norm = live_master_sha.lower() + if exp_norm != live_norm: + journal["failure_reason"] = ( + f"stale concurrency pin: expected {exp_norm[:12]} != live {live_norm[:12]}" + ) + save_phase_journal(journal, journal_dir=lock_dir) + return { + "success": False, + "reason_code": "stale_concurrency_pin", + "message": ( + f"Expected base SHA {exp_norm[:12]} does not match live master SHA {live_norm[:12]} (fail closed)." + ), + "expected_base_sha": expected_base_sha, + "live_master_sha": live_master_sha, + "exact_next_action": ( + "Re-evaluate assignment against current live master SHA and retry with updated expected_base_sha." + ), + } + + journal["resolved_base_sha"] = live_master_sha + journal["phases"][PHASE_1_REQUEST_ACCEPTED] = { + "status": "completed", + "live_master_sha": live_master_sha, + "expected_base_sha": expected_base_sha, + } + journal["current_phase"] = PHASE_2_BRANCH_CONFIRMED + save_phase_journal(journal, journal_dir=lock_dir) + + if dry_run: + return { + "success": True, + "dry_run": True, + "message": f"Dry-run: validated bootstrap intent for issue #{issue_number}", + "issue_number": issue_number, + "branch_name": target_branch, + "worktree_path": target_worktree, + "base_sha": live_master_sha, + "phase_journal": journal, + "exact_next_action": "Run without dry_run=True to execute bootstrap.", + } + + # Phase 2: BRANCH_CONFIRMED + was_branch_created_previously = journal["artifacts_created"].get("branch_created", False) + pending_branch = (journal.get("pending_creations") or {}).get("branch_name") + + branch_check = subprocess.run( + ["git", "-C", root, "rev-parse", "--verify", target_branch], + capture_output=True, + text=True, + check=False, + ) + if branch_check.returncode == 0: + branch_head = branch_check.stdout.strip() + # Verify branch head descends from base + anc_check = subprocess.run( + [ + "git", + "-C", + root, + "merge-base", + "--is-ancestor", + live_master_sha, + branch_head, + ], + capture_output=True, + text=True, + check=False, + ) + # F-8 / review #531 Finding 3: require the existing branch head to + # *contain* live master (is-ancestor) or equal it. Sharing any + # historical merge-base is not enough — that would accept stale or + # diverged branches that merely share history with master. + if anc_check.returncode != 0 and branch_head.lower() != live_master_sha.lower(): + journal["failure_reason"] = ( + f"existing branch '{target_branch}' HEAD ({branch_head[:12]}) " + f"does not contain live master ({live_master_sha[:12]})" + ) + save_phase_journal(journal, journal_dir=lock_dir) + return { + "success": False, + "reason_code": "incompatible_existing_branch", + "message": ( + f"Existing branch '{target_branch}' HEAD ({branch_head[:12]}) " + f"does not contain live master ({live_master_sha[:12]}). " + "Stale or diverged branches are refused (fail closed)." + ), + "exact_next_action": ( + "Update the branch by merging current master (no rebase/" + "force-push), or choose a branch that already contains master." + ), + } + # Preserve creation provenance monotonically across interruption and replay + if was_branch_created_previously or pending_branch == target_branch: + journal["artifacts_created"]["branch_created"] = True + else: + journal["artifacts_created"]["branch_created"] = False + else: + # Persist creation intent/provenance to disk BEFORE executing external mutation + journal.setdefault("pending_creations", {})["branch_name"] = target_branch + journal["artifacts_created"]["branch_created"] = True + save_phase_journal(journal, journal_dir=lock_dir) + + # Create branch + create_res = subprocess.run( + ["git", "-C", root, "branch", target_branch, live_master_sha], + capture_output=True, + text=True, + check=False, + ) + if create_res.returncode != 0: + journal["artifacts_created"]["branch_created"] = False + journal.get("pending_creations", {}).pop("branch_name", None) + journal["failure_reason"] = ( + f"failed to create git branch '{target_branch}': {create_res.stderr.strip()}" + ) + save_phase_journal(journal, journal_dir=lock_dir) + return { + "success": False, + "reason_code": "branch_creation_failed", + "message": f"Failed to create git branch '{target_branch}': {create_res.stderr.strip()}", + "exact_next_action": "Verify branch availability and retry.", + } + + journal["phases"][PHASE_2_BRANCH_CONFIRMED] = { + "status": "completed", + "branch_name": target_branch, + "created": journal["artifacts_created"]["branch_created"], + } + journal["current_phase"] = PHASE_3_PATH_RESERVED + save_phase_journal(journal, journal_dir=lock_dir) + + # Phase 3: PATH_RESERVED & Phase 4: WORKTREE_CONFIRMED + if not author_mutation_worktree.is_path_under_branches( + target_worktree, root + ): + journal["failure_reason"] = ( + f"target_worktree '{target_worktree}' is outside canonical branches/ root" + ) + run_compensating_recovery(journal, root, journal_dir=lock_dir) + return { + "success": False, + "reason_code": "path_outside_canonical_branches_root", + "message": ( + f"Worktree path '{target_worktree}' is outside canonical branches/ root (fail closed)." + ), + "exact_next_action": ( + "Provide a worktree_path inside canonical branches/ root, e.g., 'branches/issue-...'" + ), + } + + was_dir_created_previously = journal["artifacts_created"].get("worktree_dir_created", False) + was_registered_previously = journal["artifacts_created"].get("worktree_registered", False) + pending_wt = (journal.get("pending_creations") or {}).get("worktree_path") + + dir_exists = os.path.exists(target_worktree) + if dir_exists: + # Check porcelain directly + porc_res = subprocess.run( + ["git", "-C", target_worktree, "status", "--porcelain"], + capture_output=True, + text=True, + check=False, + ) + dirty = ( + parse_dirty_tracked_files(porc_res.stdout) + if porc_res.returncode == 0 + else [] + ) + if dirty or (porc_res.returncode == 0 and porc_res.stdout.strip()): + journal["failure_reason"] = ( + f"target worktree '{target_worktree}' contains dirty tracked/untracked files" + ) + run_compensating_recovery(journal, root, journal_dir=lock_dir) + return { + "success": False, + "reason_code": "preexisting_dirty_worktree", + "message": ( + f"Preexisting worktree '{target_worktree}' has dirty tracked/untracked files (fail closed)." + ), + "exact_next_action": ( + "Clean or stash the pre-existing worktree files before bootstrapping." + ), + } + + # Check registered branch + wt_state = issue_lock_worktree.read_worktree_git_state(target_worktree) + wt_branch = (wt_state.get("current_branch") or "").strip() + if wt_branch and wt_branch != target_branch: + journal["failure_reason"] = ( + f"existing worktree '{target_worktree}' is on branch '{wt_branch}' != expected '{target_branch}'" + ) + run_compensating_recovery(journal, root, journal_dir=lock_dir) + return { + "success": False, + "reason_code": "incompatible_existing_directory", + "message": ( + f"Existing worktree '{target_worktree}' is registered to branch '{wt_branch}' instead of '{target_branch}'." + ), + "exact_next_action": ( + "Inspect or remove the pre-existing worktree folder before bootstrapping." + ), + } + if was_dir_created_previously or pending_wt == target_worktree: + journal["artifacts_created"]["worktree_dir_created"] = True + journal["artifacts_created"]["worktree_registered"] = True + else: + journal["artifacts_created"]["worktree_dir_created"] = False + journal["artifacts_created"]["worktree_registered"] = False + else: + # Persist creation intent/provenance to disk BEFORE executing external worktree add mutation + journal.setdefault("pending_creations", {})["worktree_path"] = target_worktree + journal["artifacts_created"]["worktree_dir_created"] = True + journal["artifacts_created"]["worktree_registered"] = True + save_phase_journal(journal, journal_dir=lock_dir) + + wt_add_res = subprocess.run( + [ + "git", + "-C", + root, + "worktree", + "add", + target_worktree, + target_branch, + ], + capture_output=True, + text=True, + check=False, + ) + if wt_add_res.returncode != 0: + journal["artifacts_created"]["worktree_dir_created"] = False + journal["artifacts_created"]["worktree_registered"] = False + journal.get("pending_creations", {}).pop("worktree_path", None) + journal["failure_reason"] = ( + f"git worktree add failed: {wt_add_res.stderr.strip()}" + ) + run_compensating_recovery(journal, root, journal_dir=lock_dir) + return { + "success": False, + "reason_code": "worktree_add_failed", + "message": f"Failed to execute git worktree add: {wt_add_res.stderr.strip()}", + "exact_next_action": "Verify git worktree capabilities and retry.", + } + + journal["phases"][PHASE_3_PATH_RESERVED] = { + "status": "completed", + "worktree_path": target_worktree, + "preexisting_dir": dir_exists, + } + journal["current_phase"] = PHASE_4_WORKTREE_CONFIRMED + save_phase_journal(journal, journal_dir=lock_dir) + + # Phase 5: REGISTRATION_VERIFIED + wt_list_res = subprocess.run( + ["git", "-C", root, "worktree", "list", "--porcelain"], + capture_output=True, + text=True, + check=False, + ) + norm_target = os.path.realpath(target_worktree) + found_registration = False + if wt_list_res.returncode == 0: + for block in wt_list_res.stdout.split("\n\n"): + lines = block.strip().splitlines() + worktree_line = next( + (l[9:].strip() for l in lines if l.startswith("worktree ")), + None, + ) + if worktree_line and os.path.realpath(worktree_line) == norm_target: + found_registration = True + break + + if not found_registration: + journal["failure_reason"] = ( + f"worktree registration for '{target_worktree}' not found in git worktree list" + ) + run_compensating_recovery(journal, root, journal_dir=lock_dir) + return { + "success": False, + "reason_code": "worktree_registration_verification_failed", + "message": f"Worktree '{target_worktree}' registration verification failed.", + "exact_next_action": "Check git worktree list integrity and retry.", + } + + journal["phases"][PHASE_4_WORKTREE_CONFIRMED] = { + "status": "completed", + "worktree_path": target_worktree, + } + journal["phases"][PHASE_5_REGISTRATION_VERIFIED] = { + "status": "completed", + "registered": True, + } + journal["current_phase"] = PHASE_6_STATE_ESTABLISHED + save_phase_journal(journal, journal_dir=lock_dir) + + # Phase 6: STATE_ESTABLISHED — Issue Lock Acquisition + from datetime import datetime, timezone + try: + lock_data = { + "remote": remote, + "org": org or "Scaled-Tech-Consulting", + "repo": repo or "Gitea-Tools", + "issue_number": issue_number, + "branch": target_branch, + "branch_name": target_branch, + "worktree_path": target_worktree, + "owner_session": session, + "claimant": { + "username": identity, + "profile": profile, + }, + "assignment_id": assignment_id, + "lease_id": lease_id, + "expected_base_sha": live_master_sha, + "created_at": datetime.now(timezone.utc).isoformat(), + } + journal.setdefault("pending_creations", {})["lock"] = True + journal["artifacts_created"]["lock_created"] = True + save_phase_journal(journal, journal_dir=lock_dir) + + lock_res = issue_lock_store.bind_session_lock(lock_data, lock_dir=lock_dir) + except Exception as exc: + journal["artifacts_created"]["lock_created"] = False + journal.get("pending_creations", {}).pop("lock", None) + journal["failure_reason"] = f"issue lock binding failed: {exc}" + run_compensating_recovery(journal, root, journal_dir=lock_dir) + return { + "success": False, + "reason_code": "issue_lock_acquisition_failed", + "message": f"Could not bind canonical issue lock for issue #{issue_number}: {exc}", + "exact_next_action": "Verify lease/assignment state and retry.", + } + + journal["phases"][PHASE_6_STATE_ESTABLISHED] = { + "status": "completed", + "lock": lock_res, + } + journal["phases"][PHASE_7_TRANSITION_COMPLETED] = { + "status": "completed", + } + journal["current_phase"] = PHASE_7_TRANSITION_COMPLETED + journal["completed"] = True + save_phase_journal(journal, journal_dir=lock_dir) + + return { + "success": True, + "replayed": False, + "message": ( + f"Successfully bootstrapped author issue worktree for issue #{issue_number} " + f"at branch '{target_branch}' and worktree '{target_worktree}'." + ), + "issue_number": issue_number, + "branch_name": target_branch, + "worktree_path": target_worktree, + "base_sha": live_master_sha, + "lease_id": lease_id, + "assignment_id": assignment_id, + "idempotency_key": key, + "lock_state": lock_res, + "phase_journal": journal, + "exact_next_action": ( + "Call gitea_whoami, then gitea_resolve_task_capability(task='work_issue') " + "and proceed with author implementation in the bootstrapped worktree." + ), + } diff --git a/author_mutation_worktree.py b/author_mutation_worktree.py index b1fb48d..d0d6551 100644 --- a/author_mutation_worktree.py +++ b/author_mutation_worktree.py @@ -40,22 +40,88 @@ def _normalize_path(path: str) -> str: return (path or "").replace("\\", "/").rstrip("/") +def get_canonical_branches_root(project_root: str | None = None) -> str: + """Return the absolute path of the canonical branches directory for *project_root*.""" + root = os.path.realpath(project_root) if project_root else os.path.realpath(os.getcwd()) + canonical_repo_root = resolve_canonical_repo_root(root, root) + return os.path.realpath(os.path.join(canonical_repo_root, "branches")) + + def is_path_under_branches(path: str, project_root: str | None = None) -> bool: - """True when *path* resolves inside ``/branches/``.""" - normalized = _normalize_path(path) - if not normalized: + """True when *path* resolves inside a canonical ``branches/`` directory.""" + if not path or not str(path).strip(): return False - if "/branches/" in f"{normalized}/": - return True - if normalized.endswith("/branches"): - return True - if project_root: - root = _normalize_path(os.path.realpath(project_root)) - real = _normalize_path(os.path.realpath(path)) - if real.startswith(f"{root}/"): - rel = real[len(root) + 1 :] - return rel == "branches" or rel.startswith("branches/") - return False + try: + real_path = os.path.realpath(os.path.abspath(str(path).strip())) + except Exception: + return False + + branches_root = get_canonical_branches_root(project_root or real_path) + try: + common = os.path.commonpath([branches_root, real_path]) + except Exception: + return False + + if common != branches_root: + return False + + rel = os.path.relpath(real_path, branches_root) + return rel != "." and not rel.startswith("..") + + +def resolve_canonical_repo_root(workspace_path: str, fallback_project_root: str) -> str: + """Return the stable repository root for *workspace_path* via git metadata (#460).""" + p = (workspace_path or "").strip() + if p: + try: + res = subprocess.run( + ["git", "-C", p, "rev-parse", "--git-common-dir"], + capture_output=True, + text=True, + check=True, + ) + common = _realpath_git_common_dir(p, res.stdout) + if common.endswith(f"{os.sep}.git") or os.path.basename(common) == ".git": + candidate_root = os.path.dirname(common) + real_p = os.path.realpath(p) + try: + if os.path.commonpath([candidate_root, real_p]) == candidate_root: + return candidate_root + except Exception: + pass + except Exception: + pass + + # Fallback when git metadata is unavailable. Never string-split on + # "/branches/" (review #531 F2 / #551): recover the repo root only via + # resolved-path commonpath ancestry. Do **not** require on-disk isdir — + # MCP may launch with project_root = branches/ before that path + # exists, and #274 path-shaped worktree-as-project-root must still resolve. + fallback = os.path.realpath(fallback_project_root or workspace_path or ".") + cur = fallback + for _ in range(64): + parent = os.path.dirname(cur) + if parent == cur: + break + branches_dir = os.path.realpath(os.path.join(parent, "branches")) + try: + # Path-shaped: fallback is under parent/branches/ (commonpath). + if os.path.commonpath([branches_dir, fallback]) == branches_dir: + return parent + except ValueError: + pass + # Fallback path itself is the branches directory. + if os.path.basename(os.path.realpath(cur)) == "branches": + try: + if os.path.commonpath([os.path.realpath(cur), fallback]) == os.path.realpath( + cur + ): + return parent + except ValueError: + pass + cur = parent + + return fallback def resolve_mutation_workspace( @@ -87,29 +153,6 @@ def _realpath_git_common_dir(workspace_path: str, common_dir: str) -> str: return os.path.realpath(os.path.join(workspace_path, raw)) -def resolve_canonical_repo_root(workspace_path: str, fallback_project_root: str) -> str: - """Return the stable repository root for *workspace_path* via git metadata (#460).""" - path = (workspace_path or "").strip() - fallback = os.path.realpath(fallback_project_root) - if not path: - return fallback - try: - res = subprocess.run( - ["git", "-C", path, "rev-parse", "--git-common-dir"], - capture_output=True, - text=True, - check=True, - ) - common = _realpath_git_common_dir(path, res.stdout) - except Exception: - return fallback - if common.endswith(f"{os.sep}.git"): - return os.path.dirname(common) - if os.path.basename(common) == ".git": - return os.path.dirname(common) - return fallback - - def resolve_author_mutation_context( worktree_path: str | None, process_project_root: str, diff --git a/branch_cleanup_guard.py b/branch_cleanup_guard.py index e91ea02..584d34a 100644 --- a/branch_cleanup_guard.py +++ b/branch_cleanup_guard.py @@ -163,7 +163,19 @@ _TERMINAL_OWNERSHIP_STATUSES = frozenset( {"released", "abandoned", "done", "blocked", "terminal", "closed"} ) _EXPIRED_STATUSES = frozenset({"expired"}) -_STALE_STATUSES = frozenset({"stale", "stale_dead_process", "stale_missing_worktree"}) +_STALE_STATUSES = frozenset( + { + "stale", + "stale_dead_process", + "stale_missing_worktree", + # #790 Slice A heartbeat-lifecycle bands. Listed here so they are + # *classified* rather than falling through to the unknown-status branch; + # they still block unless the ownership record proves + # ``reclaim_allowed is True``, so the O2 fail-closed rule is unchanged. + "stale_missed_heartbeat", + "stale_absolute_cap", + } +) def _norm_str(value: Any) -> str: diff --git a/control_plane_db.py b/control_plane_db.py index cec67e4..a1f90d2 100644 --- a/control_plane_db.py +++ b/control_plane_db.py @@ -238,6 +238,34 @@ CREATE INDEX IF NOT EXISTS idx_session_checkpoints_session ON session_checkpoints(remote, org, repo, session_id); CREATE INDEX IF NOT EXISTS idx_session_checkpoints_work ON session_checkpoints(remote, org, repo, work_kind, work_number); + +-- Model usage, token cost, latency, and performance events (#651) +CREATE TABLE IF NOT EXISTS usage_events ( + usage_id INTEGER PRIMARY KEY AUTOINCREMENT, + session_id TEXT, + remote TEXT NOT NULL DEFAULT 'dadeschools', + org TEXT NOT NULL DEFAULT '', + repo TEXT NOT NULL DEFAULT '', + project_id TEXT, + role TEXT NOT NULL DEFAULT 'unknown', + model TEXT NOT NULL DEFAULT 'unknown', + issue_number INTEGER, + pr_number INTEGER, + stage TEXT NOT NULL DEFAULT 'unknown', + input_tokens INTEGER, + output_tokens INTEGER, + total_tokens INTEGER, + estimated_cost_usd REAL, + latency_ms INTEGER, + duration_ms INTEGER, + status TEXT NOT NULL DEFAULT 'success', + metadata TEXT, + created_at TEXT NOT NULL +); + +CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo); +CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model); +CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage); """ @@ -391,6 +419,7 @@ class ControlPlaneDB: self._migrate_incident_links_null_scope(conn) self._migrate_lease_lifecycle_columns(conn) self._migrate_session_ownership_columns(conn) + self._migrate_usage_events_table(conn) conn.execute( "INSERT OR REPLACE INTO schema_meta(key, value) VALUES (?, ?)", ("schema_version", str(SCHEMA_VERSION)), @@ -570,6 +599,207 @@ class ControlPlaneDB: f"UPDATE incident_links SET {col} = '' WHERE {col} IS NULL" ) + def _migrate_usage_events_table(self, conn: sqlite3.Connection) -> None: + """Create usage_events table and indexes if they do not exist (#651).""" + conn.execute(""" + CREATE TABLE IF NOT EXISTS usage_events ( + usage_id INTEGER PRIMARY KEY AUTOINCREMENT, + session_id TEXT, + remote TEXT NOT NULL DEFAULT 'dadeschools', + org TEXT NOT NULL DEFAULT '', + repo TEXT NOT NULL DEFAULT '', + project_id TEXT, + role TEXT NOT NULL DEFAULT 'unknown', + model TEXT NOT NULL DEFAULT 'unknown', + issue_number INTEGER, + pr_number INTEGER, + stage TEXT NOT NULL DEFAULT 'unknown', + input_tokens INTEGER, + output_tokens INTEGER, + total_tokens INTEGER, + estimated_cost_usd REAL, + latency_ms INTEGER, + duration_ms INTEGER, + status TEXT NOT NULL DEFAULT 'success', + metadata TEXT, + created_at TEXT NOT NULL + ); + """) + conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo);") + conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model);") + conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage);") + + # #651 retention: cap growth so unauthenticated or high-volume ingest + # cannot DoS the control-plane DB (PR #876 F3). Applied after every write. + USAGE_EVENTS_MAX_ROWS = 10_000 + USAGE_EVENTS_RETENTION_DAYS = 90 + + def record_usage_event( + self, + *, + session_id: str | None = None, + remote: str = "dadeschools", + org: str = "", + repo: str = "", + project_id: str | None = None, + role: str = "unknown", + model: str = "unknown", + issue_number: int | None = None, + pr_number: int | None = None, + stage: str = "unknown", + input_tokens: int | None = None, + output_tokens: int | None = None, + total_tokens: int | None = None, + estimated_cost_usd: float | None = None, + latency_ms: int | None = None, + duration_ms: int | None = None, + status: str = "success", + metadata: str | dict[str, Any] | None = None, + created_at: str | None = None, + ) -> int: + """Record a model usage, token cost, latency, or stage performance event (#651).""" + ts = created_at or _ts() + meta_str: str | None = None + if metadata is not None: + from webui import console_redaction + redacted_meta = console_redaction.redact_payload(metadata) + if isinstance(redacted_meta, str): + meta_str = redacted_meta + else: + try: + meta_str = json.dumps(redacted_meta, default=str) + except Exception: + meta_str = str(redacted_meta) + + if total_tokens is None and (input_tokens is not None or output_tokens is not None): + total_tokens = (input_tokens or 0) + (output_tokens or 0) + + with self._tx(immediate=True) as conn: + cursor = conn.execute( + """ + INSERT INTO usage_events ( + session_id, remote, org, repo, project_id, role, model, + issue_number, pr_number, stage, input_tokens, output_tokens, + total_tokens, estimated_cost_usd, latency_ms, duration_ms, + status, metadata, created_at + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + """, + ( + session_id, + remote, + org, + repo, + project_id, + role, + model, + issue_number, + pr_number, + stage, + input_tokens, + output_tokens, + total_tokens, + estimated_cost_usd, + latency_ms, + duration_ms, + status, + meta_str, + ts, + ), + ) + usage_id = cursor.lastrowid + self._enforce_usage_events_retention(conn) + return usage_id + + def _enforce_usage_events_retention(self, conn: sqlite3.Connection) -> None: + """Drop aged and excess usage_events rows (PR #876 F3).""" + # Age-based: ISO-8601 UTC timestamps compare lexicographically. + cutoff = ( + datetime.now(timezone.utc) + - timedelta(days=int(self.USAGE_EVENTS_RETENTION_DAYS)) + ).strftime("%Y-%m-%dT%H:%M:%SZ") + conn.execute( + "DELETE FROM usage_events WHERE created_at < ?", + (cutoff,), + ) + # Count-based: keep the newest USAGE_EVENTS_MAX_ROWS by usage_id. + max_rows = int(self.USAGE_EVENTS_MAX_ROWS) + if max_rows > 0: + conn.execute( + """ + DELETE FROM usage_events + WHERE usage_id NOT IN ( + SELECT usage_id FROM usage_events + ORDER BY usage_id DESC + LIMIT ? + ) + """, + (max_rows,), + ) + + def query_usage_events( + self, + *, + remote: str | None = None, + org: str | None = None, + repo: str | None = None, + project_id: str | None = None, + role: str | None = None, + model: str | None = None, + issue_number: int | None = None, + pr_number: int | None = None, + stage: str | None = None, + session_id: str | None = None, + limit: int = 500, + offset: int = 0, + ) -> list[dict[str, Any]]: + """Query stored usage events matching filters (#651).""" + conditions = [] + params = [] + if remote: + conditions.append("remote = ?") + params.append(remote) + if org: + conditions.append("org = ?") + params.append(org) + if repo: + conditions.append("repo = ?") + params.append(repo) + if project_id: + conditions.append("project_id = ?") + params.append(project_id) + if role: + conditions.append("role = ?") + params.append(role) + if model: + conditions.append("model = ?") + params.append(model) + if issue_number is not None: + conditions.append("issue_number = ?") + params.append(issue_number) + if pr_number is not None: + conditions.append("pr_number = ?") + params.append(pr_number) + if stage: + conditions.append("stage = ?") + params.append(stage) + if session_id: + conditions.append("session_id = ?") + params.append(session_id) + + where_clause = f"WHERE {' AND '.join(conditions)}" if conditions else "" + sql = f""" + SELECT * FROM usage_events + {where_clause} + ORDER BY usage_id ASC + LIMIT ? OFFSET ? + """ + params.extend([limit, offset]) + + with self._tx(immediate=False) as conn: + cursor = conn.execute(sql, params) + rows = cursor.fetchall() + return [dict(row) for row in rows] + # ── sessions ────────────────────────────────────────────────────────── def upsert_session( diff --git a/create_issue_bootstrap.py b/create_issue_bootstrap.py index 80595ba..25a59dd 100644 --- a/create_issue_bootstrap.py +++ b/create_issue_bootstrap.py @@ -247,7 +247,8 @@ def bootstrap_permits_control_checkout( """ if not isinstance(assessment, dict): return False - if not is_create_issue_task(task): + import author_issue_bootstrap + if not is_create_issue_task(task) and not author_issue_bootstrap.is_author_issue_bootstrap_task(task): return False # Positive proof: the assessment must affirmatively allow, with no diff --git a/docs/mcp-tool-inventory.md b/docs/mcp-tool-inventory.md index b1aed12..66094ec 100644 --- a/docs/mcp-tool-inventory.md +++ b/docs/mcp-tool-inventory.md @@ -69,6 +69,7 @@ that gates each call, not which tools exist. - `gitea_audit_worktree_cleanup` - `gitea_authorize_reconciliation_cleanup_phase` - `gitea_authorize_review_correction` +- `gitea_bootstrap_author_issue_worktree` - `gitea_capability_stop_terminal_report` - `gitea_capture_branches_worktree_snapshot` - `gitea_check_pr_eligibility` @@ -100,6 +101,7 @@ that gates each call, not which tools exist. - `gitea_get_profile` - `gitea_get_runtime_context` - `gitea_get_shell_health` +- `gitea_heartbeat_issue_lock` - `gitea_heartbeat_reviewer_pr_lease` - `gitea_inspect_workflow_lease` - `gitea_issue_irrecoverable_provenance_authorization` diff --git a/docs/observability/analytics-instrumentation.md b/docs/observability/analytics-instrumentation.md new file mode 100644 index 0000000..55d7057 --- /dev/null +++ b/docs/observability/analytics-instrumentation.md @@ -0,0 +1,143 @@ +# Model Usage, Token Cost, Latency, and Workflow Analytics (Phase 4) + +- **Tracking Issue:** [#651](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651) +- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651) +- **Console Surface:** `/analytics`, `/api/v1/analytics`, `/api/v1/analytics/usage` + +## 1. Overview + +The Web Console Analytics module provides durable, aggregate visibility into **model usage, token cost, latency percentiles, and workflow-stage performance** across projects, worker roles, AI models, issues, and PRs. + +### Non-Goals +- No mandatory client-side telemetry that leaks prompts or secret keys. +- No third-party payment provider or billing integration. +- No automatic model routing changes without controller policy (#647). + +--- + +## 2. Event Schema (`usage_events`) + +Usage metrics are stored in the control-plane database under table `usage_events`. + +| Column | Type | Description | +|---|---|---| +| `usage_id` | `INTEGER` | Primary key (autoincrement) | +| `session_id` | `TEXT` | Optional active session identifier | +| `remote` | `TEXT` | Known Gitea instance (`dadeschools` or `prgs`) | +| `org` | `TEXT` | Repository owner / organization | +| `repo` | `TEXT` | Repository name | +| `project_id` | `TEXT` | Optional project identifier | +| `role` | `TEXT` | Active worker role (`author`, `reviewer`, `merger`, `reconciler`, `controller`) | +| `model` | `TEXT` | LLM model identifier (e.g. `gemini-3.6-flash`, `claude-3-5-sonnet`) | +| `issue_number` | `INTEGER` | Correlated Gitea issue number (optional) | +| `pr_number` | `INTEGER` | Correlated Gitea PR number (optional) | +| `stage` | `TEXT` | Workflow stage (`preflight`, `implementation`, `review`, `merge`, `reconciliation`) | +| `input_tokens` | `INTEGER` | Input token count (optional / nullable) | +| `output_tokens` | `INTEGER` | Output token count (optional / nullable) | +| `total_tokens` | `INTEGER` | Total token count (optional / nullable) | +| `estimated_cost_usd` | `REAL` | Estimated USD cost (optional / nullable) | +| `latency_ms` | `INTEGER` | Request latency in milliseconds (optional / nullable) | +| `duration_ms` | `INTEGER` | Stage execution duration in milliseconds (optional / nullable) | +| `status` | `TEXT` | Outcome status (`success`, `failure`, `timeout`) | +| `metadata` | `TEXT` | Redacted metadata or summary string | +| `created_at` | `TEXT` | ISO 8601 UTC timestamp | + +--- + +## 3. Handling of Missing Data ("Unknown" vs. Zero Fabrication) + +To ensure operational metrics accurately reflect evidence: +- **Untracked or missing metrics are displayed as `Unknown`**, never zero-fabricated. +- If an event omits `estimated_cost_usd`, `latency_ms`, or token counts, the aggregator marks those fields as missing (`None`) rather than defaulting to `0` or `$0.00`. +- Summary tables and KPI cards explicitly indicate when data is unmeasured or partially reported. + +--- + +## 4. Redaction & Security Rules + +Per `#633` security policy: +- Free-text fields (`metadata`, `prompt_summary`, `session_id`) are run through `console_redaction.redact_text` before persistence and output serialization. +- Secret tokens, keychain commands, authorization headers, passwords, and JWTs are stripped automatically. + +--- + +## 5. Opt-in Instrumentation Guide + +Applications, MCP servers, and background sessions can report usage metrics through either Python API or HTTP ingestion. + +### Python Ingestion + +```python +from webui.analytics_loader import record_usage + +record_usage( + remote="dadeschools", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + role="author", + model="gemini-3.6-flash", + issue_number=651, + stage="implementation", + input_tokens=1420, + output_tokens=380, + total_tokens=1800, + estimated_cost_usd=0.00045, + latency_ms=320, + duration_ms=4500, + status="success", + metadata={"note": "Implementation of analytics module"}, +) +``` + +### HTTP Ingestion API (authorized write) + +`POST /api/v1/analytics/usage` is a **gated write**. It runs through +`console_authz` action `record_analytics_usage` (operator+, Phase 2 execution). +Unauthenticated or phase-inactive requests receive **403** and do not write. +Prefer in-process `record_usage` for MCP / session instrumentation. + +```http +POST /api/v1/analytics/usage HTTP/1.1 +Content-Type: application/json +# Requires authenticated principal with record_analytics_usage execution enabled + +{ + "remote": "dadeschools", + "org": "Scaled-Tech-Consulting", + "repo": "Gitea-Tools", + "role": "author", + "model": "gemini-3.6-flash", + "issue_number": 651, + "stage": "implementation", + "input_tokens": 1420, + "output_tokens": 380, + "total_tokens": 1800, + "estimated_cost_usd": 0.00045, + "latency_ms": 320, + "duration_ms": 4500, + "status": "success", + "metadata": "Analytics schema landed" +} +``` + +### Retention + +`usage_events` is retained with hard caps applied on every write: + +| Limit | Default | +|---|---| +| Max rows | 10,000 (`ControlPlaneDB.USAGE_EVENTS_MAX_ROWS`) | +| Max age | 90 days (`ControlPlaneDB.USAGE_EVENTS_RETENTION_DAYS`) | + +Older rows (by `created_at`) and excess oldest rows (by `usage_id`) are deleted +after each insert so unbounded growth / DoS-by-volume cannot fill the DB. + +--- + +## 6. Querying Analytics API + +```http +GET /api/v1/analytics?role=author&stage=implementation HTTP/1.1 +``` + +Returns `AnalyticsSnapshot` JSON containing aggregations (`by_model`, `by_stage`, `by_role`, `by_work_item`, `by_project`) and latency percentiles (`p50`, `p90`, `p95`, `p99`). diff --git a/docs/post-restart-reconcile.md b/docs/post-restart-reconcile.md new file mode 100644 index 0000000..42511b3 --- /dev/null +++ b/docs/post-restart-reconcile.md @@ -0,0 +1,56 @@ +# Post-restart MCP reconciliation (#662) + +After an MCP process restart, sessions, leases, capabilities, worktrees, and +interrupted mutations must be reconciled before operators claim a clean runtime. +This document describes the #662 completion-proof path. + +## Components + +| Piece | Where | Responsibility | +|-------|-------|----------------| +| `post_restart_reconcile.reconcile_after_restart` | `post_restart_reconcile.py` | Pure classification: inventory → completion proof DTO. No I/O. | +| `RestartCompletionProof` | `post_restart_reconcile.py` | Machine-readable proof (`.as_dict()` is JSON-serializable). | +| `gitea_reconcile_after_restart` | `gitea_mcp_server.py` | MCP tool: gathers inventory from the #613 control-plane DB + master-parity, classifies, returns the proof. Read-only. | +| Boot hook | `gitea_assess_master_parity` | First post-restart parity probe also runs reconcile once (log-only by default). | + +## Dimensions + +The assessor classifies: + +- **service_health** — process healthy / parity mutation-safe +- **clients** — connected client descriptors (optional inventory) +- **sessions** — active session rows with dead owner pids are unresolved +- **checkpoints** — soft-depends on #660; skipped with reason when schema absent +- **leases** — live control-plane leases after restart +- **capabilities** — master-parity / stale-runtime (#610) +- **worktrees** — lease-bound paths missing on disk +- **interrupted_mutations** — mutating lease phases or explicit pending inventory; **never auto-resumed** +- **duplicates** — multiple live claims on the same work item +- **queue** — allocator resume safety + +## Modes + +| Mode | Env / arg | Behavior | +|------|-----------|----------| +| `log_only` (default) | unset or `GITEA_POST_RESTART_RECONCILE_MODE=log_only` | Proof only; `mutation_hold=false` | +| `enforce` | `GITEA_POST_RESTART_RECONCILE_MODE=enforce` or `mode=enforce` | Sets `mutation_hold=true` when overall status is degraded/failed or interrupted mutations remain | + +## Follow-up issues + +Unresolved dimensions produce `proposed_follow_ups` entries suitable for durable +Gitea issues. The MCP tool **does not create** those issues in v1 (rollout is +log-only first). Controllers may file them from the proof payload. + +## Links + +- Umbrella: #655 +- Vision: #652 · Roadmap: #653 +- Checkpoint schema: #660 (soft dependency) +- Drain proof: #661 (soft) +- This issue: #662 + +## Non-goals + +- HA multi-instance failover +- Automatic silent mutation replay +- Implementing the #660 checkpoint schema itself diff --git a/docs/sanctioned-restart-controls.md b/docs/sanctioned-restart-controls.md new file mode 100644 index 0000000..8801695 --- /dev/null +++ b/docs/sanctioned-restart-controls.md @@ -0,0 +1,122 @@ +# Sanctioned restart and graceful reload controls (#642) + +Sessions used to recover MCP connectivity by killing the host daemon +(`pkill -f mcp_server.py`, #630). That is forbidden and stays forbidden: it +kills every namespace on the host, contaminates whichever session survives, and +leaves no audit trail. This document describes the sanctioned replacement, +implemented in `webui/sanctioned_restart.py`. + +## What the console will and will not do + +The console **never** restarts anything. It authorizes an intent, records it, +and hands off to a host supervisor. There is no code path in which the console +sends a signal, spawns a process, or renders a kill command — a regression test +asserts the module contains no `subprocess`, `signal`, `os.kill`, `os.system`, +or `popen` reference, and that no returned payload contains a kill command. + +## Operations + +| Mode | Action | Minimum role | Behaviour | +|------|--------|--------------|-----------| +| `reload` | `system.reload_namespace` | controller | Host supervisor reloads the namespace in place, draining in-flight requests. | +| `restart` | `system.restart_namespace` | admin | Host supervisor restarts the namespace. In-flight requests are lost. | + +Scope is always exactly one namespace. A fleet-wide restart is an explicit +non-goal: `all`, `*`, `fleet`, and an empty scope are refused with +`fleet_scope_not_permitted`, because that is precisely the blast radius the +forbidden kill already had. An unrecognised namespace is refused rather than +passed through to the host. + +## The gate sequence + +`assess_restart_request()` applies every gate in order and reports the first +failure with a stable reason code: + +| Order | Gate | Reason code on failure | +|-------|------|------------------------| +| 1 | Mode is `restart` or `reload` | `unknown_mode` | +| 2 | Scope is a single known namespace | `fleet_scope_not_permitted`, `unknown_namespace` | +| 3 | Principal holds the required console role | `unauthorized` | +| 4 | Confirmation phrase supplied | `confirmation_required` | +| 5 | Confirmation names this namespace and mode | `confirmation_mismatch` | +| 6 | Out-of-band operator authorization present | `operator_authorization_missing` | +| 7 | Runtime is not contaminated | `contaminated_runtime` | +| 8 | Host restart hook configured | `restart_hook_not_configured` | + +Passing every gate yields `host_action_required`, never "restarted". + +### Confirmation binds the namespace + +The required phrase is `" "` — for example +`restart gitea-author`. Binding the namespace into the phrase is the point: a +confirmation typed for one namespace cannot be replayed against another. + +### Operator authorization is not self-assertable + +Host daemon maintenance is authorized out of band through +`GITEA_OPERATOR_DAEMON_MAINTENANCE_AUTHORIZATION`, read from the process +environment and nowhere else (#630; #710 finding F1). A worker session cannot +set an environment variable for an already-running daemon, so this cannot be +faked the way a tool argument could. + +### The host hook + +`GITEA_SANCTIONED_RESTART_HOOK` holds an opaque reference the *host* resolves — +a supervisor label such as a launchd job name, never a command line. With no +hook configured the request is refused; the console does not fall back to a +process kill. The value is read server-side and never rendered to a client. + +## Manual kill remains contamination + +`classify_restart_command()` classifies an operator-proposed recovery command. +A manual `pkill`/`kill`/`killall` of the MCP daemon is contamination, not a +restart: it returns `clean_claim_allowed: false` and builds a durable +contamination marker (redacted command only, never secrets) naming +`system.restart_namespace` as the sanctioned alternative. + +A live, uncleared contamination marker also blocks a restart. This is stricter +than #630's task-scoped gate, which deliberately lets a contaminated worker keep +commenting and handing off: restarting a contaminated runtime would launder the +contamination rather than resolve it. Clear the marker through the reconciler +path first. + +## Post-restart health verification + +After the host supervisor acts, `verify_post_restart_health()` decides whether +the session may claim to be clean: + +| Status | Meaning | Clean claim | +|--------|---------|-------------| +| `clean` | Required tool callable, proven through the live client namespace | Allowed | +| `unproven` | Reported healthy without live client-namespace evidence | Refused | +| `unhealthy` | Probe failed | Refused | + +Only `probe_source=client_namespace` evidence clears a session. Static tool +registration is not proof, and neither is an offline subprocess probe — an IDE +client can hold a registered tool list while live calls fail with +`client is closing: EOF` (see +[`mcp-namespace-health.md`](mcp-namespace-health.md)). + +## Audit + +Every attempt — allowed or denied — is recorded through +`webui.console_audit` with actor, target namespace, mode, result, and reason +code, and is redacted before it is persisted. `system.restart_namespace` is +break-glass, so its records are retained for 730 days. Records carry +`process_kill_executed: false`, which is a fact about the code path rather than +a claim: no such path exists. + +## Environment variables + +| Variable | Purpose | +|----------|---------| +| `GITEA_SANCTIONED_RESTART_HOOK` | Host supervisor reference; absent means restart is refused. | +| `GITEA_OPERATOR_DAEMON_MAINTENANCE_AUTHORIZATION` | Out-of-band operator authorization reference. | +| `WEBUI_AUDIT_LOG` | Console audit sink; absent means records are built but not persisted. | + +## Non-goals + +* No unrestricted `kill` from the UI, in any role, in any phase. +* No fleet-wide restart. +* No silent auto-restart loop: every attempt is confirmed and audited. +* This does not implement the Phase 1 health API (#634). diff --git a/docs/webui-authz-audit.md b/docs/webui-authz-audit.md index 8776429..2dcffac 100644 --- a/docs/webui-authz-audit.md +++ b/docs/webui-authz-audit.md @@ -91,6 +91,9 @@ already define, and a regression test asserts each mapping matches. | `close_pr` | controller | privileged | `gitea.pr.close` | Yes | No | No | 3 | | `merge_pr` | controller | privileged | `gitea.pr.merge` | Yes | **Yes** | **Yes** | 3 | | `delete_branch` | admin | destructive | `gitea.branch.delete` | Yes | **Yes** | **Yes** | 3 | +| `record_analytics_usage` | operator | gated_write | `runtime.record_analytics_usage` | Yes | No | No | 2 | +| `system.reload_namespace` | controller | privileged | `runtime.reload_namespace` | Yes | No | No | 2 | +| `system.restart_namespace` | admin | destructive | `runtime.restart_namespace` | Yes | **Yes** | **Yes** | 2 | **Dual control** means the acting principal may not be the sole authority: a second distinct principal must confirm. **Break-glass** means the action is @@ -102,6 +105,13 @@ honouring it. `delete_branch` is admin-only rather than controller because it is the one irreversible action in the set. +`system.restart_namespace` is admin-only for the same reason: restarting a +namespace drops every in-flight request on it. `system.reload_namespace` drains +first, so it is privileged but not destructive. Neither action is ever executed +by the console — both hand off to a host supervisor, and neither exposes a raw +process kill. See +[`sanctioned-restart-controls.md`](sanctioned-restart-controls.md) (#642). + ### Authorization decision `authorize(action_id, principal, for_execution=False)` returns a decision @@ -199,8 +209,8 @@ breaking the request it describes. | Class | Applies to | Default | |-------|-----------|---------| | `standard` | Routine gated writes | 90 days | -| `privileged` | `review_pr`, `close_pr`, and any unclassifiable action | 365 days | -| `break_glass` | `merge_pr`, `delete_branch` | 730 days | +| `privileged` | `review_pr`, `close_pr`, `system.reload_namespace`, and any unclassifiable action | 365 days | +| `break_glass` | `merge_pr`, `delete_branch`, `system.restart_namespace` | 730 days | Each record carries its own class, day count, and computed `expires_at`, so retention is auditable per record rather than inferred from file age. An diff --git a/gitea_mcp_server.py b/gitea_mcp_server.py index 0199910..5fb0988 100644 --- a/gitea_mcp_server.py +++ b/gitea_mcp_server.py @@ -977,6 +977,32 @@ def _create_issue_bootstrap_assessment( """ import create_issue_bootstrap as _cib + import author_issue_bootstrap as _aib + if _aib.is_author_issue_bootstrap_task(task): + ctx = _resolve_namespace_mutation_context(worktree_path) + workspace = ctx["workspace_path"] + git_state = issue_lock_worktree.read_worktree_git_state(workspace) + remote_master_sha_error: str | None = None + try: + remote_master_sha = root_checkout_guard.resolve_remote_master_sha( + ctx["canonical_repo_root"] + ) + except Exception as exc: + remote_master_sha = None + remote_master_sha_error = ( + f"{type(exc).__name__}: {exc}".strip() or "resolver failed" + ) + return _aib.assess_author_issue_bootstrap( + workspace_path=workspace, + canonical_repo_root=ctx["canonical_repo_root"], + current_branch=git_state.get("current_branch"), + head_sha=git_state.get("head_sha"), + porcelain_status=git_state.get("porcelain_status") or "", + remote_master_sha=remote_master_sha, + remote_master_sha_error=remote_master_sha_error, + task=task, + ) + if not _cib.is_create_issue_task(task): return None @@ -2039,6 +2065,7 @@ import allocator_dependencies # noqa: E402 import dependency_graph # noqa: E402 # #784 durable dependency edges import control_plane_db # noqa: E402 import lease_lifecycle # noqa: E402 +import lease_policy # noqa: E402 import workflow_dashboard # noqa: E402 # #605 live queue/lease dashboard import restart_coordinator # noqa: E402 # #658 MCP restart coordinator/impact import incident_bridge # noqa: E402 @@ -2284,7 +2311,6 @@ import canonical_comment_validator as ccv # noqa: E402 # GITEA_ISSUE_LOCK_DIR, bound to the current MCP session via a per-PID pointer. # Legacy global path retained only for test/doc references — do not seed manually. ISSUE_LOCK_FILE = "/tmp/gitea_issue_lock.json" -WORK_LEASE_TTL_HOURS = 4 AUTHOR_ISSUE_WORK_LEASE = "author_issue_work" VALID_WORK_LEASE_OPERATIONS = frozenset({ AUTHOR_ISSUE_WORK_LEASE, @@ -2599,7 +2625,12 @@ def _build_author_issue_work_lease( host: str | None, ) -> dict: created = _work_lease_now() - expires = created + timedelta(hours=WORK_LEASE_TTL_HOURS) + # #790 Slice A: the window comes from the central policy, not a literal here. + # It is also now a *sliding* window — the lease lives ``initial_ttl_minutes`` + # past its last valid heartbeat rather than a fixed four hours past its + # creation, so an abandoned task stops holding the claim within one TTL. + policy = lease_policy.policy_for(lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK) + expires = created + timedelta(minutes=policy.initial_ttl_minutes) return { "operation_type": AUTHOR_ISSUE_WORK_LEASE, "issue_number": issue_number, @@ -2610,6 +2641,15 @@ def _build_author_issue_work_lease( "created_at": _work_lease_timestamp(created), "expires_at": _work_lease_timestamp(expires), "last_heartbeat_at": _work_lease_timestamp(created), + # #790 AC-N1: the ownership key for this task. Distinct from the recorded + # PID, which is the shared daemon and identifies no individual task. + "task_session_id": issue_lock_store.mint_task_session_id( + AUTHOR_ISSUE_WORK_LEASE + ), + # #790 AC-N8: the explicit lifecycle marker. Its absence — never a + # timestamp comparison — is what makes a lock legacy. + "lifecycle_version": lease_policy.LIFECYCLE_HEARTBEAT_V1, + "heartbeat_count": 1, } @@ -4380,6 +4420,135 @@ def gitea_lock_issue( @mcp.tool() +def gitea_heartbeat_issue_lock( + issue_number: int, + branch_name: str, + task_session_id: str | None = None, + remote: str = "dadeschools", + host: str | None = None, + org: str | None = None, + repo: str | None = None, + worktree_path: str | None = None, + expected_generation: int | None = None, +) -> dict: + """Prove an owned author issue lease is still active (#790 Slice A). + + The task-liveness signal the lifecycle was missing. Before this, an author + lease carried a fixed four-hour expiry that nothing could shorten, and the + only liveness evidence was the recorded PID — the long-lived MCP daemon, + which stays alive across every task it serves and so proved nothing about + whether the authoring task still held the work. + + Each successful call slides the lease ``initial_ttl_minutes`` past *now* + from the central policy, so an actively heartbeating session is never + evicted while an abandoned one releases its claim within one TTL. + + What this tool cannot do, by construction: + + * **Acquire.** It refuses when no durable lock exists. + * **Take over.** Exact issue, branch, realpath-normalized worktree, + claimant username, claimant profile, and recorded task-session identifier + must all match; a superseded session holding an older identifier is + refused. + * **Revive.** A lease already past its grace is not heartbeatable — that + would let a session restore ownership it had stopped proving. It must use + the sanctioned reclaim path, which mints a new generation. + + A lock predating the heartbeat lifecycle is rebound rather than heartbeated: + its exact owner is re-verified and a genuine task-session identifier and + first heartbeat are minted (#790 AC-N8). The rebind is decided server-side + from the durable lifecycle marker; there is no caller-facing switch. + + Args: + issue_number: The locked issue number. + branch_name: The branch recorded on the lock. + task_session_id: The identifier this session received when it acquired + or rebound the lock. It is a fencing token, not an ownership + assertion: it is compared against durable state and can only ever + cause a refusal, never grant anything. Omitted only when rebinding a + legacy lock, which has no identifier yet and mints one. + remote: Known instance — 'dadeschools' or 'prgs'. + host: Override the Gitea host. + org: Override the owner/organization. + repo: Override the repository name. + worktree_path: Author worktree recorded on the lock. + expected_generation: Optional fencing value. The per-issue flock already + serializes the read and the write, so this is for a caller that + wants to pin the generation it last observed across calls; a moved + generation fails closed. + + Returns: + dict with 'success', 'performed', the sliding 'expires_at', + 'last_heartbeat_at', 'lock_generation', 'task_session_id', the applied + 'policy', and post-write 'freshness'; on refusal 'success'/'performed' + False with 'reasons' naming exactly what did not match. + """ + blocked = _profile_permission_block( + task_capability_map.required_permission("heartbeat_issue_lock"), + issue_number=issue_number, + remote=remote, + host=host, + org=org, + repo=repo, + org_explicit=org is not None, + repo_explicit=repo is not None, + ) + if blocked: + return blocked + + resolved_worktree = issue_lock_worktree.resolve_author_worktree_path( + worktree_path, _canonical_local_git_root() + ) + h, o, r = _resolve(remote, host, org, repo) + claimant = _work_lease_claimant(h) + identity = claimant.get("username") + profile = claimant.get("profile") + + existing = _load_existing_issue_lock( + remote=remote, org=o, repo=r, issue_number=issue_number + ) + if not existing: + return { + "success": False, + "performed": False, + "issue_number": issue_number, + "reasons": [ + f"no durable lock for issue #{issue_number}; heartbeat cannot " + "acquire a claim (fail closed)" + ], + } + + if issue_lock_store.is_legacy_lease(existing): + outcome = issue_lock_store.rebind_legacy_lock( + remote=remote, + org=o, + repo=r, + issue_number=issue_number, + branch_name=branch_name, + worktree_path=resolved_worktree, + identity=identity, + profile=profile, + expected_generation=expected_generation, + ) + outcome["operation"] = "legacy_rebind" + return outcome + + outcome = issue_lock_store.heartbeat_session_lock( + remote=remote, + org=o, + repo=r, + issue_number=issue_number, + branch_name=branch_name, + worktree_path=resolved_worktree, + identity=identity, + profile=profile, + task_session_id=str(task_session_id or ""), + expected_generation=expected_generation, + ) + outcome["operation"] = "heartbeat" + return outcome + + @mcp.tool() def gitea_recover_dirty_orphaned_issue_worktree( issue_number: int, @@ -4653,6 +4822,7 @@ def gitea_recover_dirty_orphaned_issue_worktree( ) return result + @mcp.tool() def gitea_rebind_dirty_same_claimant_author_session( issue_number: int, @@ -4909,8 +5079,6 @@ def gitea_rebind_dirty_same_claimant_author_session( } return result - return result - @mcp.tool() def gitea_assess_work_issue_duplicate( @@ -9580,6 +9748,7 @@ def gitea_commit_files( host: str | None = None, org: str | None = None, repo: str | None = None, + worktree_path: str | None = None, ) -> dict: """Commit changes to multiple files in a Gitea repository in a single atomic commit. @@ -9592,10 +9761,46 @@ def gitea_commit_files( host: Override the Gitea host. org: Override the owner/organization. repo: Override the repository name. + worktree_path: Optional worktree path for author mutation context. Returns: dict with success status and commit/branch information. """ + if worktree_path is None: + lock_data = issue_lock_store.read_session_issue_lock() or {} + worktree_path = lock_data.get("worktree_path") + if not worktree_path: + try: + prof = get_profile() + uname = prof.get("username") or prof.get("profile_name") + for path in issue_lock_store.iter_lock_files(): + lk = issue_lock_store.read_lock_file(path) or {} + claimant = lk.get("claimant") or {} + if lk.get("remote") == remote and (claimant.get("username") == uname or lk.get("profile") == prof.get("profile_name")): + issue_lock_store.bind_session_lock(lk, renewal_sanctioned=True) + worktree_path = lk.get("worktree_path") + break + except Exception: + pass + + if worktree_path is None and files: + for f in files: + p = f.get("workspace_path") or f.get("local_path") or "" + if p and os.path.isabs(p): + real_p = os.path.realpath(p) + real_root = os.path.realpath(PROJECT_ROOT) + branches_dir = os.path.join(real_root, "branches") + if real_p.startswith(branches_dir + os.sep): + rel_sub = os.path.relpath(real_p, branches_dir) + wt_folder = rel_sub.split(os.sep)[0] + if wt_folder and wt_folder != "..": + worktree_path = os.path.join(branches_dir, wt_folder) + break + + if worktree_path: + os.environ["GITEA_AUTHOR_WORKTREE"] = worktree_path + os.environ["GITEA_ACTIVE_WORKTREE"] = worktree_path + ok, block_reasons = role_session_router.check_author_mutation_after_reviewer_stop( "commit_files" ) @@ -9608,7 +9813,7 @@ def gitea_commit_files( "reasons": block_reasons, } blocked = _namespace_mutation_block( - "commit_files", commit="", branch="", remote=remote + "commit_files", commit="", branch="", remote=remote, worktree_path=worktree_path ) if blocked: return blocked @@ -9636,7 +9841,7 @@ def gitea_commit_files( ) # #735: forward explicit org/repo into shared anti-stomp preflight. - verify_preflight_purity(remote, task="commit_files", org=org, repo=repo) + verify_preflight_purity(remote=remote, worktree_path=worktree_path, task="commit_files", org=org, repo=repo) processed_files, source_proofs = _prepare_commit_payload_files(files) h, o, r = _resolve(remote, host, org, repo) @@ -9913,6 +10118,96 @@ def gitea_publish_unpublished_issue_branch( } +@mcp.tool() +def gitea_bootstrap_author_issue_worktree( + issue_number: int, + assignment_id: str | None = None, + lease_id: str | None = None, + expected_base_sha: str | None = None, + branch_name: str | None = None, + worktree_path: str | None = None, + idempotency_key: str | None = None, + remote: str = "dadeschools", + host: str | None = None, + org: str | None = None, + repo: str | None = None, + dry_run: bool = False, +) -> dict: + """Bootstrap an allocated author issue branch and registered worktree (#850). + + Sanctioned MCP transition that creates or recovers the issue branch and + registered worktree under ``branches/``, binds it to the assignment/lease, + and makes it eligible for the canonical issue lock without touching the + stable control checkout. + + Args: + issue_number: Allocated issue number to bootstrap. + assignment_id: Optional allocation assignment ID. + lease_id: Optional workflow lease ID. + expected_base_sha: Authoritative expected base SHA / concurrency pin. + branch_name: Optional custom branch name (must match issue- pattern). + worktree_path: Optional custom worktree path under branches/. + idempotency_key: Optional key for idempotent replay/resume. + remote: Known instance — 'dadeschools' or 'prgs'. + host: Override Gitea host. + org: Override Org. + repo: Override Repo. + dry_run: Report planned transition without mutating repository. + """ + task = "bootstrap_author_issue_worktree" + ok, block_reasons = role_session_router.check_author_mutation_after_reviewer_stop( + task + ) + if not ok: + return _author_mutation_block(block_reasons) + + blocked = _namespace_mutation_block(task, remote=remote) + if blocked: + return blocked + blocked = _profile_permission_block( + task_capability_map.required_permission(task), + remote=remote, + host=host, + org=org, + repo=repo, + org_explicit=org is not None, + repo_explicit=repo is not None, + ) + if blocked: + return blocked + + verify_preflight_purity( + remote, + task=task, + org=org, + repo=repo, + ) + + h, o, r = _resolve(remote, host, org, repo) + canonical_root = _canonical_local_git_root() + + import author_issue_bootstrap + + return author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=issue_number, + canonical_repo_root=canonical_root, + assignment_id=assignment_id, + lease_id=lease_id, + expected_base_sha=expected_base_sha, + branch_name=branch_name, + worktree_path=worktree_path, + idempotency_key=idempotency_key, + remote=remote, + host=h, + org=o, + repo=r, + active_identity=_active_username(), + active_profile=_active_profile_name(), + owner_session=_current_session_id(), + dry_run=dry_run, + ) + + # Merge methods supported by the Gitea merge API. _MERGE_METHODS = ("merge", "squash", "rebase") @@ -12014,163 +12309,127 @@ def gitea_reconcile_merged_cleanups( if dry_run: report["dry_run"] = True report["executed"] = False - # #851: surface planned lifecycle order so dry-run matches execute. - report["planned_execution_orders"] = { - str(entry.get("pr_number")): entry.get("planned_execution_order") or [] - for entry in (report.get("entries") or []) - } return {"success": True, "performed": False, **report} verify_preflight_purity( remote, task="reconcile_merged_cleanups", org=org, repo=repo ) actions: list[dict] = [] - project_root = _canonical_local_git_root() - - def _ownership_records_for_branch( - head_branch: str, pr_num_int: int | None - ) -> list[dict]: - ownership_bundle = _collect_branch_ownership_records( - remote=remote, - host=h, - org=o, - repo=r, - branch=head_branch, - pr_number=pr_num_int, - project_root=project_root, - auth=auth, - base_api=base, - ) - ownership_records = list(ownership_bundle.get("records") or []) - if ownership_bundle.get("inventory_error"): - ownership_records.append( - { - "category": ( - branch_cleanup_guard.OWNERSHIP_CATEGORY_INVENTORY_ERROR - ), - "status": "unknown", - "remote": remote, - "host": h, - "org": o, - "repo": r, - "branch": head_branch, - "reclaim_allowed": False, - "role": "inventory", - } - ) - return ownership_records - - def _attempt_owned_remote_delete( - *, - head_branch: str, - pr_num_int: int | None, - after_worktree_removal: bool = False, - ) -> dict: - """Fail-closed remote delete with live ownership reassessment (#851).""" - import urllib.parse - - ownership_records = _ownership_records_for_branch(head_branch, pr_num_int) - ownership = branch_cleanup_guard.assess_active_branch_ownership( - remote=remote, - org=o, - repo=r, - branch=head_branch, - host=h, - records=ownership_records, - ) - if ownership.get("block"): - return { - "action": "delete_remote_branch", - "branch": head_branch, - "success": False, - "performed": False, - "delete_acknowledged": False, - "verified_absent": False, - "blocker_kind": "active_branch_ownership", - "reasons": ownership.get("reasons") or [], - "blocking_categories": ownership.get("blocking_categories") or [], - "after_worktree_removal": after_worktree_removal, - "ownership_reassessed": after_worktree_removal, - } - - encoded = urllib.parse.quote(head_branch, safe="") - url = f"{base}/branches/{encoded}" - with _audited( - "delete_branch", - host=h, - remote=remote, - org=o, - repo=r, - target_branch=head_branch, - request_metadata={ - "branch": head_branch, - "source": "reconcile_merged_cleanups", - "ownership_checked": True, - "after_worktree_removal": after_worktree_removal, - }, - ): - api_request("DELETE", url, auth) - readback = _probe_remote_branch(h, o, r, auth, head_branch) - readback_assessment = branch_cleanup_guard.assess_post_delete_readback( - readback - ) - verified = bool(readback_assessment.get("verified_absent")) - return { - "action": "delete_remote_branch", - "branch": head_branch, - "success": bool(readback_assessment.get("ok")), - "performed": True, - "delete_acknowledged": True, - "verified_absent": verified, - "readback": readback_assessment.get("readback"), - "reasons": readback_assessment.get("reasons") or [], - "after_worktree_removal": after_worktree_removal, - "ownership_reassessed": after_worktree_removal, - } - for entry in report.get("entries") or []: head_branch = entry.get("head_branch") or "" remote_assessment = entry.get("remote_branch") or {} local_assessment = entry.get("local_worktree") or {} - pr_num = entry.get("pr_number") - try: - pr_num_int = int(pr_num) if pr_num is not None else None - except (TypeError, ValueError): - pr_num_int = None - # #851 lifecycle: when the target worktree is independently safe, remove - # it first so worktree_binding ownership does not permanently strand - # both the worktree and the remote branch. Never skip worktree removal - # merely because remote delete would be blocked by that binding. - # Ownership protection for remote delete remains fail-closed below. - worktree_removed = False + if remote_assessment.get("safe_to_delete_remote"): + import urllib.parse + + pr_num = entry.get("pr_number") + try: + pr_num_int = int(pr_num) if pr_num is not None else None + except (TypeError, ValueError): + pr_num_int = None + ownership_bundle = _collect_branch_ownership_records( + remote=remote, + host=h, + org=o, + repo=r, + branch=head_branch, + pr_number=pr_num_int, + project_root=_canonical_local_git_root(), + auth=auth, + base_api=base, + ) + ownership_records = list(ownership_bundle.get("records") or []) + if ownership_bundle.get("inventory_error"): + ownership_records.append( + { + "category": ( + branch_cleanup_guard.OWNERSHIP_CATEGORY_INVENTORY_ERROR + ), + "status": "unknown", + "remote": remote, + "host": h, + "org": o, + "repo": r, + "branch": head_branch, + "reclaim_allowed": False, + "role": "inventory", + } + ) + ownership = branch_cleanup_guard.assess_active_branch_ownership( + remote=remote, + org=o, + repo=r, + branch=head_branch, + host=h, + records=ownership_records, + ) + if ownership.get("block"): + actions.append( + { + "action": "delete_remote_branch", + "branch": head_branch, + "success": False, + "performed": False, + "delete_acknowledged": False, + "verified_absent": False, + "blocker_kind": "active_branch_ownership", + "reasons": ownership.get("reasons") or [], + "blocking_categories": ownership.get( + "blocking_categories" + ) + or [], + } + ) + continue + + encoded = urllib.parse.quote(head_branch, safe="") + url = f"{base}/branches/{encoded}" + with _audited( + "delete_branch", + host=h, + remote=remote, + org=o, + repo=r, + target_branch=head_branch, + request_metadata={ + "branch": head_branch, + "source": "reconcile_merged_cleanups", + "ownership_checked": True, + }, + ): + api_request("DELETE", url, auth) + readback = _probe_remote_branch(h, o, r, auth, head_branch) + readback_assessment = branch_cleanup_guard.assess_post_delete_readback( + readback + ) + verified = bool(readback_assessment.get("verified_absent")) + actions.append( + { + "action": "delete_remote_branch", + "branch": head_branch, + "success": bool(readback_assessment.get("ok")), + "performed": True, + "delete_acknowledged": True, + "verified_absent": verified, + "readback": readback_assessment.get("readback"), + "reasons": readback_assessment.get("reasons") or [], + } + ) + if local_assessment.get("safe_to_remove_worktree"): result = merged_cleanup_reconcile.remove_local_worktree( - project_root, + _canonical_local_git_root(), head_branch, worktree_path=local_assessment.get("worktree_path"), ) actions.append({"action": "remove_local_worktree", **result}) - # Idempotent resume: absent worktree is already gone. - msg = (result.get("message") or "").lower() - worktree_removed = bool(result.get("success")) or ( - "not found" in msg - ) - - if remote_assessment.get("safe_to_delete_remote"): - actions.append( - _attempt_owned_remote_delete( - head_branch=head_branch, - pr_num_int=pr_num_int, - after_worktree_removal=worktree_removed, - ) - ) for scratch in report.get("reviewer_scratch_entries") or []: if not scratch.get("safe_to_remove_worktree"): continue result = merged_cleanup_reconcile.remove_reviewer_scratch_worktree( - project_root, scratch.get("worktree_path") or "" + _canonical_local_git_root(), scratch.get("worktree_path") or "" ) actions.append({"action": "remove_reviewer_scratch_worktree", **result}) @@ -17763,6 +18022,17 @@ def gitea_assess_master_parity( } if parity["restart_required"] and enforced: out["report"] = master_parity_gate.parity_report(parity) + # #662 AC1: first post-restart master-parity probe also runs the boot + # reconcile once (log-only by default; never raises). + boot_proof = _ensure_boot_post_restart_reconcile() + if boot_proof is not None: + out["post_restart_reconcile"] = { + "reconcile_id": boot_proof.get("reconcile_id"), + "overall_status": boot_proof.get("overall_status"), + "mutation_hold": boot_proof.get("mutation_hold"), + "unresolved_count": boot_proof.get("unresolved_count"), + "mode": boot_proof.get("mode"), + } return out @@ -20150,6 +20420,12 @@ def gitea_resolve_task_capability( task: str, remote: str = "dadeschools", host: str | None = None, + org: str | None = None, + repo: str | None = None, + issue_number: int | None = None, + worktree_path: str | None = None, + pr_number: int | None = None, + **kwargs: Any, ) -> dict: """Read-only / side-effect free: resolve capability, profile, and namespace for a task. @@ -20166,13 +20442,14 @@ def gitea_resolve_task_capability( remote: Known remote instance name. host: Optional override for the Gitea host. """ + task_key = task_capability_map._canonical_preflight_task(task) TASK_MAP = task_capability_map.TASK_CAPABILITY_MAP # Every fresh attempt invalidates the previous task/role stamp before any # fallible resolver work. Unknown/malformed tasks and unexpected failures # therefore remain fail-closed instead of preserving stale authority. _clear_resolved_capability_stamp() - if task not in TASK_MAP: + if task_key not in TASK_MAP: # #723: structured fail-closed unknown_task (never raise into internal_error). profile = get_profile() h = host or (REMOTES.get(remote, {}).get("host") if remote in REMOTES else None) @@ -20218,8 +20495,8 @@ def gitea_resolve_task_capability( result["cleared_stale_denial"] = True return result - required_permission = task_capability_map.required_permission(task) - required_role = task_capability_map.required_role(task) + required_permission = task_capability_map.required_permission(task_key) + required_role = task_capability_map.required_role(task_key) role_exclusive_tasks = task_capability_map.ROLE_EXCLUSIVE_TASKS infra_assessment = role_session_router.assess_infra_stop(PROJECT_ROOT) @@ -20405,7 +20682,10 @@ def gitea_resolve_task_capability( available_in_session = allowed_in_current_session runtime_stale_blocker = False - if "PYTEST_CURRENT_TEST" not in os.environ or "GITEA_FORCE_MCP_RUNTIME_CHECK" in os.environ: + if ( + "PYTEST_CURRENT_TEST" not in os.environ + or "GITEA_FORCE_MCP_RUNTIME_CHECK" in os.environ + ) and os.environ.get("GITEA_ALLOW_STALE_RUNTIME") != "1": runtime_reasons = _check_mcp_runtimes_diagnostics(task, matching_profiles) if runtime_reasons: restart_required = True @@ -22201,6 +22481,249 @@ def gitea_request_mcp_restart( return payload +# --- #662 post-restart reconciliation --------------------------------------- + +_POST_RESTART_LAST_PROOF: dict | None = None +_POST_RESTART_BOOT_RAN = False + + +def _post_restart_reconcile_mode() -> str: + """Return log_only (default) or enforce from process environment.""" + raw = (os.environ.get("GITEA_POST_RESTART_RECONCILE_MODE") or "").strip().lower() + if raw in {"enforce", "enforced", "hold"}: + return "enforce" + return "log_only" + + +def _gather_post_restart_inventory( + *, + remote: str, + org: str, + repo: str, + limit: int = 200, +) -> dict: + """Gather live control-plane facts for post-restart reconcile (#662). + + Best-effort and fail-closed: any inventory section that cannot be read is + recorded on ``incomplete_reasons`` and ``inventory_complete`` is cleared. + Never mutates Gitea or the control-plane DB. + """ + import master_parity_gate + import post_restart_reconcile as prr + + inventory_complete = True + incomplete_reasons: list[str] = [] + sessions: list[dict] = [] + leases: list[dict] = [] + worktree_bindings: list[dict] = [] + + db, db_errs = _control_plane_db_or_error() + if db is None: + inventory_complete = False + incomplete_reasons.extend( + db_errs or ["control-plane DB unavailable; cannot reconcile after restart"] + ) + else: + try: + sessions = db.list_sessions(statuses=("active",), limit=max(1, int(limit))) + except Exception as exc: # noqa: BLE001 + inventory_complete = False + incomplete_reasons.append( + f"session inventory failed: {_redact(str(exc))}" + ) + try: + lease_result = lease_lifecycle.list_active_leases( + db, + remote=remote if remote in REMOTES else remote, + org=org, + repo=repo, + role=None, + include_non_active=True, + limit=max(1, int(limit)), + ) + leases = list(lease_result.get("leases") or []) + except Exception as exc: # noqa: BLE001 + inventory_complete = False + incomplete_reasons.append( + f"lease inventory failed: {_redact(str(exc))}" + ) + leases = [] + + # Derive worktree bindings from live leases that carry a path. + for row in leases: + wt = (row.get("worktree_path") or "").strip() + if not wt: + continue + exists = bool(lease_lifecycle.worktree_exists(wt)) + worktree_bindings.append( + { + "path": wt, + "exists": exists, + "missing": not exists, + "lease_id": row.get("lease_id"), + "session_id": row.get("session_id"), + "work_kind": row.get("work_kind"), + "work_number": row.get("work_number"), + } + ) + + # Capability / master-parity dimension (code parity after restart). + try: + parity = _current_master_parity() + capabilities = { + "stale": bool(parity.get("stale")), + "startup_head": parity.get("startup_head"), + "current_head": parity.get("current_head"), + "in_parity": parity.get("in_parity"), + "mutation_safe": parity.get("mutation_safe"), + } + service_health = { + "healthy": bool(parity.get("mutation_safe", not parity.get("stale"))), + "parity_summary": master_parity_gate.format_parity(parity), + } + boot_head = parity.get("startup_head") or _process_boot_head_sha + current_head = parity.get("current_head") + except Exception as exc: # noqa: BLE001 + inventory_complete = False + incomplete_reasons.append( + f"master-parity inventory failed: {_redact(str(exc))}" + ) + capabilities = {"stale": None} + service_health = {"healthy": None, "error": "parity assessment failed"} + boot_head = _process_boot_head_sha + current_head = None + + # #660 soft dependency: checkpoint schema not landed → skip dimension. + checkpoints_available = False + try: + import importlib + + importlib.import_module("session_checkpoint_schema") + checkpoints_available = True + except Exception: # noqa: BLE001 + checkpoints_available = False + + return { + "inventory_complete": inventory_complete, + "incomplete_reasons": incomplete_reasons, + "service_health": service_health, + "clients": [], # client transport inventory is host/IDE-owned + "sessions": sessions, + "leases": leases, + "checkpoints_available": checkpoints_available, + "checkpoints": [] if checkpoints_available else None, + "worktree_bindings": worktree_bindings, + "pending_mutations": [], + "capabilities": capabilities, + "boot_head_sha": boot_head, + "current_head_sha": current_head, + "queue_state": {"safe_to_resume": inventory_complete}, + "reconcile_version": prr.RECONCILE_VERSION, + } + + +def _run_post_restart_reconcile( + *, + remote: str = "prgs", + org: str | None = None, + repo: str | None = None, + mode: str | None = None, + limit: int = 200, +) -> dict: + """Gather + classify post-restart state; cache the latest proof (#662).""" + global _POST_RESTART_LAST_PROOF, _POST_RESTART_BOOT_RAN + import post_restart_reconcile as prr + + try: + _h, o, r = _resolve(remote, None, org, repo) + except ValueError as exc: + return { + "success": False, + "read_only": True, + "reasons": [str(exc)], + } + + inventory = _gather_post_restart_inventory( + remote=remote, org=o, repo=r, limit=limit + ) + proof = prr.reconcile_after_restart( + inventory, + mode=mode or _post_restart_reconcile_mode(), + ) + payload = proof.as_dict() + payload["success"] = True + payload["read_only"] = True + payload["remote"] = remote + payload["org"] = o + payload["repo"] = r + payload["follow_up_create_supported"] = False + payload["follow_up_create_note"] = ( + "proposed_follow_ups lists durable issues the apply path may create; " + "this tool never creates them (log-only by default, #662 rollout)" + ) + _POST_RESTART_LAST_PROOF = payload + _POST_RESTART_BOOT_RAN = True + return payload + + +def _ensure_boot_post_restart_reconcile() -> dict | None: + """Run post-restart reconcile once per process (boot hook, #662 AC1).""" + global _POST_RESTART_BOOT_RAN + if _POST_RESTART_BOOT_RAN: + return _POST_RESTART_LAST_PROOF + # Best-effort: never raise from the boot hook. + try: + return _run_post_restart_reconcile() + except Exception: # noqa: BLE001 + _POST_RESTART_BOOT_RAN = True + return None + + +@mcp.tool() +def gitea_reconcile_after_restart( + remote: str = "dadeschools", + host: str | None = None, + org: str | None = None, + repo: str | None = None, + mode: str | None = None, + limit: int = 200, +) -> dict: + """Run post-restart MCP reconciliation and return a completion proof (#662). + + Gathers live control-plane sessions, leases, worktree bindings, and + master-parity evidence, then classifies them with the pure + ``post_restart_reconcile.reconcile_after_restart`` assessor. Returns a + machine-readable completion proof listing resolved / unresolved dimensions + and proposed durable follow-up issues. + + Read-only by design: never restarts MCP, never auto-resumes write + mutations, and never creates Gitea issues (those are a separate apply + path). Default mode is ``log_only``; set + ``GITEA_POST_RESTART_RECONCILE_MODE=enforce`` (or pass ``mode='enforce'``) + to set ``mutation_hold`` when anything remains unresolved. + + Soft-depends on #660 for session checkpoints: when the checkpoint schema + module is absent the checkpoints dimension is ``skipped`` with an explicit + reason rather than inventing a schema. + """ + read_block = _profile_operation_gate("gitea.read") + if read_block: + return { + "success": False, + "read_only": True, + "reasons": read_block, + "permission_report": _permission_block_report("gitea.read"), + } + + return _run_post_restart_reconcile( + remote=remote, + org=org, + repo=repo, + mode=mode, + limit=limit, + ) + + @mcp.tool() def gitea_inspect_workflow_lease( lease_id: str, diff --git a/issue_lock_store.py b/issue_lock_store.py index d011c1c..f216656 100644 --- a/issue_lock_store.py +++ b/issue_lock_store.py @@ -15,15 +15,27 @@ import json import os import re import tempfile +import uuid from contextlib import contextmanager from datetime import datetime, timedelta, timezone from typing import Any +import lease_policy + LOCK_DIR_ENV = "GITEA_ISSUE_LOCK_DIR" DEFAULT_LOCK_DIR = os.path.expanduser("~/.cache/gitea-tools/issue-locks") -WORK_LEASE_TTL_HOURS = 4 AUTHOR_ISSUE_WORK_LEASE = "author_issue_work" +# Freshness classifications. ``STATUS_STALE`` remains the dead-PID band that +# #753 recovery keys on; the two bands below are new in #790 Slice A and apply +# only to leases minted under the heartbeat lifecycle. +STATUS_LIVE = "live" +STATUS_EXPIRED = "expired" +STATUS_ABSENT = "absent" +STATUS_STALE = "stale" +STATUS_STALE_MISSED_HEARTBEAT = "stale_missed_heartbeat" +STATUS_STALE_ABSOLUTE_CAP = "stale_absolute_cap" + _SAFE_SEGMENT_RE = re.compile(r"[^A-Za-z0-9._+-]+") @@ -257,6 +269,331 @@ def bind_session_lock( return path +def _ownership_refusals( + lock: dict[str, Any], + *, + issue_number: int, + branch_name: str, + worktree_path: str, + identity: str | None, + profile: str | None, +) -> list[str]: + """Exact-ownership mismatches between a durable lock and a live caller. + + Shared by the heartbeat writer and the legacy rebind path so the two cannot + disagree about what "the same owner" means. Every field is compared against + durable state; nothing is taken on the caller's word beyond the identity the + server itself resolved. + """ + reasons: list[str] = [] + if lock.get("issue_number") != issue_number: + reasons.append( + f"lock targets issue #{lock.get('issue_number')}, not #{issue_number}" + ) + if str(lock.get("branch_name") or "") != str(branch_name or ""): + reasons.append( + f"lock branch '{lock.get('branch_name')}' does not match '{branch_name}'" + ) + if not _same_realpath(str(lock.get("worktree_path") or ""), worktree_path): + reasons.append( + f"lock worktree '{lock.get('worktree_path')}' does not match " + f"'{worktree_path}'" + ) + lease = lock.get("work_lease") if isinstance(lock, dict) else None + claimant = lease.get("claimant") if isinstance(lease, dict) else None + claimant = claimant if isinstance(claimant, dict) else {} + recorded_identity = str(claimant.get("username") or "").strip() + recorded_profile = str(claimant.get("profile") or "").strip() + if not recorded_identity or not recorded_profile: + reasons.append("lock does not record both a claimant username and profile") + if recorded_identity and recorded_identity != str(identity or "").strip(): + reasons.append( + f"lock claimant '{recorded_identity}' does not match active identity " + f"'{str(identity or '').strip() or 'unknown'}'" + ) + if recorded_profile and recorded_profile != str(profile or "").strip(): + reasons.append( + f"lock profile '{recorded_profile}' does not match active profile " + f"'{str(profile or '').strip() or 'unknown'}'" + ) + return reasons + + +def _refusal(reasons: list[str], **extra: Any) -> dict[str, Any]: + return {"success": False, "performed": False, "reasons": reasons, **extra} + + +def heartbeat_session_lock( + *, + remote: str, + org: str, + repo: str, + issue_number: int, + branch_name: str, + worktree_path: str, + identity: str | None, + profile: str | None, + task_session_id: str, + expected_generation: int | None = None, + lock_dir: str | None = None, + now: datetime | None = None, +) -> dict[str, Any]: + """Slide a heartbeat-lifecycle lease forward (#790 Slice A, A4). + + The write happens inside the same per-issue ``flock`` that serializes + acquisition, and under the #772 generation compare-and-swap, so a heartbeat + can never race a concurrent reclaim: whichever lands first moves the + generation and the other fails closed. + + Refuses — never revives — in every ambiguous case. A lease that has already + lapsed past its grace is *not* heartbeatable: allowing that would let a + session that stopped proving liveness restore ownership retroactively, which + is precisely the revival AC-N5 forbids. Such a session must go through the + sanctioned reclaim path, which mints a fresh generation. + """ + current = _lease_now(now) + root = _ensure_lock_dir(lock_dir) + path = lock_file_path( + remote=remote, org=org, repo=repo, issue_number=issue_number, lock_dir=root + ) + declared_session = str(task_session_id or "").strip() + if not declared_session: + return _refusal(["no task_session_id supplied (fail closed)"]) + + sentinel = flock_path(path) + try: + with _exclusive_file_lock(sentinel): + lock = read_lock_file(path) + if not lock: + return _refusal([f"no durable lock for issue #{issue_number}"]) + + if is_legacy_lease(lock): + return _refusal( + [ + "lock predates the heartbeat lifecycle; it must be rebound " + "by its exact owner before it can be heartbeated" + ], + lifecycle=lease_lifecycle_version(lock), + legacy_lease=True, + ) + + reasons = _ownership_refusals( + lock, + issue_number=issue_number, + branch_name=branch_name, + worktree_path=worktree_path, + identity=identity, + profile=profile, + ) + recorded_session = lease_task_session_id(lock) + if not recorded_session: + reasons.append( + "lock declares the heartbeat lifecycle but records no " + "task_session_id (fail closed)" + ) + elif recorded_session != declared_session: + # A superseded session holding an old identifier cannot heartbeat + # over the session that replaced it. + reasons.append( + "task_session_id does not match the session recorded on the lock" + ) + if reasons: + return _refusal(reasons) + + current_generation = lock_generation(lock) + if ( + expected_generation is not None + and current_generation != expected_generation + ): + return _refusal( + [ + f"lock generation changed: expected {expected_generation}, " + f"found {current_generation}; another session reclaimed or " + "replaced this claim (fail closed)" + ], + lock_generation=current_generation, + ) + + freshness = assess_lock_freshness(lock, now=current) + if not freshness.get("live"): + return _refusal( + [ + f"lease is not live ({freshness.get('status')}): " + f"{freshness.get('reason')}; a lapsed lease must be " + "reclaimed, not heartbeated" + ], + freshness=freshness, + ) + + policy = lease_policy.policy_for(lease_task_class(lock)) + expires = current + timedelta(minutes=policy.initial_ttl_minutes) + record = dict(lock) + lease = dict(record.get("work_lease") or {}) + prior_heartbeat = lease.get("last_heartbeat_at") + lease["last_heartbeat_at"] = _format_lease_timestamp(current) + lease["expires_at"] = _format_lease_timestamp(expires) + try: + lease["heartbeat_count"] = int(lease.get("heartbeat_count") or 0) + 1 + except (TypeError, ValueError): + lease["heartbeat_count"] = 1 + record["work_lease"] = lease + record["lock_generation"] = current_generation + 1 + save_lock_file(path, record) + except LockContentionError as exc: + return _refusal([f"issue #{issue_number} lock contention: {exc} (fail closed)"]) + + return { + "success": True, + "performed": True, + "issue_number": issue_number, + "branch_name": branch_name, + "worktree_path": worktree_path, + "task_session_id": declared_session, + "lock_generation": record["lock_generation"], + "prior_generation": current_generation, + "prior_heartbeat_at": prior_heartbeat, + "last_heartbeat_at": lease["last_heartbeat_at"], + "expires_at": lease["expires_at"], + "heartbeat_count": lease["heartbeat_count"], + "lock_file_path": path, + "policy": lease_policy.describe(lease_task_class(record)), + "freshness": assess_lock_freshness(record, now=current), + } + + +def rebind_legacy_lock( + *, + remote: str, + org: str, + repo: str, + issue_number: int, + branch_name: str, + worktree_path: str, + identity: str | None, + profile: str | None, + expected_generation: int | None = None, + lock_dir: str | None = None, + now: datetime | None = None, +) -> dict[str, Any]: + """Move a legacy lock into the heartbeat lifecycle (#790 AC-N8). + + One of the two sanctioned exits from the preserved-expiry legacy state; the + other is terminal retirement, which is Slice B. Only the exact recorded + owner may rebind, and only while the legacy lock is still live under its + original absolute expiry — an already-expired legacy lease belongs to the + #760 renewal path or #601 reclaim, and this must not become a second, weaker + way to revive one. + + The rebind mints a genuine task-session identifier and a genuine first + heartbeat. It does not fabricate history: the original creation and expiry + are preserved under ``legacy_origin`` for audit, and the new lifecycle's + absolute cap runs from the rebind, not from the legacy claim. + """ + current = _lease_now(now) + root = _ensure_lock_dir(lock_dir) + path = lock_file_path( + remote=remote, org=org, repo=repo, issue_number=issue_number, lock_dir=root + ) + sentinel = flock_path(path) + try: + with _exclusive_file_lock(sentinel): + lock = read_lock_file(path) + if not lock: + return _refusal([f"no durable lock for issue #{issue_number}"]) + if not is_legacy_lease(lock): + return _refusal( + [ + "lock is already on the heartbeat lifecycle; use the " + "heartbeat path" + ], + lifecycle=lease_lifecycle_version(lock), + legacy_lease=False, + ) + + reasons = _ownership_refusals( + lock, + issue_number=issue_number, + branch_name=branch_name, + worktree_path=worktree_path, + identity=identity, + profile=profile, + ) + if reasons: + return _refusal(reasons) + + current_generation = lock_generation(lock) + if ( + expected_generation is not None + and current_generation != expected_generation + ): + return _refusal( + [ + f"lock generation changed: expected {expected_generation}, " + f"found {current_generation} (fail closed)" + ], + lock_generation=current_generation, + ) + + freshness = assess_lock_freshness(lock, now=current) + if not freshness.get("live"): + return _refusal( + [ + f"legacy lease is not live ({freshness.get('status')}): " + f"{freshness.get('reason')}; rebinding is not a recovery " + "path for a lapsed lease" + ], + freshness=freshness, + ) + + policy = lease_policy.policy_for(lease_task_class(lock)) + expires = current + timedelta(minutes=policy.initial_ttl_minutes) + session_id = mint_task_session_id(lease_task_class(lock)) + record = dict(lock) + lease = dict(record.get("work_lease") or {}) + legacy_origin = { + "created_at": lease.get("created_at"), + "expires_at": lease.get("expires_at"), + "last_heartbeat_at": lease.get("last_heartbeat_at"), + "lifecycle": lease_policy.LIFECYCLE_LEGACY, + } + lease["lifecycle_version"] = lease_policy.LIFECYCLE_HEARTBEAT_V1 + lease["task_session_id"] = session_id + lease["created_at"] = _format_lease_timestamp(current) + lease["last_heartbeat_at"] = _format_lease_timestamp(current) + lease["expires_at"] = _format_lease_timestamp(expires) + lease["heartbeat_count"] = 1 + record["work_lease"] = lease + record["legacy_rebind"] = { + "rebound_at": _format_lease_timestamp(current), + "task_session_id": session_id, + "prior_generation": current_generation, + "legacy_origin": legacy_origin, + "reason": ( + "legacy lock rebound into the heartbeat lifecycle by its exact " + "recorded owner" + ), + } + record["lock_generation"] = current_generation + 1 + save_lock_file(path, record) + except LockContentionError as exc: + return _refusal([f"issue #{issue_number} lock contention: {exc} (fail closed)"]) + + return { + "success": True, + "performed": True, + "issue_number": issue_number, + "task_session_id": session_id, + "lock_generation": record["lock_generation"], + "prior_generation": current_generation, + "lifecycle": lease_policy.LIFECYCLE_HEARTBEAT_V1, + "legacy_rebind": record["legacy_rebind"], + "expires_at": lease["expires_at"], + "last_heartbeat_at": lease["last_heartbeat_at"], + "lock_file_path": path, + "freshness": assess_lock_freshness(record, now=current), + } + + def read_session_issue_lock(lock_dir: str | None = None) -> dict[str, Any] | None: root = (lock_dir or default_lock_dir()).strip() pointer = read_lock_file(session_pointer_path(root)) @@ -340,6 +677,16 @@ def _parse_lease_timestamp(value: str | None) -> datetime | None: return None +def _format_lease_timestamp(value: datetime) -> str: + """Serialize a lease timestamp in the durable ``...Z`` form already on disk.""" + return ( + value.astimezone(timezone.utc) + .replace(microsecond=0) + .isoformat() + .replace("+00:00", "Z") + ) + + def lease_expires_at(lock: dict[str, Any] | None) -> datetime | None: if not lock: return None @@ -360,26 +707,113 @@ def is_lease_live(lock: dict[str, Any] | None, *, now: datetime | None = None) - return assess_lock_freshness(lock, now=now)["live"] +def lease_task_class(lock_data: dict[str, Any] | None) -> str: + """Policy task class for a durable lock; author work when unrecorded.""" + lease = lock_data.get("work_lease") if isinstance(lock_data, dict) else None + if isinstance(lease, dict): + recorded = str(lease.get("operation_type") or "").strip() + if recorded: + return recorded + return AUTHOR_ISSUE_WORK_LEASE + + +def lease_lifecycle_version(lock_data: dict[str, Any] | None) -> str: + """Read the durable lifecycle marker (#790 AC-N8). + + The marker is the *only* discriminator between a heartbeat-lifecycle lease + and a legacy one. Timestamps are deliberately not consulted: a lock minted + before this lifecycle existed has ``last_heartbeat_at == created_at`` + forever, and reading that equality as "recently heartbeated" would treat + every never-heartbeated legacy lock as fresh — the precise inversion AC-N8 + forbids. A newly minted heartbeat lease also has the two equal, so the + equality carries no information in either direction. + """ + lease = lock_data.get("work_lease") if isinstance(lock_data, dict) else None + if isinstance(lease, dict): + recorded = str(lease.get("lifecycle_version") or "").strip() + if recorded: + return recorded + return lease_policy.LIFECYCLE_LEGACY + + +def is_legacy_lease(lock_data: dict[str, Any] | None) -> bool: + """True when a lock predates the shared heartbeat lifecycle.""" + return lease_lifecycle_version(lock_data) != lease_policy.LIFECYCLE_HEARTBEAT_V1 + + +def lease_task_session_id(lock_data: dict[str, Any] | None) -> str: + """Recorded per-task session identifier, or empty for a legacy lock.""" + lease = lock_data.get("work_lease") if isinstance(lock_data, dict) else None + if isinstance(lease, dict): + return str(lease.get("task_session_id") or "").strip() + return "" + + +def mint_task_session_id(task_class: str = AUTHOR_ISSUE_WORK_LEASE) -> str: + """Mint an ownership key for one task (#790 AC-N1). + + Deliberately contains no process identifier. The recorded PID belongs to the + long-lived MCP daemon, which outlives any individual task and is reused by + every task it serves, so PID digits cannot identify *which* task holds a + claim. The PID is still recorded alongside this value as evidence. + """ + prefix = _sanitize_segment(str(task_class or AUTHOR_ISSUE_WORK_LEASE)) + return f"{prefix}-{uuid.uuid4().hex[:16]}" + + +def _lease_heartbeat_at(lock_data: dict[str, Any] | None) -> datetime | None: + lease = lock_data.get("work_lease") if isinstance(lock_data, dict) else None + heartbeat_at = None + if isinstance(lock_data, dict): + heartbeat_at = _parse_lease_timestamp(lock_data.get("last_heartbeat_at")) + if heartbeat_at is None and isinstance(lease, dict): + heartbeat_at = _parse_lease_timestamp(lease.get("last_heartbeat_at")) + return heartbeat_at + + def assess_lock_freshness( lock_data: dict[str, Any] | None, *, now: datetime | None = None, ) -> dict[str, Any]: - """Classify a lock as live, expired, stale, or absent.""" + """Classify a lock as live, expired, stale, or absent. + + #790 Slice A makes the heartbeat load-bearing. Before this change + ``last_heartbeat_at`` was parsed and then never consulted: liveness was + decided entirely by the absolute ``expires_at`` and by PID liveness, and + since the recorded PID is the long-lived MCP daemon, an abandoned author + task stayed "live" for the full four-hour TTL. + + Two rules govern the rewrite: + + * **An alive PID never establishes freshness** (AC-N2). It proves the daemon + is up, nothing about the task. It is recorded as evidence and no branch + returns ``live`` because of it. + * **A dead PID still corroborates staleness.** The dead-PID band is + unchanged and still precedes every heartbeat evaluation, so #753 + dead-session recovery keys on exactly the classification it always did. + + Legacy leases (AC-N8) keep their recorded absolute expiry and are never + evaluated against the short heartbeat grace, so deploying this change cannot + make an existing claim instantly reclaimable. + """ current = _lease_now(now) if not lock_data: return { - "status": "absent", + "status": STATUS_ABSENT, "live": False, "stale": False, "reason": "no lock record", } - expires_at = lease_expires_at(lock_data) lease = lock_data.get("work_lease") - heartbeat_at = _parse_lease_timestamp(lock_data.get("last_heartbeat_at")) - if heartbeat_at is None and isinstance(lease, dict): - heartbeat_at = _parse_lease_timestamp(lease.get("last_heartbeat_at")) + expires_at = lease_expires_at(lock_data) + heartbeat_at = _lease_heartbeat_at(lock_data) + created_at = ( + _parse_lease_timestamp(lease.get("created_at")) + if isinstance(lease, dict) + else None + ) pid = lock_data.get("session_pid") if pid is None: @@ -397,56 +831,103 @@ def assess_lock_freshness( pid_int = None pid_alive = is_process_alive(pid_int) if pid_int is not None else False - if expires_at and expires_at <= current: - return { - "status": "expired", - "live": False, - "stale": True, - "reason": f"lease expired at {expires_at.isoformat()}", - "pid_alive": pid_alive, - "pid_missing": pid_missing, - } + lifecycle = lease_lifecycle_version(lock_data) + legacy = lifecycle != lease_policy.LIFECYCLE_HEARTBEAT_V1 + policy = lease_policy.policy_for(lease_task_class(lock_data)) - # #860: a PID-less lock must never be considered live merely because - # expiration / heartbeat fields are absent. Missing PID is insufficient - # evidence of a live owner; treat as malformed/stale so recovery routes - # can evaluate corroborating pins instead of blocking on a false live flag. - if pid_missing: - return { - "status": "malformed", - "live": False, - "stale": True, - "reason": ( - "lock has no usable session pid; cannot prove live ownership " - "(PID-less locks are never live by missing expiry alone)" - ), - "pid_alive": False, - "pid_missing": True, - "heartbeat_at": heartbeat_at.isoformat() if heartbeat_at else None, - "expires_at": expires_at.isoformat() if expires_at else None, - } - - if pid_int is not None and not pid_alive: - return { - "status": "stale", - "live": False, - "stale": True, - "reason": f"owner pid {pid_int} is not alive", - "pid_alive": False, - "pid_missing": False, - } - - return { - "status": "live", - "live": True, - "stale": False, - "reason": "lock heartbeat and lease are fresh", + evidence: dict[str, Any] = { "pid_alive": pid_alive, - "pid_missing": False, + "pid_missing": pid_missing, + "lifecycle": lifecycle, + "legacy_lease": legacy, + "task_session_id": lease_task_session_id(lock_data) or None, "heartbeat_at": heartbeat_at.isoformat() if heartbeat_at else None, "expires_at": expires_at.isoformat() if expires_at else None, } + def _result(status: str, *, live: bool, reason: str, **extra: Any) -> dict[str, Any]: + return { + "status": status, + "live": live, + "stale": not live and status != STATUS_ABSENT, + "reason": reason, + **evidence, + **extra, + } + + if legacy: + # AC-N8: the preserved absolute expiry is the only clock for a lock + # written before task-session heartbeats existed. + if expires_at and expires_at <= current: + return _result( + STATUS_EXPIRED, + live=False, + reason=f"lease expired at {expires_at.isoformat()}", + ) + if pid is not None and not pid_alive: + return _result( + STATUS_STALE, live=False, reason=f"owner pid {pid} is not alive" + ) + return _result( + STATUS_LIVE, + live=True, + reason=( + "legacy lease is within its recorded absolute expiry; the " + "heartbeat grace does not apply retroactively" + ), + legacy_expiry_preserved=True, + ) + + # ── Heartbeat lifecycle ── + if pid is not None and not pid_alive: + # Unchanged dead-PID band: #753 recovery depends on this exact status. + return _result(STATUS_STALE, live=False, reason=f"owner pid {pid} is not alive") + + if heartbeat_at is None: + # Contradictory: a heartbeat lease must carry a heartbeat. Fail closed. + return _result( + STATUS_STALE_MISSED_HEARTBEAT, + live=False, + reason=( + f"lease declares lifecycle '{lifecycle}' but records no " + "last_heartbeat_at (fail closed)" + ), + ) + + if policy.absolute_cap_hours and created_at is not None: + cap_at = created_at + timedelta(hours=policy.absolute_cap_hours) + if cap_at <= current: + return _result( + STATUS_STALE_ABSOLUTE_CAP, + live=False, + reason=( + f"lease exceeded its {policy.absolute_cap_hours}h absolute cap " + f"at {cap_at.isoformat()}; canonical re-adoption is required" + ), + absolute_cap_at=cap_at.isoformat(), + ) + + grace_at = heartbeat_at + timedelta(minutes=policy.missed_heartbeat_grace_minutes) + if grace_at <= current or (expires_at is not None and expires_at <= current): + return _result( + STATUS_STALE_MISSED_HEARTBEAT, + live=False, + reason=( + f"no valid heartbeat since {heartbeat_at.isoformat()}; the " + f"{policy.missed_heartbeat_grace_minutes}min grace lapsed at " + f"{grace_at.isoformat()}" + ), + missed_heartbeat_since=grace_at.isoformat(), + ) + + warning_at = heartbeat_at + timedelta(minutes=policy.stale_warning_minutes) + return _result( + STATUS_LIVE, + live=True, + reason="lease heartbeat is fresh within the configured grace", + heartbeat_warning=warning_at <= current, + ) + def _same_realpath(left: str | None, right: str | None) -> bool: if not left or not right: @@ -483,6 +964,27 @@ def assess_expired_lock_reclaim( "reasons": ["lock is still live; cannot reclaim (fail closed)"], "freshness": freshness, } + status = str(freshness.get("status") or "") + if status in (STATUS_STALE_MISSED_HEARTBEAT, STATUS_STALE_ABSOLUTE_CAP): + # #790: under the heartbeat lifecycle the heartbeat *is* the liveness + # proof, so a session that stopped heartbeating past its grace has + # released its claim by definition. Requiring a dead PID on top of that + # would reinstate the original defect — the recorded PID is the shared + # daemon, which stays alive across every abandoned task it ever served. + # + # This band is unreachable for a legacy lease (AC-N8), so no lock + # written before this lifecycle can be reclaimed by this path. + return { + "reclaim_allowed": True, + "reasons": [ + f"heartbeat-lifecycle lease is {status}: {freshness.get('reason')}" + ], + "freshness": freshness, + "prior_branch": existing_lock.get("branch_name"), + "prior_worktree": existing_lock.get("worktree_path"), + "prior_pid": existing_lock.get("session_pid") or existing_lock.get("pid"), + "prior_task_session_id": lease_task_session_id(existing_lock) or None, + } pid = existing_lock.get("session_pid") if pid is None: pid = existing_lock.get("pid") @@ -557,7 +1059,28 @@ def assess_same_issue_lease_conflict( ) if recovery_sanctioned and existing_issue == issue_number and existing_branch == branch_name: return None - if is_lease_expired(existing_lock, now=now): + expired = is_lease_expired(existing_lock, now=now) + # #790 review #502/#516: a heartbeat-lifecycle lease that is non-live but + # whose absolute expires_at is still in the future must enter the same + # reclaim/renewal disposition as an expired lease. This happens for + # stale_absolute_cap (a session that keeps heartbeating past the 8h cap has + # expires_at = last_heartbeat + TTL in the future) and for + # stale_missed_heartbeat under an independent TTL>grace policy. Keying this + # gate on is_lease_expired alone left assess_expired_lock_reclaim — which + # already permits exactly those bands — unreachable from the acquisition + # path, so the load-bearing heartbeat was not load-bearing for foreign + # reclaim: the precise abandonment scenario #790 exists to fix. + # assess_foreign_lock_overwrite already keys on is_lease_live; this makes the + # same-issue path consistent with it. Legacy leases keep their absolute-expiry + # clock (AC-N8): is_lease_live already decides a legacy lease from expires_at + # / dead-PID alone and is_legacy_lease excludes it here, so this widening is a + # no-op for every pre-lifecycle lock. + heartbeat_non_live_reclaimable = ( + not expired + and not is_legacy_lease(existing_lock) + and not is_lease_live(existing_lock, now=now) + ) + if expired or heartbeat_non_live_reclaimable: # #760 AC1/AC2: exact-owner renewal is a different disposition from # foreign takeover and is evaluated first. Before this, both branches # below returned unconditionally, so the same_owner allowance further @@ -571,10 +1094,12 @@ def assess_same_issue_lease_conflict( reclaim = assess_expired_lock_reclaim(existing_lock, now=now) if reclaim.get("reclaim_allowed"): # #601: expired + dead pid / missing worktree may be reclaimed - # through the normal lock path (sanctioned overwrite). + # through the normal lock path (sanctioned overwrite). #790: the new + # stale heartbeat bands are reclaimable here on the same evidence. return None + descriptor = "expired" if expired else "non-live" return ( - f"Issue #{issue_number} has an expired {operation_type} lease on " + f"Issue #{issue_number} has an {descriptor} {operation_type} lease on " f"branch '{existing_branch}' from worktree '{existing_worktree}'. " "Recovery review is required before takeover (fail closed)" ) diff --git a/lease_policy.py b/lease_policy.py new file mode 100644 index 0000000..e778cd3 --- /dev/null +++ b/lease_policy.py @@ -0,0 +1,212 @@ +"""Central lease policy configuration (#790 Slice A, AC-N7). + +The single authoritative source for every lease duration in the project. Before +this module the numbers were scattered: a four-hour author TTL was declared +twice (``issue_lock_store`` and ``gitea_mcp_server``), the reviewer/merger +sliding window lived in ``reviewer_pr_lease``, the conflict-fix window in +``pr_work_lease``, and the control-plane default in ``control_plane_db``. +Nothing tied them together, so tuning one class silently diverged from the +others and no reader could answer "how long does a lease live?" without +grepping four files. + +AC-N7 requires that this configuration exist *before* the first heartbeat and +TTL behavior that reads from it, so it ships in Slice A rather than trailing the +code it governs. + +Deliberate boundaries: + +* **Declaration is not rewiring.** Every task class is declared here, but only + those with ``heartbeat_lifecycle_active`` were migrated onto the shared + heartbeat lifecycle in Slice A — currently ``author_issue_work`` alone. + Reviewer, merger, and conflict-fix leases keep their own existing behavior + until Slice C moves them; their numbers are recorded here so the two cannot + drift apart unnoticed, and ``tests/test_issue_790_lease_policy.py`` asserts + the recorded values still equal the constants those modules use. +* **No policy decision lives here.** This module answers "how long", never "may + this session proceed". Freshness, reclaim, and renewal dispositions stay in + ``issue_lock_store``. +""" + +from __future__ import annotations + +import os +from dataclasses import dataclass +from typing import Any + +# Task classes. Only the first is migrated onto the shared lifecycle in Slice A. +TASK_CLASS_AUTHOR_ISSUE_WORK = "author_issue_work" +TASK_CLASS_REVIEWER_PR = "reviewer_pr" +TASK_CLASS_MERGER_PR = "merger_pr" +TASK_CLASS_CONFLICT_FIX = "conflict_fix" + +# Durable marker for a lease minted under the shared heartbeat lifecycle. +# +# #790 AC-N8: this explicit marker — never a timestamp comparison — is what +# distinguishes a heartbeat-lifecycle lease from a legacy one. A lock written +# before this lifecycle existed carries no marker and reads as +# ``LIFECYCLE_LEGACY``. +LIFECYCLE_HEARTBEAT_V1 = "heartbeat-v1" +LIFECYCLE_LEGACY = "legacy" + +_ENV_PREFIX = "GITEA_LEASE_POLICY" + + +@dataclass(frozen=True) +class LeasePolicy: + """Durations governing one task class. + + All intervals are minutes except ``absolute_cap_hours``. ``None`` for the + cap means the class has no maximum continuous duration. + """ + + task_class: str + initial_ttl_minutes: float + heartbeat_cadence_minutes: float + stale_warning_minutes: float + missed_heartbeat_grace_minutes: float + absolute_cap_hours: float | None + recovery_grace_minutes: float + terminal_race_drain_minutes: float + terminal_retirement_eligible: bool + heartbeat_lifecycle_active: bool + + +# Defaults. ``author_issue_work`` adopts the reviewer window proven by #747 +# rather than inventing new numbers: a lease expires 10 minutes after its last +# valid heartbeat, warns at half that, and an actively heartbeating session is +# never evicted. The prior value was a fixed four hours (240 minutes) that no +# heartbeat could shorten — the defect this issue exists to correct. +_DEFAULTS: dict[str, LeasePolicy] = { + TASK_CLASS_AUTHOR_ISSUE_WORK: LeasePolicy( + task_class=TASK_CLASS_AUTHOR_ISSUE_WORK, + initial_ttl_minutes=10.0, + heartbeat_cadence_minutes=2.0, + stale_warning_minutes=5.0, + missed_heartbeat_grace_minutes=10.0, + absolute_cap_hours=8.0, + recovery_grace_minutes=10.0, + terminal_race_drain_minutes=2.0, + terminal_retirement_eligible=True, + heartbeat_lifecycle_active=True, + ), + # Declared, not rewired. These mirror reviewer_pr_lease.LEASE_TTL_MINUTES + # and STALE_WARNING_MINUTES; Slice C migrates the call sites. + TASK_CLASS_REVIEWER_PR: LeasePolicy( + task_class=TASK_CLASS_REVIEWER_PR, + initial_ttl_minutes=10.0, + heartbeat_cadence_minutes=2.0, + stale_warning_minutes=5.0, + missed_heartbeat_grace_minutes=10.0, + absolute_cap_hours=None, + recovery_grace_minutes=10.0, + terminal_race_drain_minutes=2.0, + terminal_retirement_eligible=False, + heartbeat_lifecycle_active=False, + ), + TASK_CLASS_MERGER_PR: LeasePolicy( + task_class=TASK_CLASS_MERGER_PR, + initial_ttl_minutes=10.0, + heartbeat_cadence_minutes=2.0, + stale_warning_minutes=5.0, + missed_heartbeat_grace_minutes=10.0, + absolute_cap_hours=None, + recovery_grace_minutes=10.0, + terminal_race_drain_minutes=2.0, + terminal_retirement_eligible=False, + heartbeat_lifecycle_active=False, + ), + # Mirrors pr_work_lease.DEFAULT_CONFLICT_FIX_TTL_MINUTES. Deliberately left + # at its current window; shortening it is Slice C's call, not this slice's. + TASK_CLASS_CONFLICT_FIX: LeasePolicy( + task_class=TASK_CLASS_CONFLICT_FIX, + initial_ttl_minutes=120.0, + heartbeat_cadence_minutes=2.0, + stale_warning_minutes=5.0, + missed_heartbeat_grace_minutes=10.0, + absolute_cap_hours=None, + recovery_grace_minutes=10.0, + terminal_race_drain_minutes=2.0, + terminal_retirement_eligible=False, + heartbeat_lifecycle_active=False, + ), +} + +_NUMERIC_FIELDS = ( + "initial_ttl_minutes", + "heartbeat_cadence_minutes", + "stale_warning_minutes", + "missed_heartbeat_grace_minutes", + "absolute_cap_hours", + "recovery_grace_minutes", + "terminal_race_drain_minutes", +) + + +def env_var_name(task_class: str, field: str) -> str: + """Environment variable that overrides one field of one task class.""" + return f"{_ENV_PREFIX}_{task_class.upper()}_{field.upper()}" + + +def _override(task_class: str, field: str, default: float | None) -> float | None: + """Read one override, falling back to *default* on anything unusable. + + A malformed or non-positive override is ignored rather than raised: a typo + in an environment variable must not be able to mint a zero-length lease that + makes every claim instantly reclaimable, nor crash the server at import. + """ + raw = (os.environ.get(env_var_name(task_class, field)) or "").strip() + if not raw: + return default + try: + value = float(raw) + except (TypeError, ValueError): + return default + if value <= 0: + return default + return value + + +def policy_for(task_class: str) -> LeasePolicy: + """Return the effective policy for *task_class*. + + Unknown task classes fall back to the author policy, which is the most + conservative migrated class, rather than raising — a new caller must never + be able to crash a lock write by naming a class this table has not learned. + """ + key = str(task_class or "").strip() or TASK_CLASS_AUTHOR_ISSUE_WORK + base = _DEFAULTS.get(key) or _DEFAULTS[TASK_CLASS_AUTHOR_ISSUE_WORK] + resolved = { + field: _override(base.task_class, field, getattr(base, field)) + for field in _NUMERIC_FIELDS + } + if all(resolved[field] == getattr(base, field) for field in _NUMERIC_FIELDS): + return base + return LeasePolicy( + task_class=base.task_class, + terminal_retirement_eligible=base.terminal_retirement_eligible, + heartbeat_lifecycle_active=base.heartbeat_lifecycle_active, + **resolved, + ) + + +def known_task_classes() -> tuple[str, ...]: + """Every declared task class, migrated or not.""" + return tuple(_DEFAULTS) + + +def describe(task_class: str) -> dict[str, Any]: + """Serializable view of a policy, for audit records and tool payloads.""" + policy = policy_for(task_class) + return { + "task_class": policy.task_class, + "initial_ttl_minutes": policy.initial_ttl_minutes, + "heartbeat_cadence_minutes": policy.heartbeat_cadence_minutes, + "stale_warning_minutes": policy.stale_warning_minutes, + "missed_heartbeat_grace_minutes": policy.missed_heartbeat_grace_minutes, + "absolute_cap_hours": policy.absolute_cap_hours, + "recovery_grace_minutes": policy.recovery_grace_minutes, + "terminal_race_drain_minutes": policy.terminal_race_drain_minutes, + "terminal_retirement_eligible": policy.terminal_retirement_eligible, + "heartbeat_lifecycle_active": policy.heartbeat_lifecycle_active, + "lifecycle_version": LIFECYCLE_HEARTBEAT_V1, + } diff --git a/post_restart_reconcile.py b/post_restart_reconcile.py new file mode 100644 index 0000000..6477197 --- /dev/null +++ b/post_restart_reconcile.py @@ -0,0 +1,791 @@ +"""Post-restart MCP reconciliation and completion proof (#662). + +After an MCP process restart, sessions, leases, capabilities, worktrees, and +interrupted mutations are not systematically reconciled; operators rebuild +context from chat. This module is the pure classification core of the +post-restart reconcile path. + +Design rules (mirrors ``restart_coordinator`` / ``workflow_dashboard``): + +* **Pure classification.** :func:`reconcile_after_restart` takes an already + gathered inventory and returns a structured *completion proof*. It never + touches the network, the filesystem, or a live process, so multi-session + fixtures can drive every branch in unit tests. +* **Fail closed.** Incomplete inventory never reports overall ``complete``. + Ambiguous interrupted mutations are ``unresolved`` (never silently resumed). +* **No blind write resume.** The proof never authorizes replaying a mutation; + it only classifies evidence and names follow-up work. +* **#660 soft dependency.** When durable session checkpoints are not present + in the inventory, the checkpoint dimension is ``skipped`` with an explicit + reason rather than inventing a schema (#660 lands separately). +* **Log-only then enforce.** Default mode is ``log_only``. ``enforce`` sets + ``mutation_hold`` when anything remains unresolved so callers can block + write ops until reconcile is complete or degraded mode is documented. + +The single sanctioned gather+classify entry point is the MCP tool +``gitea_reconcile_after_restart`` (read-only inventory gather + pure classify). +Creating durable follow-up Gitea issues from unresolved items is an explicit +apply step outside this pure module. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from datetime import datetime, timezone +from typing import Any, Mapping, Sequence +from uuid import uuid4 + +import lease_lifecycle + +RECONCILE_VERSION = "1.0.0-issue-662" +SCHEMA_VERSION = 1 + +# Overall proof statuses. +STATUS_COMPLETE = "complete" +STATUS_DEGRADED = "degraded" +STATUS_FAILED = "failed" + +# Per-dimension item statuses. +ITEM_RESOLVED = "resolved" +ITEM_UNRESOLVED = "unresolved" +ITEM_DEGRADED = "degraded" +ITEM_SKIPPED = "skipped" + +# Modes. +MODE_LOG_ONLY = "log_only" +MODE_ENFORCE = "enforce" + +# Lease / session phases that imply a write critical section was in flight. +MUTATING_PHASES = frozenset( + { + "implementing", + "publishing", + "merging", + "reviewing", + "committing", + "pushing", + "closing", + "mutating", + "critical_section", + } +) + +# Dimensions the acceptance criteria require. +DIM_SERVICE_HEALTH = "service_health" +DIM_CLIENTS = "clients" +DIM_SESSIONS = "sessions" +DIM_CHECKPOINTS = "checkpoints" +DIM_LEASES = "leases" +DIM_CAPABILITIES = "capabilities" +DIM_WORKTREES = "worktrees" +DIM_MUTATIONS = "interrupted_mutations" +DIM_DUPLICATES = "duplicates" +DIM_QUEUE = "queue" + + +def _utc_now() -> datetime: + return datetime.now(timezone.utc) + + +def _ts(dt: datetime) -> str: + return dt.astimezone(timezone.utc).isoformat().replace("+00:00", "Z") + + +@dataclass(frozen=True) +class ReconcileItem: + """One dimension of the post-restart reconcile report.""" + + dimension: str + status: str + summary: str + details: dict[str, Any] = field(default_factory=dict) + follow_up_required: bool = False + + def as_dict(self) -> dict[str, Any]: + return { + "dimension": self.dimension, + "status": self.status, + "summary": self.summary, + "details": dict(self.details), + "follow_up_required": self.follow_up_required, + } + + +@dataclass(frozen=True) +class FollowUpIssue: + """A durable follow-up issue the apply path may create for unresolved work.""" + + title: str + body: str + dimension: str + severity: str = "high" + + def as_dict(self) -> dict[str, Any]: + return { + "title": self.title, + "body": self.body, + "dimension": self.dimension, + "severity": self.severity, + } + + +@dataclass(frozen=True) +class RestartCompletionProof: + """Machine-readable post-restart completion proof (#662 AC2).""" + + schema_version: int + reconcile_version: str + reconcile_id: str + started_at: str + finished_at: str + boot_head_sha: str | None + current_head_sha: str | None + inventory_complete: bool + incomplete_reasons: tuple[str, ...] + mode: str + mutation_hold: bool + overall_status: str + items: tuple[ReconcileItem, ...] + proposed_follow_ups: tuple[FollowUpIssue, ...] + resolved_count: int + unresolved_count: int + skipped_count: int + note: str + + def as_dict(self) -> dict[str, Any]: + return { + "schema_version": self.schema_version, + "reconcile_version": self.reconcile_version, + "reconcile_id": self.reconcile_id, + "started_at": self.started_at, + "finished_at": self.finished_at, + "boot_head_sha": self.boot_head_sha, + "current_head_sha": self.current_head_sha, + "inventory_complete": self.inventory_complete, + "incomplete_reasons": list(self.incomplete_reasons), + "mode": self.mode, + "mutation_hold": self.mutation_hold, + "overall_status": self.overall_status, + "items": [i.as_dict() for i in self.items], + "proposed_follow_ups": [f.as_dict() for f in self.proposed_follow_ups], + "resolved_count": self.resolved_count, + "unresolved_count": self.unresolved_count, + "skipped_count": self.skipped_count, + "note": self.note, + "links": { + "umbrella": 655, + "vision": 652, + "roadmap": 653, + "issue": 662, + "checkpoint_schema": 660, + "drain_proof": 661, + }, + } + + +def _item( + dimension: str, + status: str, + summary: str, + *, + details: dict[str, Any] | None = None, + follow_up: bool = False, +) -> ReconcileItem: + return ReconcileItem( + dimension=dimension, + status=status, + summary=summary, + details=dict(details or {}), + follow_up_required=follow_up, + ) + + +def _lease_freshness(lease: Mapping[str, Any]) -> str: + fr = lease.get("freshness") + if isinstance(fr, Mapping): + return str(fr.get("freshness") or fr.get("status") or "unknown") + if isinstance(fr, str): + return fr + # Fall back to pure classifier when raw lease rows are supplied. + try: + return str(lease_lifecycle.classify_lease_freshness(dict(lease)).get("freshness") or "unknown") + except Exception: # noqa: BLE001 - pure path must not raise on bad rows + return "unknown" + + +def _is_live_freshness(freshness: str) -> bool: + return freshness in {"active", "live", "fresh"} + + +def _is_mutating_phase(phase: str | None) -> bool: + p = (phase or "").strip().lower() + if not p: + return False + if p in MUTATING_PHASES: + return True + # Soft match for compound phases like "author_implementing". + return any(token in p for token in MUTATING_PHASES) + + +def _detect_interrupted_mutations( + leases: Sequence[Mapping[str, Any]], + pending_mutations: Sequence[Mapping[str, Any]], +) -> list[dict[str, Any]]: + """Return interrupted-mutation evidence (never auto-resumes writes).""" + found: list[dict[str, Any]] = [] + + for raw in pending_mutations or (): + if not isinstance(raw, Mapping): + continue + found.append( + { + "source": "pending_mutation_inventory", + "status": "unresolved", + "phase": raw.get("phase"), + "session_id": raw.get("session_id"), + "work_kind": raw.get("work_kind") or raw.get("kind"), + "work_number": raw.get("work_number") or raw.get("number"), + "reason": raw.get("reason") + or "pending mutation recorded across process restart", + "resume_allowed": False, + } + ) + + for lease in leases or (): + if not isinstance(lease, Mapping): + continue + phase = lease.get("phase") + freshness = _lease_freshness(lease) + if not _is_mutating_phase(str(phase) if phase is not None else None): + continue + # A mutating phase whose owner is not live is interrupted. + if _is_live_freshness(freshness): + # Still live after restart is itself surprising — flag for review. + found.append( + { + "source": "lease_mutating_phase", + "status": "unresolved", + "phase": phase, + "freshness": freshness, + "lease_id": lease.get("lease_id"), + "session_id": lease.get("session_id"), + "work_kind": lease.get("work_kind"), + "work_number": lease.get("work_number"), + "worktree_path": lease.get("worktree_path"), + "reason": ( + "mutating lease phase still classified live after restart; " + "do not auto-resume writes" + ), + "resume_allowed": False, + } + ) + else: + found.append( + { + "source": "lease_mutating_phase", + "status": "unresolved", + "phase": phase, + "freshness": freshness, + "lease_id": lease.get("lease_id"), + "session_id": lease.get("session_id"), + "work_kind": lease.get("work_kind"), + "work_number": lease.get("work_number"), + "worktree_path": lease.get("worktree_path"), + "reason": ( + f"mutating lease phase '{phase}' with non-live freshness " + f"'{freshness}' — interrupted by restart" + ), + "resume_allowed": False, + } + ) + return found + + +def _detect_duplicate_work( + leases: Sequence[Mapping[str, Any]], +) -> list[dict[str, Any]]: + """Surface duplicate live claims on the same work item.""" + by_work: dict[tuple[Any, Any], list[Mapping[str, Any]]] = {} + for lease in leases or (): + if not isinstance(lease, Mapping): + continue + if not _is_live_freshness(_lease_freshness(lease)): + continue + key = (lease.get("work_kind"), lease.get("work_number")) + if key[0] is None or key[1] is None: + continue + by_work.setdefault(key, []).append(lease) + + dups: list[dict[str, Any]] = [] + for (kind, number), rows in sorted(by_work.items(), key=lambda kv: str(kv[0])): + if len(rows) < 2: + continue + dups.append( + { + "work_kind": kind, + "work_number": number, + "claim_count": len(rows), + "session_ids": [r.get("session_id") for r in rows], + "lease_ids": [r.get("lease_id") for r in rows], + } + ) + return dups + + +def _follow_up_for_item(item: ReconcileItem) -> FollowUpIssue | None: + if not item.follow_up_required: + return None + title = f"[post-restart] unresolved {item.dimension} after MCP restart" + body = ( + f"## Post-restart reconcile follow-up (#662)\n\n" + f"**Dimension:** `{item.dimension}`\n" + f"**Status:** `{item.status}`\n" + f"**Summary:** {item.summary}\n\n" + f"```json\n{item.details!r}\n```\n\n" + f"Parent umbrella: #655 · Vision: #652 · Roadmap: #653 · Reconcile: #662\n" + f"Do **not** auto-resume write mutations; reconcile evidence first.\n" + ) + return FollowUpIssue( + title=title, + body=body, + dimension=item.dimension, + severity="high" if item.dimension == DIM_MUTATIONS else "medium", + ) + + +def reconcile_after_restart( + inventory: Mapping[str, Any], + *, + now: datetime | None = None, + mode: str = MODE_LOG_ONLY, + reconcile_id: str | None = None, +) -> RestartCompletionProof: + """Classify a post-restart inventory into a completion proof (#662). + + Parameters + ---------- + inventory: + Gathered facts. Expected keys (all optional except completeness): + + * ``inventory_complete`` (bool) — fail closed when false + * ``incomplete_reasons`` (list[str]) + * ``service_health`` (dict with ``healthy`` bool) + * ``clients`` (list) — connected client descriptors + * ``sessions`` (list) + * ``leases`` (list, optionally with ``freshness``) + * ``checkpoints`` (list | None) — durable session checkpoints (#660) + * ``checkpoints_available`` (bool) — False when #660 schema absent + * ``worktree_bindings`` (list) + * ``pending_mutations`` (list) — explicit interrupted-mutation evidence + * ``capabilities`` (dict with optional ``stale`` / heads) + * ``boot_head_sha`` / ``current_head_sha`` + * ``queue_state`` (dict) + mode: + ``log_only`` (default) or ``enforce`` (sets mutation_hold on unresolved). + """ + started = now or _utc_now() + mode_norm = (mode or MODE_LOG_ONLY).strip().lower() + if mode_norm not in {MODE_LOG_ONLY, MODE_ENFORCE}: + mode_norm = MODE_LOG_ONLY + + inventory_complete = bool(inventory.get("inventory_complete", False)) + incomplete_reasons = tuple( + str(r) for r in (inventory.get("incomplete_reasons") or []) if str(r).strip() + ) + + items: list[ReconcileItem] = [] + + # --- service health ------------------------------------------------- + health = inventory.get("service_health") or {} + if not isinstance(health, Mapping): + health = {} + if not inventory_complete and "service_health" not in inventory: + items.append( + _item( + DIM_SERVICE_HEALTH, + ITEM_UNRESOLVED, + "service health unknown because inventory is incomplete", + details={"inventory_complete": False}, + follow_up=True, + ) + ) + elif health.get("healthy") is True: + items.append( + _item( + DIM_SERVICE_HEALTH, + ITEM_RESOLVED, + "service health verified", + details=dict(health), + ) + ) + elif health.get("healthy") is False: + items.append( + _item( + DIM_SERVICE_HEALTH, + ITEM_UNRESOLVED, + "service health check failed", + details=dict(health), + follow_up=True, + ) + ) + else: + items.append( + _item( + DIM_SERVICE_HEALTH, + ITEM_DEGRADED, + "service health not reported; treating as degraded", + details=dict(health), + follow_up=True, + ) + ) + + # --- clients -------------------------------------------------------- + clients = list(inventory.get("clients") or []) + disconnected = [ + c + for c in clients + if isinstance(c, Mapping) and c.get("connected") is False + ] + if "clients" not in inventory: + items.append( + _item( + DIM_CLIENTS, + ITEM_SKIPPED, + "client inventory not supplied", + details={}, + ) + ) + elif disconnected: + items.append( + _item( + DIM_CLIENTS, + ITEM_UNRESOLVED, + f"{len(disconnected)} disconnected client(s) need reconnect", + details={"disconnected": disconnected, "total": len(clients)}, + follow_up=True, + ) + ) + else: + items.append( + _item( + DIM_CLIENTS, + ITEM_RESOLVED, + f"{len(clients)} client(s) accounted for", + details={"total": len(clients)}, + ) + ) + + # --- sessions ------------------------------------------------------- + sessions = [s for s in (inventory.get("sessions") or []) if isinstance(s, Mapping)] + orphan_sessions = [ + s + for s in sessions + if str(s.get("status") or "").lower() == "active" + and s.get("pid") is not None + and not lease_lifecycle.is_process_alive(s.get("pid")) + ] + if orphan_sessions: + items.append( + _item( + DIM_SESSIONS, + ITEM_UNRESOLVED, + f"{len(orphan_sessions)} active session row(s) with dead owner pid", + details={ + "orphan_session_ids": [s.get("session_id") for s in orphan_sessions], + "total_sessions": len(sessions), + }, + follow_up=True, + ) + ) + else: + items.append( + _item( + DIM_SESSIONS, + ITEM_RESOLVED, + f"{len(sessions)} session row(s) reconciled (no dead-pid orphans)", + details={"total_sessions": len(sessions)}, + ) + ) + + # --- checkpoints (#660 soft) ---------------------------------------- + checkpoints_available = inventory.get("checkpoints_available") + checkpoints = inventory.get("checkpoints") + if checkpoints_available is False or ( + checkpoints is None and "checkpoints" not in inventory + ): + items.append( + _item( + DIM_CHECKPOINTS, + ITEM_SKIPPED, + "durable session checkpoint schema not available yet (#660)", + details={"depends_on": 660}, + ) + ) + else: + cp_list = [c for c in (checkpoints or []) if isinstance(c, Mapping)] + stale_cp = [c for c in cp_list if c.get("stale") or c.get("invalid")] + if stale_cp: + items.append( + _item( + DIM_CHECKPOINTS, + ITEM_UNRESOLVED, + f"{len(stale_cp)} checkpoint(s) invalid or stale vs live state", + details={"stale_count": len(stale_cp), "total": len(cp_list)}, + follow_up=True, + ) + ) + else: + items.append( + _item( + DIM_CHECKPOINTS, + ITEM_RESOLVED, + f"{len(cp_list)} checkpoint(s) consistent with live state", + details={"total": len(cp_list)}, + ) + ) + + # --- leases / locks ------------------------------------------------- + leases = [L for L in (inventory.get("leases") or []) if isinstance(L, Mapping)] + live_leases = [L for L in leases if _is_live_freshness(_lease_freshness(L))] + items.append( + _item( + DIM_LEASES, + ITEM_RESOLVED if inventory_complete else ITEM_DEGRADED, + f"{len(live_leases)} live lease(s) of {len(leases)} inventoried", + details={ + "live_count": len(live_leases), + "total": len(leases), + "live_lease_ids": [L.get("lease_id") for L in live_leases], + }, + follow_up=not inventory_complete, + ) + ) + + # --- capabilities / stale runtime ----------------------------------- + caps = inventory.get("capabilities") or {} + if not isinstance(caps, Mapping): + caps = {} + if caps.get("stale") is True: + items.append( + _item( + DIM_CAPABILITIES, + ITEM_UNRESOLVED, + "runtime code is stale vs on-disk master; restart did not reach parity", + details=dict(caps), + follow_up=True, + ) + ) + else: + items.append( + _item( + DIM_CAPABILITIES, + ITEM_RESOLVED, + "capability/runtime parity acceptable", + details=dict(caps) if caps else {"stale": False}, + ) + ) + + # --- worktrees ------------------------------------------------------ + bindings = [ + b for b in (inventory.get("worktree_bindings") or []) if isinstance(b, Mapping) + ] + missing_wt = [ + b + for b in bindings + if b.get("missing") is True or b.get("exists") is False + ] + if "worktree_bindings" not in inventory: + items.append( + _item( + DIM_WORKTREES, + ITEM_SKIPPED, + "worktree binding inventory not supplied", + ) + ) + elif missing_wt: + items.append( + _item( + DIM_WORKTREES, + ITEM_UNRESOLVED, + f"{len(missing_wt)} worktree binding(s) missing on disk", + details={"missing": missing_wt, "total": len(bindings)}, + follow_up=True, + ) + ) + else: + items.append( + _item( + DIM_WORKTREES, + ITEM_RESOLVED, + f"{len(bindings)} worktree binding(s) present", + details={"total": len(bindings)}, + ) + ) + + # --- interrupted mutations (AC4) ------------------------------------ + pending = [ + m + for m in (inventory.get("pending_mutations") or []) + if isinstance(m, Mapping) + ] + interrupted = _detect_interrupted_mutations(leases, pending) + if interrupted: + items.append( + _item( + DIM_MUTATIONS, + ITEM_UNRESOLVED, + f"{len(interrupted)} interrupted mutation(s); write resume forbidden", + details={"interrupted": interrupted}, + follow_up=True, + ) + ) + else: + items.append( + _item( + DIM_MUTATIONS, + ITEM_RESOLVED, + "no interrupted mutations detected", + details={"interrupted": []}, + ) + ) + + # --- duplicates ----------------------------------------------------- + dups = _detect_duplicate_work(leases) + if dups: + items.append( + _item( + DIM_DUPLICATES, + ITEM_UNRESOLVED, + f"{len(dups)} work item(s) have multiple live claims", + details={"duplicates": dups}, + follow_up=True, + ) + ) + else: + items.append( + _item( + DIM_DUPLICATES, + ITEM_RESOLVED, + "no duplicate live claims detected", + details={"duplicates": []}, + ) + ) + + # --- queue ---------------------------------------------------------- + queue = inventory.get("queue_state") + if queue is None: + items.append( + _item( + DIM_QUEUE, + ITEM_SKIPPED, + "allocator queue state not supplied", + ) + ) + elif isinstance(queue, Mapping) and queue.get("safe_to_resume") is False: + items.append( + _item( + DIM_QUEUE, + ITEM_UNRESOLVED, + "allocator queue not safe to resume", + details=dict(queue), + follow_up=True, + ) + ) + else: + items.append( + _item( + DIM_QUEUE, + ITEM_RESOLVED, + "allocator queue state acceptable", + details=dict(queue) if isinstance(queue, Mapping) else {}, + ) + ) + + # Incomplete inventory always degrades the whole proof. + if not inventory_complete: + # Ensure at least one follow-up names the incomplete inventory. + items.append( + _item( + "inventory", + ITEM_UNRESOLVED, + "control-plane inventory incomplete; reconcile cannot claim success", + details={"reasons": list(incomplete_reasons)}, + follow_up=True, + ) + ) + + resolved = sum(1 for i in items if i.status == ITEM_RESOLVED) + unresolved = sum(1 for i in items if i.status in {ITEM_UNRESOLVED, ITEM_DEGRADED}) + skipped = sum(1 for i in items if i.status == ITEM_SKIPPED) + + if not inventory_complete or any(i.status == ITEM_UNRESOLVED for i in items): + if any(i.status == ITEM_UNRESOLVED for i in items) and inventory_complete: + overall = STATUS_DEGRADED + elif not inventory_complete: + overall = STATUS_FAILED + else: + overall = STATUS_DEGRADED + elif any(i.status == ITEM_DEGRADED for i in items): + overall = STATUS_DEGRADED + else: + overall = STATUS_COMPLETE + + # Enforce mode holds mutations whenever anything is unresolved/failed. + mutation_hold = False + if mode_norm == MODE_ENFORCE and overall in {STATUS_DEGRADED, STATUS_FAILED}: + mutation_hold = True + if mode_norm == MODE_ENFORCE and any( + i.dimension == DIM_MUTATIONS and i.status == ITEM_UNRESOLVED for i in items + ): + mutation_hold = True + + follow_ups = tuple( + fu for i in items if (fu := _follow_up_for_item(i)) is not None + ) + + finished = _utc_now() if now is None else now + note = ( + "Read-only completion proof. Never auto-resumes write mutations. " + "Unresolved items require durable follow-up before claiming clean restart. " + f"Mode={mode_norm}." + ) + + return RestartCompletionProof( + schema_version=SCHEMA_VERSION, + reconcile_version=RECONCILE_VERSION, + reconcile_id=(reconcile_id or f"reconcile-{uuid4().hex[:12]}"), + started_at=_ts(started), + finished_at=_ts(finished), + boot_head_sha=( + str(inventory.get("boot_head_sha")).strip() + if inventory.get("boot_head_sha") + else None + ), + current_head_sha=( + str(inventory.get("current_head_sha")).strip() + if inventory.get("current_head_sha") + else None + ), + inventory_complete=inventory_complete, + incomplete_reasons=incomplete_reasons, + mode=mode_norm, + mutation_hold=mutation_hold, + overall_status=overall, + items=tuple(items), + proposed_follow_ups=follow_ups, + resolved_count=resolved, + unresolved_count=unresolved, + skipped_count=skipped, + note=note, + ) + + +def mutations_allowed(proof: RestartCompletionProof | Mapping[str, Any] | None) -> bool: + """Return whether write mutations may proceed under the given proof.""" + if proof is None: + return True # no proof yet → caller decides; enforce path sets hold + if isinstance(proof, RestartCompletionProof): + return not proof.mutation_hold + if isinstance(proof, Mapping): + return not bool(proof.get("mutation_hold")) + return True diff --git a/role_session_router.py b/role_session_router.py index f756c71..e557266 100644 --- a/role_session_router.py +++ b/role_session_router.py @@ -73,6 +73,8 @@ AUTHOR_TASKS = frozenset({ "claim_issue", "create_branch", "push_branch", + "bootstrap_author_issue_worktree", + "gitea_bootstrap_author_issue_worktree", "create_pr", "comment_pr", "address_pr_change_requests", diff --git a/task_capability_map.py b/task_capability_map.py index 213e701..878cf0a 100644 --- a/task_capability_map.py +++ b/task_capability_map.py @@ -32,6 +32,15 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = { "permission": "gitea.issue.comment", "role": "author", }, + # #790 Slice A: prove an owned author lease is still active. Strictly + # narrower than lock_issue — it can only slide a lease this exact session + # already owns, never acquire, take over, or revive one — so it gates on the + # same authority rather than introducing an operation name that every + # already-configured author profile would be missing. + "heartbeat_issue_lock": { + "permission": "gitea.issue.comment", + "role": "author", + }, # #860: dirty orphaned same-claimant worktree recovery (explicit operation). "recover_dirty_orphaned_issue_worktree": { "permission": "gitea.issue.comment", @@ -78,6 +87,14 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = { "permission": "gitea.branch.create", "role": "author", }, + "bootstrap_author_issue_worktree": { + "permission": "gitea.branch.create", + "role": "author", + }, + "gitea_bootstrap_author_issue_worktree": { + "permission": "gitea.branch.create", + "role": "author", + }, "push_branch": { "permission": "gitea.branch.push", "role": "author", @@ -115,6 +132,16 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = { "permission": "gitea.branch.push", "role": "author", }, + # #662: post-restart reconcile is read-only inventory + pure classification. + # Durable follow-up issue creation is a separate apply path (not this task). + "reconcile_after_restart": { + "permission": "gitea.read", + "role": "author", + }, + "gitea_reconcile_after_restart": { + "permission": "gitea.read", + "role": "author", + }, # PR synchronization lifecycle: assess is read-only (any role with gitea.read); # update-by-merge is author-only and mutates the PR head via Gitea API. "assess_pr_sync_status": { @@ -355,6 +382,21 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = { "role": "controller", }, + # #642: sanctioned host-daemon lifecycle controls. Deliberately *not* a + # ``gitea.*`` operation — restarting an MCP namespace is a host action, not + # a Gitea API call, and no configured Gitea profile should be able to + # satisfy it by accident. Authority comes from the console RBAC model plus + # out-of-band operator authorization (#630); these entries exist so the + # console cannot invent an authority the capability layer never declared. + "restart_namespace": { + "permission": "runtime.restart_namespace", + "role": "controller", + }, + "reload_namespace": { + "permission": "runtime.reload_namespace", + "role": "controller", + }, + # #601 first-class lease lifecycle — inspect/list need read; mutations gate on # ownership in the control-plane DB (not a separate Gitea write permission). "list_workflow_leases": { @@ -487,6 +529,15 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = { "permission": "gitea.issue.comment", "role": "author", }, + + # #651 console analytics ingest — control-plane DB write, not a Gitea API + # call. Authority comes from console RBAC (operator+) plus phase gating; + # permission string is a non-Gitea runtime capability so no Gitea profile + # can satisfy it by accident. + "record_analytics_usage": { + "permission": "runtime.record_analytics_usage", + "role": "author", + }, } @@ -497,6 +548,10 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = { # merger lease (#763). _PREFLIGHT_TASK_TRANSITIONS = frozenset({ ("review_pr", "acquire_reviewer_pr_lease"), + # #850: native author issue worktree bootstrap + ("work_issue", "bootstrap_author_issue_worktree"), + ("bootstrap_author_issue_worktree", "lock_issue"), + # #860: dirty-orphan recovery and related work_issue transitions (master) ("work_issue", "lock_issue"), ("work_issue", "recover_dirty_orphaned_issue_worktree"), ("work_issue", "gitea_recover_dirty_orphaned_issue_worktree"), @@ -548,6 +603,8 @@ ROLE_EXCLUSIVE_TASKS: frozenset[str] = frozenset( "gitea_release_merger_pr_lease", "create_branch", "push_branch", + "bootstrap_author_issue_worktree", + "gitea_bootstrap_author_issue_worktree", "publish_unpublished_branch", "create_pr", "commit_files", @@ -573,6 +630,7 @@ ISSUE_MUTATION_TOOL_TASKS: dict[str, str] = { "gitea_set_issue_labels": "set_issue_labels", "gitea_cleanup_terminal_pr_labels": "cleanup_terminal_pr_labels", "gitea_create_label": "create_label", + "gitea_bootstrap_author_issue_worktree": "bootstrap_author_issue_worktree", "gitea_commit_files": "commit_files", } diff --git a/tests/test_author_issue_bootstrap.py b/tests/test_author_issue_bootstrap.py new file mode 100644 index 0000000..f1de515 --- /dev/null +++ b/tests/test_author_issue_bootstrap.py @@ -0,0 +1,601 @@ +"""Regression test suite for native author issue worktree bootstrap (#850).""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +import tempfile +import unittest +from unittest import mock + +import author_issue_bootstrap +import task_capability_map + + +def _concurrent_bootstrap_worker(args: tuple[str, int, str, str, str, str]) -> dict: + repo_dir, issue_num, key, lock_dir, journal_dir, master_sha = args + os.environ["GITEA_BOOTSTRAP_JOURNAL_DIR"] = journal_dir + return author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=issue_num, + canonical_repo_root=repo_dir, + expected_base_sha=master_sha, + idempotency_key=key, + lock_dir=lock_dir, + owner_session="session-concurrent-test", + active_identity="jcwalker3", + active_profile="prgs-author", + ) + + +class TestAuthorIssueBootstrap(unittest.TestCase): + """Test suite covering AC1-AC10 and comment #14959 specification.""" + + def setUp(self): + self.tmp_dir = tempfile.mkdtemp(prefix="test_bootstrap_") + self.repo_dir = os.path.join(self.tmp_dir, "repo") + os.makedirs(self.repo_dir) + + # Initialize synthetic git repo + subprocess.run(["git", "init", "-b", "master"], cwd=self.repo_dir, check=True, capture_output=True) + subprocess.run(["git", "config", "user.name", "Test User"], cwd=self.repo_dir, check=True) + subprocess.run(["git", "config", "user.email", "test@example.com"], cwd=self.repo_dir, check=True) + + readme = os.path.join(self.repo_dir, "README.md") + with open(readme, "w", encoding="utf-8") as f: + f.write("# Test Repo\n") + subprocess.run(["git", "add", "README.md"], cwd=self.repo_dir, check=True, capture_output=True) + subprocess.run(["git", "commit", "-m", "initial commit"], cwd=self.repo_dir, check=True, capture_output=True) + + rev_res = subprocess.run(["git", "rev-parse", "HEAD"], cwd=self.repo_dir, capture_output=True, text=True, check=True) + self.master_sha = rev_res.stdout.strip() + + self.branches_dir = os.path.join(self.repo_dir, "branches") + os.makedirs(self.branches_dir, exist_ok=True) + self.lock_dir = os.path.join(self.tmp_dir, "locks") + os.makedirs(self.lock_dir, exist_ok=True) + self.journal_dir = os.path.join(self.tmp_dir, "journals") + os.makedirs(self.journal_dir, exist_ok=True) + os.environ["GITEA_BOOTSTRAP_JOURNAL_DIR"] = self.journal_dir + + def tearDown(self): + os.environ.pop("GITEA_BOOTSTRAP_JOURNAL_DIR", None) + shutil.rmtree(self.tmp_dir, ignore_errors=True) + + def test_bootstrap_success_path(self): + """AC1/AC3/AC8: Successful bootstrap creates branch, worktree, registration, and lock proof.""" + key = "test_key_success_1" + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + # assignment_id/lease_id omitted: optional unless verified live. + expected_base_sha=self.master_sha, + idempotency_key=key, + remote="prgs", + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertTrue(res.get("success"), f"Bootstrap failed: {res}") + self.assertFalse(res.get("replayed")) + self.assertEqual(res.get("issue_number"), 850) + self.assertEqual(res.get("base_sha"), self.master_sha) + self.assertIn("branches/fix-issue-850-native-mcp-bootstrap", res.get("worktree_path")) + + # Verify worktree directory exists and is registered + worktree_path = res["worktree_path"] + self.assertTrue(os.path.isdir(worktree_path)) + + wt_list = subprocess.run(["git", "-C", self.repo_dir, "worktree", "list"], capture_output=True, text=True, check=True) + self.assertIn(worktree_path, wt_list.stdout) + + # Verify phase journal written + journal = author_issue_bootstrap.load_phase_journal(key, journal_dir=self.lock_dir) + self.assertIsNotNone(journal) + self.assertTrue(journal.get("completed")) + self.assertEqual(journal.get("current_phase"), author_issue_bootstrap.PHASE_7_TRANSITION_COMPLETED) + + def test_idempotent_replay(self): + """Item 2: Replaying with identical key returns cached transition without duplicate creation.""" + key = "test_key_idempotent_1" + res1 = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + idempotency_key=key, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertTrue(res1["success"], f"res1 failed: {res1}") + self.assertFalse(res1.get("replayed")) + + # Second call + res2 = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + idempotency_key=key, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertTrue(res2["success"], f"res2 failed: {res2}") + self.assertTrue(res2.get("replayed")) + self.assertEqual(res1["worktree_path"], res2["worktree_path"]) + + def test_stale_concurrency_pin_refusal(self): + """Item 3: Mismatched expected base SHA fails closed without silent rebasing.""" + stale_sha = "0000000000000000000000000000000000000000" + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + expected_base_sha=stale_sha, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertFalse(res["success"]) + self.assertEqual(res.get("reason_code"), "stale_concurrency_pin") + self.assertIn("exact_next_action", res) + + def test_path_outside_branches_root_refusal(self): + """Item 6: Worktree path outside branches/ root is refused.""" + outside_path = os.path.join(self.tmp_dir, "outside_worktree") + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + worktree_path=outside_path, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertFalse(res["success"]) + self.assertEqual(res.get("reason_code"), "path_outside_canonical_branches_root") + + def test_preexisting_dirty_worktree_preservation(self): + """Item 6: Preexisting dirty worktree fails closed and is NOT modified or cleaned.""" + branch = "fix/issue-850-dirty-test" + wt_path = os.path.join(self.branches_dir, "fix-issue-850-dirty-test") + subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", "-b", branch, wt_path], check=True, capture_output=True) + + # Create dirty untracked file + dirty_file = os.path.join(wt_path, "dirty.txt") + with open(dirty_file, "w") as f: + f.write("dirty edits\n") + + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + branch_name=branch, + worktree_path=wt_path, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertFalse(res["success"]) + self.assertEqual(res.get("reason_code"), "preexisting_dirty_worktree") + + # Prove dirty file is preserved byte-for-byte + self.assertTrue(os.path.exists(dirty_file)) + with open(dirty_file, "r") as f: + self.assertEqual(f.read(), "dirty edits\n") + + def test_compensating_recovery_on_failed_phase(self): + """AC4/Item 4: Failure during transition rolls back ONLY newly created artifacts.""" + key = "test_key_recovery_1" + # Simulate partial progress in journal + journal = { + "idempotency_key": key, + "issue_number": 850, + "branch_name": "fix/issue-850-recovery-test", + "worktree_path": os.path.join(self.branches_dir, "fix-issue-850-recovery-test"), + "artifacts_created": { + "branch_created": True, + "worktree_dir_created": True, + "worktree_registered": True, + "lock_created": False, + }, + "failure_reason": "simulated lock failure", + "current_phase": author_issue_bootstrap.PHASE_5_REGISTRATION_VERIFIED, + "completed": False, + } + # Create the branch and worktree manually to simulate partial state + subprocess.run(["git", "-C", self.repo_dir, "branch", journal["branch_name"]], check=True, capture_output=True) + subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", journal["worktree_path"], journal["branch_name"]], check=True, capture_output=True) + + # Run compensating recovery + rec = author_issue_bootstrap.run_compensating_recovery(journal, self.repo_dir) + self.assertTrue(rec["executed"]) + self.assertIn(f"worktree_path:{journal['worktree_path']}", rec["rolled_back"]) + self.assertIn(f"branch:{journal['branch_name']}", rec["rolled_back"]) + + # Prove worktree directory and branch were rolled back + self.assertFalse(os.path.exists(journal["worktree_path"])) + branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", journal["branch_name"]], capture_output=True, text=True, check=False) + self.assertNotEqual(branch_check.returncode, 0) + + def test_cross_process_concurrency(self): + """Review #525 Finding 1: Genuine cross-process concurrency locking prevents corruption.""" + import concurrent.futures + + key = "test_concurrent_key_850" + args = (self.repo_dir, 850, key, self.lock_dir, self.journal_dir, self.master_sha) + + with concurrent.futures.ProcessPoolExecutor(max_workers=2) as executor: + fut1 = executor.submit(_concurrent_bootstrap_worker, args) + fut2 = executor.submit(_concurrent_bootstrap_worker, args) + res1 = fut1.result(timeout=10) + res2 = fut2.result(timeout=10) + + self.assertTrue(res1["success"], f"res1 failed: {res1}") + self.assertTrue(res2["success"], f"res2 failed: {res2}") + # One process performs creation, the other process receives idempotent replay + replayed_count = sum(1 for r in (res1, res2) if r.get("replayed")) + created_count = sum(1 for r in (res1, res2) if not r.get("replayed")) + self.assertEqual(replayed_count, 1) + self.assertEqual(created_count, 1) + self.assertEqual(res1["worktree_path"], res2["worktree_path"]) + + def test_interrupted_replay_preserves_artifacts_created_provenance(self): + """Review #525 Finding 2: Replaying incomplete journal preserves creation provenance monotonically.""" + key = "test_key_interrupted_replay_1" + branch = "fix/issue-850-interrupted-replay" + wt_path = os.path.join(self.branches_dir, "fix-issue-850-interrupted-replay") + + # Simulate Phase 2/3 completion where branch and worktree directory were created by this transition + journal = { + "idempotency_key": key, + "issue_number": 850, + "branch_name": branch, + "worktree_path": wt_path, + "active_identity": "jcwalker3", + "active_profile": "prgs-author", + "remote": "prgs", + "org": "Scaled-Tech-Consulting", + "repo": "Gitea-Tools", + "phases": { + author_issue_bootstrap.PHASE_1_REQUEST_ACCEPTED: {"status": "completed"}, + author_issue_bootstrap.PHASE_2_BRANCH_CONFIRMED: {"status": "completed", "created": True}, + }, + "artifacts_created": { + "branch_created": True, + "worktree_dir_created": True, + "worktree_registered": True, + "lock_created": False, + }, + "current_phase": author_issue_bootstrap.PHASE_3_PATH_RESERVED, + "completed": False, + } + # Pre-create the branch and worktree on disk to simulate partial state after crash + subprocess.run(["git", "-C", self.repo_dir, "branch", branch, self.master_sha], check=True, capture_output=True) + subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", wt_path, branch], check=True, capture_output=True) + author_issue_bootstrap.save_phase_journal(journal, journal_dir=self.lock_dir) + + # Now resume/replay the transition but simulate lock binding failure during Phase 6 + with mock.patch("issue_lock_store.bind_session_lock", side_effect=RuntimeError("Lock failure test")): + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + branch_name=branch, + worktree_path=wt_path, + idempotency_key=key, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + + self.assertFalse(res["success"]) + self.assertEqual(res.get("reason_code"), "issue_lock_acquisition_failed") + + # Verify that compensating recovery correctly deleted transition-created branch & worktree + # because creation provenance was preserved across replay (NOT downgraded to False!) + self.assertFalse(os.path.exists(wt_path)) + branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", branch], capture_output=True, text=True, check=False) + self.assertNotEqual(branch_check.returncode, 0) + + def test_transition_created_only_compensation(self): + """Review #525 Finding 4: Preexisting branch is NOT deleted by compensation when only worktree was transition-created.""" + key = "test_key_preexisting_branch_compensation" + preexisting_branch = "fix/issue-850-preexisting" + wt_path = os.path.join(self.branches_dir, "fix-issue-850-preexisting") + + # Create branch BEFORE bootstrap (preexisting branch) + subprocess.run(["git", "-C", self.repo_dir, "branch", preexisting_branch, self.master_sha], check=True, capture_output=True) + + # Call bootstrap with simulated failure during Phase 6 (lock binding) + with mock.patch("issue_lock_store.bind_session_lock", side_effect=RuntimeError("Simulated lock failure")): + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + branch_name=preexisting_branch, + worktree_path=wt_path, + idempotency_key=key, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + + self.assertFalse(res["success"]) + # Worktree dir was created by transition -> removed by compensation + self.assertFalse(os.path.exists(wt_path)) + + # Preexisting branch was NOT created by transition -> MUST BE PRESERVED! + branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", preexisting_branch], capture_output=True, text=True, check=False) + self.assertEqual(branch_check.returncode, 0, "Preexisting branch was deleted by mistake!") + + def test_incompatible_idempotency_replay_refusal(self): + """Review #525 Finding 4: Replaying key with incompatible parameters returns refusal.""" + key = "test_key_incompatible_replay" + res1 = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + branch_name="fix/issue-850-param-a", + idempotency_key=key, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertTrue(res1["success"]) + + # Second call with different branch_name + res2 = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + branch_name="fix/issue-850-param-b", + idempotency_key=key, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertFalse(res2["success"]) + self.assertEqual(res2.get("reason_code"), "incompatible_idempotency_replay") + + def test_exact_next_action_satisfiable_via_mcp(self): + """Review #525 Finding 4: exact_next_action provides satisfiable MCP actions, not shell commands.""" + key = "test_key_next_action_mcp" + stale_sha = "0000000000000000000000000000000000000000" + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + expected_base_sha=stale_sha, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + next_action = res.get("exact_next_action", "") + self.assertNotIn("scripts/worktree-start", next_action) + self.assertNotIn("git worktree add", next_action) + self.assertNotIn("bash", next_action.lower()) + + def test_missing_owner_session_refusal(self): + """Finding D: Missing owner_session context fails closed with typed refusal and zero mutation.""" + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + owner_session=None, + lock_dir=self.lock_dir, + ) + self.assertFalse(res["success"]) + self.assertEqual(res.get("reason_code"), "missing_owner_session") + self.assertIn("exact_next_action", res) + + def test_symlink_lock_file_refusal(self): + """Finding C: BootstrapTransitionLock refuses to follow symlinks.""" + key = "test_symlink_lock_key" + safe_key = "".join(c if c.isalnum() or c in ("-", "_", ".") else "_" for c in key) + lock_path = os.path.join(self.lock_dir, f"{safe_key}.lock") + target_file = os.path.join(self.tmp_dir, "fake_target") + with open(target_file, "w") as f: + f.write("target") + os.symlink(target_file, lock_path) + + with self.assertRaises(RuntimeError) as ctx: + with author_issue_bootstrap.BootstrapTransitionLock(key, journal_dir=self.lock_dir): + pass + self.assertIn("symlink", str(ctx.exception).lower()) + + def test_lock_directory_escape_refusal(self): + """Finding C: BootstrapTransitionLock refuses keys that escape lock directory.""" + with mock.patch("os.path.abspath", return_value="/tmp/outside/evil_key.lock"): + with self.assertRaises(RuntimeError) as ctx: + author_issue_bootstrap.BootstrapTransitionLock("key", journal_dir=self.lock_dir) + self.assertIn("escapes", str(ctx.exception).lower()) + + def test_missing_active_identity_refusal(self): + """F-5: Missing active_identity parameter fails closed.""" + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + owner_session="session-test-1234", + active_identity=None, + active_profile="prgs-author", + lock_dir=self.lock_dir, + ) + self.assertFalse(res["success"]) + self.assertEqual(res.get("reason_code"), "missing_active_identity") + + def test_missing_active_profile_refusal(self): + """F-5: Missing active_profile parameter fails closed.""" + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + owner_session="session-test-1234", + active_identity="jcwalker3", + active_profile=None, + lock_dir=self.lock_dir, + ) + self.assertFalse(res["success"]) + self.assertEqual(res.get("reason_code"), "missing_active_profile") + + def test_dirty_worktree_preserved_during_recovery(self): + """F-4: Compensating recovery does not delete dirty worktree.""" + branch = "fix/issue-850-rec-dirty" + wt_path = os.path.join(self.branches_dir, "fix-issue-850-rec-dirty") + subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", "-b", branch, wt_path], check=True, capture_output=True) + dirty_file = os.path.join(wt_path, "dirty.txt") + with open(dirty_file, "w") as f: + f.write("uncommitted work") + + journal = { + "idempotency_key": "test_dirty_rec", + "issue_number": 850, + "branch_name": branch, + "worktree_path": wt_path, + "artifacts_created": { + "worktree_dir_created": True, + "worktree_registered": True, + }, + "failure_reason": "test dirty recovery", + } + rec = author_issue_bootstrap.run_compensating_recovery(journal, self.repo_dir, journal_dir=self.lock_dir) + self.assertTrue(os.path.exists(wt_path)) + self.assertIn(f"worktree_path_preserved_dirty:{wt_path}", rec["rolled_back"]) + + def test_branch_with_commits_preserved_during_recovery(self): + """F-4: Compensating recovery does not delete branch with author commits.""" + branch = "fix/issue-850-rec-commits" + subprocess.run(["git", "-C", self.repo_dir, "branch", branch, self.master_sha], check=True, capture_output=True) + # Add a commit on the branch + wt_path = os.path.join(self.branches_dir, "fix-issue-850-rec-commits") + subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", wt_path, branch], check=True, capture_output=True) + cfile = os.path.join(wt_path, "commit.txt") + with open(cfile, "w") as f: + f.write("author commit") + subprocess.run(["git", "-C", wt_path, "add", "commit.txt"], check=True, capture_output=True) + subprocess.run(["git", "-C", wt_path, "commit", "-m", "author commit"], check=True, capture_output=True) + subprocess.run(["git", "-C", self.repo_dir, "worktree", "remove", "--force", wt_path], check=True, capture_output=True) + + journal = { + "idempotency_key": "test_commits_rec", + "issue_number": 850, + "branch_name": branch, + "resolved_base_sha": self.master_sha, + "artifacts_created": { + "branch_created": True, + }, + "failure_reason": "test commit branch recovery", + } + rec = author_issue_bootstrap.run_compensating_recovery(journal, self.repo_dir, journal_dir=self.lock_dir) + branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", branch], capture_output=True, text=True, check=False) + self.assertEqual(branch_check.returncode, 0, "Branch with commits was deleted!") + self.assertIn(f"branch_preserved_commits:{branch}", rec["rolled_back"]) + + def test_task_capability_map_integration(self): + """Verify task_capability_map has bootstrap_author_issue_worktree configured correctly.""" + self.assertEqual(task_capability_map.required_role("bootstrap_author_issue_worktree"), "author") + self.assertEqual(task_capability_map.required_permission("bootstrap_author_issue_worktree"), "gitea.branch.create") + self.assertTrue(task_capability_map.preflight_task_matches("work_issue", "bootstrap_author_issue_worktree")) + self.assertTrue(task_capability_map.preflight_task_matches("bootstrap_author_issue_worktree", "lock_issue")) + + def test_unverified_assignment_lease_ids_fail_closed(self): + """Review #531 Finding 4: fabricated assignment/lease IDs are refused.""" + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + assignment_id="asn-fabricated", + lease_id="lease-fabricated", + expected_base_sha=self.master_sha, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertFalse(res["success"]) + self.assertIn( + res.get("reason_code"), + { + "unknown_lease_id", + "assignment_lease_lookup_failed", + "incomplete_assignment_lease_ids", + }, + ) + + def test_partial_assignment_lease_ids_fail_closed(self): + """Review #531 Finding 4: one of assignment_id/lease_id alone is incomplete.""" + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + assignment_id="asn-only", + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertFalse(res["success"]) + self.assertEqual(res.get("reason_code"), "incomplete_assignment_lease_ids") + + def test_stale_diverged_branch_is_not_accepted_via_merge_base(self): + """Review #531 Finding 3: any common ancestor is not enough; require master ⊆ branch.""" + branch = "fix/issue-850-stale-divergent" + # Create branch from current master, then advance master so branch lacks tip. + subprocess.run( + ["git", "-C", self.repo_dir, "branch", branch, self.master_sha], + check=True, + capture_output=True, + ) + # Make a new commit on master (orphan path so branch does not contain it). + marker = os.path.join(self.repo_dir, "master-advance.txt") + with open(marker, "w") as f: + f.write("advance master\n") + subprocess.run(["git", "-C", self.repo_dir, "add", "master-advance.txt"], check=True, capture_output=True) + subprocess.run( + ["git", "-C", self.repo_dir, "commit", "-m", "advance master past branch"], + check=True, + capture_output=True, + ) + new_master = subprocess.run( + ["git", "-C", self.repo_dir, "rev-parse", "HEAD"], + capture_output=True, + text=True, + check=True, + ).stdout.strip() + res = author_issue_bootstrap.bootstrap_author_issue_worktree( + issue_number=850, + canonical_repo_root=self.repo_dir, + branch_name=branch, + expected_base_sha=new_master, + lock_dir=self.lock_dir, + owner_session="session-test-1234", + ) + self.assertFalse(res["success"]) + self.assertEqual(res.get("reason_code"), "incompatible_existing_branch") + + def test_compensating_recovery_attempts_lease_release(self): + """Review #531 Finding 5: recovery invokes lease release when lease_id is present.""" + from unittest import mock + + journal = { + "idempotency_key": "test_lease_rec", + "issue_number": 850, + "owner_session": "session-test-1234", + "lease_id": "lease-abc", + "branch_name": "fix/issue-850-lease-rec", + "artifacts_created": {"lock_created": True}, + "failure_reason": "simulated", + "completed": False, + } + with mock.patch.object( + author_issue_bootstrap.lease_lifecycle, + "release_lease", + return_value={"success": True}, + ) as rel, mock.patch.object( + author_issue_bootstrap.control_plane_db, + "ControlPlaneDB", + return_value=mock.Mock(), + ): + rec = author_issue_bootstrap.run_compensating_recovery( + journal, self.repo_dir, journal_dir=self.lock_dir + ) + self.assertTrue(rec["executed"]) + rel.assert_called_once() + self.assertIn("lease:lease-abc", rec["rolled_back"]) + + +class TestCanonicalRootNoStringSplit(unittest.TestCase): + def test_fallback_uses_commonpath_not_substring_split(self): + """Review #531 Finding 2: no norm.split('/branches/') fallback.""" + import inspect + import author_mutation_worktree as amw + + src = inspect.getsource(amw.resolve_canonical_repo_root) + self.assertNotIn('split("/branches/")', src) + self.assertNotIn("split('/branches/')", src) + + # Fallback recovers repo root from a nested branches worktree path. + with tempfile.TemporaryDirectory() as tmp: + repo = os.path.join(tmp, "repo") + wt = os.path.join(repo, "branches", "fix-issue-850-x") + os.makedirs(wt) + # git unavailable path: pass missing workspace so fallback is used. + root = amw.resolve_canonical_repo_root("/missing/path", wt) + self.assertEqual(root, os.path.realpath(repo)) + + +if __name__ == "__main__": + unittest.main() + diff --git a/tests/test_author_mutation_worktree.py b/tests/test_author_mutation_worktree.py index 23bed97..006a88f 100644 --- a/tests/test_author_mutation_worktree.py +++ b/tests/test_author_mutation_worktree.py @@ -28,6 +28,21 @@ class TestPathUnderBranches(unittest.TestCase): amw.is_path_under_branches("/repo/other-checkout", self.ROOT) ) + def test_unrelated_branches_dir_fails(self): + self.assertFalse( + amw.is_path_under_branches("/tmp/branches/evil", self.ROOT) + ) + + def test_prefix_confusion_fails(self): + self.assertFalse( + amw.is_path_under_branches(f"{self.ROOT}/branches-other/foo", self.ROOT) + ) + + def test_traversal_fails(self): + self.assertFalse( + amw.is_path_under_branches(f"{self.ROOT}/branches/../evil", self.ROOT) + ) + class TestAssessAuthorMutationWorktree(unittest.TestCase): ROOT = "/repo/Gitea-Tools" @@ -71,6 +86,13 @@ class TestAssessAuthorMutationWorktree(unittest.TestCase): self.assertTrue(result["proven"]) self.assertFalse(result["block"]) + def test_path_shaped_branches_ancestry_without_isdir(self): + """Review #551: commonpath recovery must not require on-disk isdir.""" + fake_wt = "/repo/Gitea-Tools/branches/issue-274" + root = amw.resolve_canonical_repo_root(fake_wt, fake_wt) + self.assertEqual(root, "/repo/Gitea-Tools") + self.assertNotIn('split("/branches/")', open(amw.__file__).read()) + class TestPreflightIntegration(unittest.TestCase): def test_verify_preflight_blocks_control_checkout_with_test_porcelain(self): diff --git a/tests/test_issue_628_orchestration.py b/tests/test_issue_628_orchestration.py new file mode 100644 index 0000000..91e9437 --- /dev/null +++ b/tests/test_issue_628_orchestration.py @@ -0,0 +1,199 @@ +"""Regression tests for #628 building blocks (child scope only). + +Honest scope: unit coverage of pre-existing APIs used by umbrella #628. +This module does **not** implement or verify all 21 umbrella acceptance +criteria, automatic handoff store/retrieve, multi-worker product wiring, +or end-to-end orchestration. + +Covered building blocks: +- CTH format / parse / assess (`format_cth_body`, `parse_cth_comment`, + `assess_cth_comment`) +- Exclusive-ownership skip classification (`classify_skip` with + OWNERSHIP_FOREIGN vs OWNERSHIP_OWN) +- Durable dependency edges (`upsert_dependency_edge` / list) and skip + when dependency_unmet +- Edge state transition UNMET -> MET + +Parent umbrella remains #628; this slice is a scoped child issue only. +""" + +import unittest +from unittest.mock import MagicMock, patch +import os +import json +import tempfile + +from canonical_thread_handoff import ( + format_cth_body, + parse_cth_comment, + assess_cth_comment, +) +import dependency_graph +from control_plane_db import ControlPlaneDB +from allocator_service import ( + WorkCandidate, + classify_skip, + ROLE_AUTHOR, + ROLE_REVIEWER, + ROLE_MERGER, + ROLE_RECONCILER, + OWNERSHIP_OWN, + OWNERSHIP_FOREIGN, +) + + +class TestIssue628Orchestration(unittest.TestCase): + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.db_path = os.path.join(self._tmp.name, "cp.sqlite3") + self.db = ControlPlaneDB(self.db_path) + + def tearDown(self): + self._tmp.cleanup() + + def test_canonical_handoff_serialization_and_retrieval(self): + """Building block: format/parse/assess a CTH body (not full AC1/AC2 product path).""" + handoff = format_cth_body( + cth_type="Author Handoff", + status="completed", + next_owner="reviewer", + current_blocker="none", + decision="Implementation complete, tests passing", + proof="pytest tests/test_issue_628_orchestration.py passed", + next_action="Review PR and run reviewer pre-flight", + ready_to_paste_prompt="Review PR for child issue #878 (parent #628)", + ) + self.assertIn("CTH: Author Handoff", handoff) + + parsed = parse_cth_comment(handoff) + self.assertIsNotNone(parsed) + self.assertEqual(parsed["cth_type"], "Author Handoff") + + assessment = assess_cth_comment(handoff) + self.assertFalse(assessment["block"]) + + def test_exclusive_task_unit_single_owner(self): + """Building block: classify_skip foreign vs own ownership.""" + candidate = WorkCandidate( + kind="issue", + number=878, + title="Child #878 ownership classify candidate", + state="open", + labels=["status:in-progress"], + blocked=False, + dependency_unmet=False, + ) + # Foreign ownership MUST be skipped + skip_foreign = classify_skip( + c=candidate, + role=ROLE_AUTHOR, + terminal_pr=None, + claim_ownership=OWNERSHIP_FOREIGN, + ) + self.assertIsNotNone(skip_foreign) + self.assertIn("active lease", skip_foreign) + + # Own/Self claim remains selectable for session resumption + skip_self = classify_skip( + c=candidate, + role=ROLE_AUTHOR, + terminal_pr=None, + claim_ownership=OWNERSHIP_OWN, + ) + self.assertIsNone(skip_self) + + def test_durable_dependency_graph_blocking(self): + """Building block: unmet dependency edge + classify_skip on dependency_unmet.""" + # Upsert a blocking dependency edge between issue 878 and blocker 601 + self.db.upsert_dependency_edge( + remote="prgs", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + source_kind="issue", + source_number=878, + target_kind="issue", + target_number=601, + edge_type=dependency_graph.EDGE_ISSUE_BLOCKED_BY_ISSUE, + state=dependency_graph.STATE_UNMET, + blocking_condition="Target issue #601 is not closed", + completion_condition="Target issue #601 is closed", + evidence={"source": "unit_test"}, + ) + + edges = self.db.list_dependency_edges( + remote="prgs", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + source_kind="issue", + source_number=878, + ) + self.assertEqual(len(edges), 1) + self.assertEqual(edges[0]["state"], "unmet") + self.assertEqual(edges[0]["target_number"], 601) + + # When dependency is unmet, candidate is blocked from selection + candidate = WorkCandidate( + kind="issue", + number=878, + title="Blocked candidate", + state="open", + labels=[], + blocked=False, + dependency_unmet=True, + dependency_reason="issue#878 is blocked by unmet dependency issue#601", + ) + skip_reason = classify_skip( + c=candidate, + role=ROLE_AUTHOR, + terminal_pr=None, + claim_ownership=OWNERSHIP_OWN, + ) + self.assertIsNotNone(skip_reason) + self.assertIn("issue#601", skip_reason) + + def test_dependency_completion_reevaluation(self): + """Building block: dependency edge state can transition UNMET -> MET.""" + self.db.upsert_dependency_edge( + remote="prgs", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + source_kind="issue", + source_number=878, + target_kind="issue", + target_number=601, + edge_type=dependency_graph.EDGE_ISSUE_BLOCKED_BY_ISSUE, + state=dependency_graph.STATE_UNMET, + blocking_condition="Target issue #601 is open", + completion_condition="Target issue #601 is closed", + evidence={"source": "unit_test"}, + ) + + # Mark edge as met upon target issue closure + self.db.upsert_dependency_edge( + remote="prgs", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + source_kind="issue", + source_number=878, + target_kind="issue", + target_number=601, + edge_type=dependency_graph.EDGE_ISSUE_BLOCKED_BY_ISSUE, + state=dependency_graph.STATE_MET, + blocking_condition="Target issue #601 is open", + completion_condition="Target issue #601 is closed", + evidence={"source": "target_closed_event"}, + ) + + edges = self.db.list_dependency_edges( + remote="prgs", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + source_kind="issue", + source_number=878, + ) + self.assertEqual(len(edges), 1) + self.assertEqual(edges[0]["state"], "met") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_issue_662_post_restart_reconcile.py b/tests/test_issue_662_post_restart_reconcile.py new file mode 100644 index 0000000..0782397 --- /dev/null +++ b/tests/test_issue_662_post_restart_reconcile.py @@ -0,0 +1,249 @@ +"""Tests for post-restart MCP reconciliation and completion proof (#662).""" + +from __future__ import annotations + +import json +import os +import unittest +from datetime import datetime, timezone + +import post_restart_reconcile as prr + +NOW = datetime(2026, 7, 24, 12, 0, 0, tzinfo=timezone.utc) + + +def _base_inventory(**overrides): + inv = { + "inventory_complete": True, + "incomplete_reasons": [], + "service_health": {"healthy": True}, + "clients": [{"session_id": "c1", "connected": True}], + "sessions": [ + { + "session_id": "s-live", + "status": "active", + "pid": os.getpid(), + "role": "author", + } + ], + "leases": [], + "checkpoints_available": False, + "worktree_bindings": [{"path": "/tmp/wt", "exists": True}], + "pending_mutations": [], + "capabilities": {"stale": False}, + "boot_head_sha": "a" * 40, + "current_head_sha": "a" * 40, + "queue_state": {"safe_to_resume": True}, + } + inv.update(overrides) + return inv + + +class IncompleteInventoryTest(unittest.TestCase): + def test_incomplete_inventory_fails_closed(self) -> None: + proof = prr.reconcile_after_restart( + { + "inventory_complete": False, + "incomplete_reasons": ["control-plane DB unavailable"], + }, + now=NOW, + mode=prr.MODE_ENFORCE, + reconcile_id="test-incomplete", + ) + self.assertEqual(proof.overall_status, prr.STATUS_FAILED) + self.assertTrue(proof.mutation_hold) + self.assertFalse(proof.inventory_complete) + self.assertTrue(proof.proposed_follow_ups) + self.assertIn("control-plane DB unavailable", proof.incomplete_reasons) + + +class HappyPathTest(unittest.TestCase): + def test_clean_restart_is_complete_without_mutation_hold(self) -> None: + proof = prr.reconcile_after_restart( + _base_inventory(), + now=NOW, + mode=prr.MODE_ENFORCE, + reconcile_id="test-clean", + ) + self.assertEqual(proof.overall_status, prr.STATUS_COMPLETE) + self.assertFalse(proof.mutation_hold) + self.assertEqual(proof.unresolved_count, 0) + cp = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS) + self.assertEqual(cp.status, prr.ITEM_SKIPPED) + links = proof.as_dict()["links"] + self.assertEqual(links["umbrella"], 655) + self.assertEqual(links["issue"], 662) + self.assertEqual(links["vision"], 652) + self.assertEqual(links["roadmap"], 653) + + +class InterruptedMutationTest(unittest.TestCase): + def test_mutating_lease_with_dead_owner_is_unresolved(self) -> None: + proof = prr.reconcile_after_restart( + _base_inventory( + leases=[ + { + "lease_id": "lease-mut", + "session_id": "s-dead", + "phase": "implementing", + "work_kind": "issue", + "work_number": 662, + "worktree_path": "/tmp/wt-662", + "freshness": {"freshness": "stale_dead_process"}, + } + ] + ), + now=NOW, + mode=prr.MODE_ENFORCE, + ) + mut = next(i for i in proof.items if i.dimension == prr.DIM_MUTATIONS) + self.assertEqual(mut.status, prr.ITEM_UNRESOLVED) + self.assertTrue(mut.follow_up_required) + interrupted = mut.details["interrupted"] + self.assertEqual(len(interrupted), 1) + self.assertFalse(interrupted[0]["resume_allowed"]) + self.assertTrue(proof.mutation_hold) + self.assertTrue( + any(f.dimension == prr.DIM_MUTATIONS for f in proof.proposed_follow_ups) + ) + + def test_explicit_pending_mutation_inventory(self) -> None: + proof = prr.reconcile_after_restart( + _base_inventory( + pending_mutations=[ + { + "session_id": "s1", + "phase": "publishing", + "work_kind": "pr", + "work_number": 856, + "reason": "push interrupted mid-flight", + } + ] + ), + now=NOW, + mode=prr.MODE_LOG_ONLY, + ) + mut = next(i for i in proof.items if i.dimension == prr.DIM_MUTATIONS) + self.assertEqual(mut.status, prr.ITEM_UNRESOLVED) + # log_only never holds mutations even when unresolved + self.assertFalse(proof.mutation_hold) + self.assertEqual(proof.overall_status, prr.STATUS_DEGRADED) + + +class DuplicateClaimsTest(unittest.TestCase): + def test_duplicate_live_claims_flagged(self) -> None: + proof = prr.reconcile_after_restart( + _base_inventory( + leases=[ + { + "lease_id": "l1", + "session_id": "s1", + "phase": "allocated", + "work_kind": "issue", + "work_number": 100, + "freshness": {"freshness": "active"}, + }, + { + "lease_id": "l2", + "session_id": "s2", + "phase": "allocated", + "work_kind": "issue", + "work_number": 100, + "freshness": {"freshness": "active"}, + }, + ] + ), + now=NOW, + mode=prr.MODE_ENFORCE, + ) + dups = next(i for i in proof.items if i.dimension == prr.DIM_DUPLICATES) + self.assertEqual(dups.status, prr.ITEM_UNRESOLVED) + self.assertEqual(dups.details["duplicates"][0]["claim_count"], 2) + self.assertTrue(proof.mutation_hold) + + +class OrphanSessionTest(unittest.TestCase): + def test_active_session_dead_pid_is_unresolved(self) -> None: + proof = prr.reconcile_after_restart( + _base_inventory( + sessions=[ + { + "session_id": "ghost", + "status": "active", + "pid": 2_000_000_000, + "role": "author", + } + ] + ), + now=NOW, + mode=prr.MODE_ENFORCE, + ) + sess = next(i for i in proof.items if i.dimension == prr.DIM_SESSIONS) + self.assertEqual(sess.status, prr.ITEM_UNRESOLVED) + self.assertIn("ghost", sess.details["orphan_session_ids"]) + + +class CapabilityStaleTest(unittest.TestCase): + def test_stale_runtime_unresolved(self) -> None: + proof = prr.reconcile_after_restart( + _base_inventory(capabilities={"stale": True, "startup_head": "aaa"}), + now=NOW, + mode=prr.MODE_ENFORCE, + ) + caps = next(i for i in proof.items if i.dimension == prr.DIM_CAPABILITIES) + self.assertEqual(caps.status, prr.ITEM_UNRESOLVED) + self.assertTrue(proof.mutation_hold) + + +class MutationsAllowedHelperTest(unittest.TestCase): + def test_mutations_allowed_respects_hold(self) -> None: + held = prr.reconcile_after_restart( + _base_inventory( + pending_mutations=[{"phase": "merging", "session_id": "x"}] + ), + now=NOW, + mode=prr.MODE_ENFORCE, + ) + self.assertFalse(prr.mutations_allowed(held)) + self.assertFalse(prr.mutations_allowed(held.as_dict())) + clean = prr.reconcile_after_restart( + _base_inventory(), now=NOW, mode=prr.MODE_ENFORCE + ) + self.assertTrue(prr.mutations_allowed(clean)) + + +class CheckpointSoftDependencyTest(unittest.TestCase): + def test_checkpoints_when_schema_present(self) -> None: + proof = prr.reconcile_after_restart( + _base_inventory( + checkpoints_available=True, + checkpoints=[{"session_id": "s1", "stale": False}], + ), + now=NOW, + ) + cp = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS) + self.assertEqual(cp.status, prr.ITEM_RESOLVED) + + def test_stale_checkpoints_unresolved(self) -> None: + proof = prr.reconcile_after_restart( + _base_inventory( + checkpoints_available=True, + checkpoints=[{"session_id": "s1", "stale": True}], + ), + now=NOW, + mode=prr.MODE_ENFORCE, + ) + cp = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS) + self.assertEqual(cp.status, prr.ITEM_UNRESOLVED) + + +class ProofSerializationTest(unittest.TestCase): + def test_as_dict_is_json_friendly(self) -> None: + proof = prr.reconcile_after_restart(_base_inventory(), now=NOW) + blob = json.dumps(proof.as_dict()) + self.assertIn("reconcile_id", blob) + self.assertIn("proposed_follow_ups", blob) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_issue_790_heartbeat_mcp_path.py b/tests/test_issue_790_heartbeat_mcp_path.py new file mode 100644 index 0000000..57e0d5b --- /dev/null +++ b/tests/test_issue_790_heartbeat_mcp_path.py @@ -0,0 +1,444 @@ +"""Task heartbeat through the native MCP author path (#790 Slice A, AC-N6). + +Assessor-level coverage is not sufficient here, and this project has already +paid for learning that: in review #499 on PR #791 the #760 renewal waiver was +computed correctly and then *discarded* at two later gates, so every real +renewal still failed while the unit suite stayed green. AC-N6 exists because of +that, and requires driving the real tools against a real git repository and a +real durable lock file, composing the gates in production order. + +These tests therefore call ``gitea_lock_issue`` and +``gitea_heartbeat_issue_lock`` themselves and assert on what lands on disk, +never on an assessor's return value alone. +""" + +from __future__ import annotations + +import os +import subprocess +import sys +import tempfile +import unittest +from datetime import datetime, timedelta, timezone +from unittest.mock import patch + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + +from mutation_profile_fixture import shared_mutation_env # noqa: E402 + +import issue_lock_provenance # noqa: E402 +import issue_lock_store # noqa: E402 +import lease_policy # noqa: E402 +import mcp_server # noqa: E402 + +ISSUE = 9791 +BRANCH = f"fix/issue-{ISSUE}-heartbeat-mcp" +IDENTITY = "example-user" +PROFILE = "test-author-prgs" +ORG = "Scaled-Tech-Consulting" +REPO = "Gitea-Tools" + + +def _ts(moment: datetime) -> str: + return ( + moment.astimezone(timezone.utc) + .replace(microsecond=0) + .isoformat() + .replace("+00:00", "Z") + ) + + +class _HeartbeatMcpBase(unittest.TestCase): + """Real git repo plus a real durable lock, driven through the real tools.""" + + def setUp(self): + self.lock_dir = tempfile.TemporaryDirectory() + self.addCleanup(self.lock_dir.cleanup) + self.repo = tempfile.mkdtemp(prefix="issue790-mcp-") + self.addCleanup(lambda: subprocess.run(["rm", "-rf", self.repo], check=False)) + self._init_worktree() + self.remotes = patch.dict( + mcp_server.REMOTES, + {"prgs": {"host": "gitea.prgs.cc", "org": ORG, "repo": REPO}}, + ) + self.remotes.start() + self.addCleanup(patch.stopall) + mcp_server._IDENTITY_CACHE.clear() + + def _git(self, *args): + return subprocess.run( + ["git", "-C", self.repo, *args], capture_output=True, text=True, check=True + ) + + def _init_worktree(self): + self._git("init", "-q", "-b", "master") + self._git("config", "user.email", "test@example.com") + self._git("config", "user.name", "Test") + with open(os.path.join(self.repo, "seed.txt"), "w") as fh: + fh.write("seed\n") + self._git("add", "seed.txt") + self._git("commit", "-q", "-m", "seed") + self.base_sha = self._git("rev-parse", "HEAD").stdout.strip() + # A fresh claim starts base-equivalent, which is the ordinary first-lock + # shape and exercises assess_issue_lock_worktree on its normal path. + self._git("checkout", "-q", "-b", BRANCH) + self.head_sha = self.base_sha + self.worktree = os.path.realpath(self.repo) + + def _lock_path(self): + return issue_lock_store.lock_file_path( + remote="prgs", + org=ORG, + repo=REPO, + issue_number=ISSUE, + lock_dir=self.lock_dir.name, + ) + + def _tool_env(self): + env = shared_mutation_env( + PROFILE, include_example_repo=True, GITEA_ISSUE_LOCK_DIR=self.lock_dir.name + ) + env["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name + return env + + def _git_state(self, *, porcelain="", base_equivalent=True): + return { + "current_branch": BRANCH, + "porcelain_status": porcelain, + "base_equivalent": base_equivalent, + "head_sha": self.head_sha, + "inspected_git_root": self.worktree, + "base_branch": "master", + } + + def run_lock_issue( + self, + *, + branch_entries=None, + open_prs=None, + git_state=None, + identity=IDENTITY, + profile=PROFILE, + ): + branch_entries = branch_entries if branch_entries is not None else [] + open_prs = open_prs if open_prs is not None else [] + git_state = git_state or self._git_state() + env = self._tool_env() + with patch( + "mcp_server.api_get_all", return_value=list(branch_entries) + ), patch( + "mcp_server._list_open_pulls", return_value=list(open_prs) + ), patch( + "mcp_server.get_auth_header", return_value="token x" + ), patch( + "mcp_server._work_lease_claimant", + return_value={"username": identity, "profile": profile}, + ), patch( + "mcp_server.issue_lock_worktree.read_worktree_git_state", + return_value=git_state, + ), patch( + "mcp_server.issue_duplicate_context_fetcher", + side_effect=lambda h, o, r, auth, issue_number: ( + list(open_prs), + [b.get("name") for b in branch_entries if isinstance(b, dict)], + {"status": "not_claimed"}, + ), + ), patch.dict(os.environ, env, clear=True): + os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name + return mcp_server.gitea_lock_issue( + issue_number=ISSUE, + branch_name=BRANCH, + remote="prgs", + worktree_path=self.worktree, + ) + + def run_heartbeat( + self, *, task_session_id, identity=IDENTITY, profile=PROFILE, **kwargs + ): + env = self._tool_env() + with patch( + "mcp_server._work_lease_claimant", + return_value={"username": identity, "profile": profile}, + ), patch("mcp_server.get_auth_header", return_value="token x"), patch.dict( + os.environ, env, clear=True + ): + os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name + return mcp_server.gitea_heartbeat_issue_lock( + issue_number=ISSUE, + branch_name=kwargs.pop("branch_name", BRANCH), + task_session_id=task_session_id, + remote="prgs", + worktree_path=kwargs.pop("worktree_path", self.worktree), + **kwargs, + ) + + def write_legacy_lock(self, *, hours_old: float = 3.0, ttl_hours: float = 4.0): + """A durable lock in the shape the store wrote before this slice.""" + now = datetime.now(timezone.utc) + claimant = {"username": IDENTITY, "profile": PROFILE} + created = now - timedelta(hours=hours_old) + record = { + "issue_number": ISSUE, + "branch_name": BRANCH, + "remote": "prgs", + "org": ORG, + "repo": REPO, + "worktree_path": self.worktree, + "session_pid": os.getpid(), + "pid": os.getpid(), + "lock_generation": 1, + "work_lease": { + "operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE, + "issue_number": ISSUE, + "pr_number": None, + "branch": BRANCH, + "worktree_path": self.worktree, + "claimant": claimant, + "created_at": _ts(created), + # The legacy signature: never advanced past creation. + "last_heartbeat_at": _ts(created), + "expires_at": _ts(created + timedelta(hours=ttl_hours)), + }, + "lock_provenance": issue_lock_provenance.build_sanctioned_lock_provenance( + tool="gitea_lock_issue", claimant=claimant + ), + } + path = self._lock_path() + record["lock_file_path"] = path + issue_lock_store.save_lock_file(path, record) + return record + + +class TestLockIssueMintsTheLifecycle(_HeartbeatMcpBase): + """Durable lock creation and read-back through the real tool.""" + + def test_native_lock_writes_the_marker_and_a_task_session_id(self): + result = self.run_lock_issue() + self.assertTrue(result["success"], result) + + written = issue_lock_store.read_lock_file(result["lock_file_path"]) + lease = written["work_lease"] + self.assertEqual( + lease["lifecycle_version"], lease_policy.LIFECYCLE_HEARTBEAT_V1 + ) + self.assertTrue(lease["task_session_id"]) + self.assertFalse(issue_lock_store.is_legacy_lease(written)) + # AC-N1: the ownership key is not the daemon pid, which is recorded + # separately as evidence. + self.assertNotIn(str(written["session_pid"]), lease["task_session_id"]) + self.assertEqual(written["session_pid"], os.getpid()) + + def test_native_lease_uses_the_policy_window_not_four_hours(self): + result = self.run_lock_issue() + lease = result["work_lease"] + created = datetime.fromisoformat(lease["created_at"].replace("Z", "+00:00")) + expires = datetime.fromisoformat(lease["expires_at"].replace("Z", "+00:00")) + policy = lease_policy.policy_for(lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK) + self.assertEqual( + (expires - created).total_seconds() / 60.0, policy.initial_ttl_minutes + ) + + def test_freshness_of_a_new_native_lock_is_live(self): + result = self.run_lock_issue() + self.assertEqual( + result["lock_freshness"]["status"], issue_lock_store.STATUS_LIVE + ) + self.assertTrue(result["lock_freshness"]["live"]) + + +class TestHeartbeatThroughTheTool(_HeartbeatMcpBase): + def _lock_and_session(self): + result = self.run_lock_issue() + self.assertTrue(result["success"], result) + return result, result["work_lease"]["task_session_id"] + + def test_heartbeat_slides_the_lease_and_advances_the_generation(self): + locked, session = self._lock_and_session() + before = issue_lock_store.read_lock_file(locked["lock_file_path"]) + + beat = self.run_heartbeat(task_session_id=session) + + self.assertTrue(beat["success"], beat) + self.assertEqual(beat["operation"], "heartbeat") + after = issue_lock_store.read_lock_file(locked["lock_file_path"]) + self.assertGreater( + issue_lock_store.lock_generation(after), + issue_lock_store.lock_generation(before), + ) + self.assertGreaterEqual( + after["work_lease"]["expires_at"], before["work_lease"]["expires_at"] + ) + self.assertEqual(after["work_lease"]["heartbeat_count"], 2) + + def test_heartbeat_evidence_survives_the_downstream_mutation_gate(self): + """The #499 F2 lesson, applied. + + A sanction that is computed and then discarded downstream is worthless. + After a heartbeat the lock must still satisfy the gate every author + mutation runs through. + """ + locked, session = self._lock_and_session() + self.run_heartbeat(task_session_id=session) + + written = issue_lock_store.read_lock_file(locked["lock_file_path"]) + verdict = issue_lock_store.verify_lock_for_mutation( + written, + issue_number=ISSUE, + branch_name=BRANCH, + worktree_path=self.worktree, + ) + self.assertTrue(verdict["proven"], verdict) + self.assertFalse(verdict["block"]) + + def _duplicate_gate(self, *, open_prs, branches): + env = self._tool_env() + with patch("mcp_server.get_auth_header", return_value="token x"), patch( + "mcp_server.issue_duplicate_context_fetcher", + side_effect=lambda h, o, r, auth, issue_number: ( + list(open_prs), + list(branches), + {"status": "not_claimed"}, + ), + ), patch.dict(os.environ, env, clear=True): + os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name + return mcp_server.gitea_assess_work_issue_duplicate( + issue_number=ISSUE, branch_name=BRANCH, remote="prgs" + ) + + def test_heartbeat_does_not_change_the_duplicate_gate_verdict(self): + """The gate must be invariant under heartbeating. + + The point is not that the gate passes — with a linked open PR at the + lock phase it correctly blocks (#400), heartbeat or not. The property + that matters is that sliding a lease neither loosens the gate nor + corrupts the lock state it reads: the verdict before and after a + heartbeat must be identical, for both the clear and the blocking shape. + """ + _, session = self._lock_and_session() + linked = [{"number": 4242, "head": {"ref": BRANCH, "sha": self.head_sha}}] + + clear_before = self._duplicate_gate(open_prs=[], branches=[]) + blocked_before = self._duplicate_gate(open_prs=linked, branches=[BRANCH]) + + self.assertTrue(self.run_heartbeat(task_session_id=session)["success"]) + + clear_after = self._duplicate_gate(open_prs=[], branches=[]) + blocked_after = self._duplicate_gate(open_prs=linked, branches=[BRANCH]) + + self.assertEqual(clear_before["outcome"], clear_after["outcome"]) + self.assertFalse(clear_after["block"]) + self.assertEqual(blocked_before["outcome"], blocked_after["outcome"]) + self.assertTrue(blocked_after["block"]) + self.assertEqual(blocked_after["linked_open_pr"], 4242) + + def test_foreign_session_id_is_refused_through_the_tool(self): + self._lock_and_session() + beat = self.run_heartbeat(task_session_id="author_issue_work-ffffffffffffffff") + self.assertFalse(beat["success"]) + self.assertIn("task_session_id does not match", " ".join(beat["reasons"])) + + def test_stale_generation_is_refused_through_the_tool(self): + locked, session = self._lock_and_session() + current = issue_lock_store.lock_generation( + issue_lock_store.read_lock_file(locked["lock_file_path"]) + ) + beat = self.run_heartbeat( + task_session_id=session, expected_generation=current + 5 + ) + self.assertFalse(beat["success"]) + self.assertIn("generation changed", beat["reasons"][0]) + + def test_foreign_claimant_is_refused_through_the_tool(self): + _, session = self._lock_and_session() + beat = self.run_heartbeat(task_session_id=session, identity="someone-else") + self.assertFalse(beat["success"]) + + def test_heartbeat_cannot_acquire_a_missing_lock(self): + beat = self.run_heartbeat(task_session_id="author_issue_work-000000000000") + self.assertFalse(beat["success"]) + self.assertIn("no durable lock", beat["reasons"][0]) + + def test_alive_pid_alone_does_not_keep_a_lease_live_through_the_tool(self): + """PID-only refusal, end to end. + + The recorded pid is this live process. The lock is aged past its grace + with no heartbeat, so the tool must refuse to slide it and the durable + record must classify as a missed heartbeat rather than as live. + """ + locked, session = self._lock_and_session() + record = issue_lock_store.read_lock_file(locked["lock_file_path"]) + record["work_lease"]["last_heartbeat_at"] = _ts( + datetime.now(timezone.utc) - timedelta(minutes=30) + ) + record["work_lease"]["expires_at"] = _ts( + datetime.now(timezone.utc) + timedelta(hours=2) + ) + issue_lock_store.save_lock_file(locked["lock_file_path"], record) + + self.assertTrue(issue_lock_store.is_process_alive(record["session_pid"])) + fresh = issue_lock_store.assess_lock_freshness(record) + self.assertEqual( + fresh["status"], issue_lock_store.STATUS_STALE_MISSED_HEARTBEAT + ) + self.assertTrue(fresh["pid_alive"]) + + beat = self.run_heartbeat(task_session_id=session) + self.assertFalse(beat["success"]) + self.assertIn("reclaimed", " ".join(beat["reasons"])) + + +class TestLegacyLocksThroughTheTool(_HeartbeatMcpBase): + """AC-N8 end to end: protected on deployment, and rebindable.""" + + def test_legacy_lock_stays_protected_after_deployment(self): + record = self.write_legacy_lock(hours_old=3.0, ttl_hours=4.0) + fresh = issue_lock_store.assess_lock_freshness(record) + self.assertEqual(fresh["status"], issue_lock_store.STATUS_LIVE) + self.assertTrue(fresh["legacy_lease"]) + self.assertTrue(fresh["legacy_expiry_preserved"]) + # It had never heartbeated, so under the new grace alone it would be + # long gone; the preserved absolute expiry is what protects it. + self.assertEqual( + record["work_lease"]["created_at"], + record["work_lease"]["last_heartbeat_at"], + ) + + def test_tool_rebinds_a_legacy_lock_and_mints_a_first_heartbeat(self): + self.write_legacy_lock(hours_old=3.0, ttl_hours=4.0) + + result = self.run_heartbeat(task_session_id=None) + + self.assertTrue(result["success"], result) + self.assertEqual(result["operation"], "legacy_rebind") + self.assertTrue(result["task_session_id"]) + + written = issue_lock_store.read_lock_file(self._lock_path()) + lease = written["work_lease"] + self.assertEqual( + lease["lifecycle_version"], lease_policy.LIFECYCLE_HEARTBEAT_V1 + ) + self.assertEqual(lease["heartbeat_count"], 1) + self.assertNotEqual( + lease["created_at"], + written["legacy_rebind"]["legacy_origin"]["created_at"], + ) + self.assertFalse(issue_lock_store.is_legacy_lease(written)) + + def test_rebound_lock_then_heartbeats_through_the_tool(self): + self.write_legacy_lock(hours_old=3.0, ttl_hours=4.0) + rebound = self.run_heartbeat(task_session_id=None) + beat = self.run_heartbeat(task_session_id=rebound["task_session_id"]) + self.assertTrue(beat["success"], beat) + self.assertEqual(beat["operation"], "heartbeat") + self.assertEqual(beat["heartbeat_count"], 2) + + def test_rebind_refuses_a_foreign_owner_through_the_tool(self): + self.write_legacy_lock(hours_old=3.0, ttl_hours=4.0) + result = self.run_heartbeat(task_session_id=None, identity="someone-else") + self.assertFalse(result["success"]) + self.assertEqual(result["operation"], "legacy_rebind") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_issue_790_lease_policy.py b/tests/test_issue_790_lease_policy.py new file mode 100644 index 0000000..34d52d6 --- /dev/null +++ b/tests/test_issue_790_lease_policy.py @@ -0,0 +1,594 @@ +"""Central lease policy and load-bearing heartbeat freshness (#790 Slice A). + +Before this slice, ``issue_lock_store.assess_lock_freshness`` parsed +``last_heartbeat_at`` and then never consulted it: liveness was decided by an +absolute four-hour ``expires_at`` and by PID liveness. Because the recorded PID +is the long-lived MCP daemon rather than the authoring task, an abandoned claim +stayed "live" for the full four hours, and a claim whose work had already landed +blocked reconciliation for just as long (Issue #787 / PR #789, and again Issue +#760 / PR #791). + +These tests pin the corrected semantics, including the two asymmetries that are +easy to lose in a refactor: + +* an **alive** PID must never make anything live (AC-N2), while +* a **dead** PID must still mark a lease stale, because #753 dead-session + recovery keys on exactly that classification. + +Durable-state helpers here write real lock files through the real flock path; +they are not mocks of the store. +""" + +from __future__ import annotations + +import os +import sys +import tempfile +import unittest +from datetime import datetime, timedelta, timezone +from unittest.mock import patch + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import issue_lock_store # noqa: E402 +import lease_policy # noqa: E402 +import pr_work_lease # noqa: E402 +import reviewer_pr_lease # noqa: E402 + +ISSUE = 9790 +BRANCH = f"fix/issue-{ISSUE}-heartbeat" +IDENTITY = "example-user" +PROFILE = "test-author-prgs" +ORG = "Example-Org" +REPO = "Example-Repo" +REMOTE = "prgs" +DEAD_PID = 2**22 # far above any live pid on a test host + + +def _ts(moment: datetime) -> str: + return ( + moment.astimezone(timezone.utc) + .replace(microsecond=0) + .isoformat() + .replace("+00:00", "Z") + ) + + +class _LockFixture(unittest.TestCase): + def setUp(self): + self.lock_dir = tempfile.TemporaryDirectory() + self.addCleanup(self.lock_dir.cleanup) + self.now = datetime.now(timezone.utc) + self.worktree = os.path.realpath(tempfile.mkdtemp(prefix="issue790-")) + self.addCleanup(patch.stopall) + + def _path(self): + return issue_lock_store.lock_file_path( + remote=REMOTE, + org=ORG, + repo=REPO, + issue_number=ISSUE, + lock_dir=self.lock_dir.name, + ) + + def write_lock( + self, + *, + lifecycle: str | None = lease_policy.LIFECYCLE_HEARTBEAT_V1, + created_delta: timedelta = timedelta(minutes=1), + heartbeat_delta: timedelta = timedelta(minutes=1), + expires_delta: timedelta = timedelta(minutes=9), + pid: int | None = None, + task_session_id: str | None = "author_issue_work-aaaabbbbccccdddd", + generation: int = 1, + identity: str = IDENTITY, + profile: str = PROFILE, + branch: str = BRANCH, + worktree: str | None = None, + ) -> dict: + """Write a real durable lock and return the record. + + Deltas are relative to ``self.now``; ``expires_delta`` is added, the + others subtracted, so "in the past" reads naturally at each call site. + """ + lease: dict = { + "operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE, + "issue_number": ISSUE, + "pr_number": None, + "branch": branch, + "worktree_path": worktree or self.worktree, + "claimant": {"username": identity, "profile": profile}, + "created_at": _ts(self.now - created_delta), + "last_heartbeat_at": _ts(self.now - heartbeat_delta), + "expires_at": _ts(self.now + expires_delta), + } + if lifecycle is not None: + lease["lifecycle_version"] = lifecycle + if task_session_id is not None: + lease["task_session_id"] = task_session_id + pid_value = os.getpid() if pid is None else pid + record = { + "issue_number": ISSUE, + "branch_name": branch, + "remote": REMOTE, + "org": ORG, + "repo": REPO, + "worktree_path": worktree or self.worktree, + "session_pid": pid_value, + "pid": pid_value, + "lock_generation": generation, + "work_lease": lease, + } + path = self._path() + record["lock_file_path"] = path + issue_lock_store.save_lock_file(path, record) + return record + + +class TestPolicyIsTheSingleSource(unittest.TestCase): + """AC-N7: one authoritative configuration source for every duration.""" + + def test_author_policy_carries_the_agreed_values(self): + policy = lease_policy.policy_for(lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK) + self.assertEqual(policy.initial_ttl_minutes, 10.0) + self.assertEqual(policy.heartbeat_cadence_minutes, 2.0) + self.assertEqual(policy.stale_warning_minutes, 5.0) + self.assertEqual(policy.missed_heartbeat_grace_minutes, 10.0) + self.assertEqual(policy.absolute_cap_hours, 8.0) + self.assertEqual(policy.recovery_grace_minutes, 10.0) + self.assertEqual(policy.terminal_race_drain_minutes, 2.0) + self.assertTrue(policy.terminal_retirement_eligible) + self.assertTrue(policy.heartbeat_lifecycle_active) + + def test_the_four_hour_author_ttl_literal_is_gone(self): + """The duplicated literal AC-N7 exists to remove.""" + self.assertFalse(hasattr(issue_lock_store, "WORK_LEASE_TTL_HOURS")) + import gitea_mcp_server + + self.assertFalse(hasattr(gitea_mcp_server, "WORK_LEASE_TTL_HOURS")) + + def test_declared_reviewer_values_match_the_module_still_using_them(self): + """Slice A declares reviewer/merger numbers without rewiring them. + + Recording a value in two places is only safe if drift is detectable, so + this asserts the declaration still equals the constants #747 owns. When + Slice C migrates those call sites, this test becomes the proof the + migration changed nothing. + """ + policy = lease_policy.policy_for(lease_policy.TASK_CLASS_REVIEWER_PR) + self.assertEqual( + policy.initial_ttl_minutes, float(reviewer_pr_lease.LEASE_TTL_MINUTES) + ) + self.assertEqual( + policy.stale_warning_minutes, + float(reviewer_pr_lease.STALE_WARNING_MINUTES), + ) + self.assertFalse(policy.heartbeat_lifecycle_active) + + def test_declared_conflict_fix_value_matches_its_module(self): + policy = lease_policy.policy_for(lease_policy.TASK_CLASS_CONFLICT_FIX) + self.assertEqual( + policy.initial_ttl_minutes, + float(pr_work_lease.DEFAULT_CONFLICT_FIX_TTL_MINUTES), + ) + self.assertFalse(policy.heartbeat_lifecycle_active) + + def test_environment_override_applies(self): + var = lease_policy.env_var_name( + lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK, "initial_ttl_minutes" + ) + with patch.dict(os.environ, {var: "7"}): + self.assertEqual( + lease_policy.policy_for( + lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK + ).initial_ttl_minutes, + 7.0, + ) + + def test_unusable_override_falls_back_instead_of_minting_a_zero_lease(self): + """A typo must not make every claim instantly reclaimable.""" + var = lease_policy.env_var_name( + lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK, "initial_ttl_minutes" + ) + for bad in ("0", "-5", "not-a-number", " "): + with self.subTest(value=bad), patch.dict(os.environ, {var: bad}): + self.assertEqual( + lease_policy.policy_for( + lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK + ).initial_ttl_minutes, + 10.0, + ) + + def test_unknown_task_class_does_not_raise(self): + policy = lease_policy.policy_for("something-new") + self.assertEqual(policy.task_class, lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK) + + +class TestLifecycleDiscrimination(_LockFixture): + """AC-N8: the marker, never a timestamp, decides legacy vs heartbeat.""" + + def test_missing_marker_reads_as_legacy(self): + record = self.write_lock(lifecycle=None) + self.assertTrue(issue_lock_store.is_legacy_lease(record)) + self.assertEqual( + issue_lock_store.lease_lifecycle_version(record), + lease_policy.LIFECYCLE_LEGACY, + ) + + def test_marker_present_reads_as_heartbeat_lifecycle(self): + record = self.write_lock() + self.assertFalse(issue_lock_store.is_legacy_lease(record)) + + def test_equal_created_and_heartbeat_never_implies_a_fresh_heartbeat(self): + """The exact inversion AC-N8 forbids. + + A legacy lock has ``last_heartbeat_at == created_at`` forever because + nothing ever advanced it. Reading that equality as "recently + heartbeated" would classify every never-heartbeated lock as fresh. + """ + legacy = self.write_lock( + lifecycle=None, + created_delta=timedelta(hours=3), + heartbeat_delta=timedelta(hours=3), + ) + lease = legacy["work_lease"] + self.assertEqual(lease["created_at"], lease["last_heartbeat_at"]) + self.assertTrue(issue_lock_store.is_legacy_lease(legacy)) + + # A brand-new heartbeat lease has them equal too, so the equality + # carries no information in either direction. + fresh = self.write_lock( + created_delta=timedelta(seconds=0), heartbeat_delta=timedelta(seconds=0) + ) + self.assertEqual( + fresh["work_lease"]["created_at"], + fresh["work_lease"]["last_heartbeat_at"], + ) + self.assertFalse(issue_lock_store.is_legacy_lease(fresh)) + + def test_minted_session_id_contains_no_pid(self): + """AC-N1: the ownership key must not be derived from the daemon pid.""" + minted = issue_lock_store.mint_task_session_id() + self.assertNotIn(str(os.getpid()), minted) + self.assertNotEqual(minted, issue_lock_store.mint_task_session_id()) + + +class TestFreshnessIsHeartbeatDriven(_LockFixture): + """AC-N2 and the new bands.""" + + def test_fresh_heartbeat_is_live(self): + record = self.write_lock(heartbeat_delta=timedelta(minutes=1)) + fresh = issue_lock_store.assess_lock_freshness(record, now=self.now) + self.assertEqual(fresh["status"], issue_lock_store.STATUS_LIVE) + self.assertTrue(fresh["live"]) + self.assertFalse(fresh["heartbeat_warning"]) + + def test_heartbeat_past_warning_is_still_live_but_flagged(self): + record = self.write_lock(heartbeat_delta=timedelta(minutes=6)) + fresh = issue_lock_store.assess_lock_freshness(record, now=self.now) + self.assertEqual(fresh["status"], issue_lock_store.STATUS_LIVE) + self.assertTrue(fresh["heartbeat_warning"]) + + def test_missed_heartbeat_past_grace_is_classified_explicitly(self): + record = self.write_lock( + heartbeat_delta=timedelta(minutes=11), + expires_delta=timedelta(minutes=30), + ) + fresh = issue_lock_store.assess_lock_freshness(record, now=self.now) + self.assertEqual( + fresh["status"], issue_lock_store.STATUS_STALE_MISSED_HEARTBEAT + ) + self.assertFalse(fresh["live"]) + self.assertTrue(fresh["stale"]) + + def test_alive_pid_never_establishes_freshness(self): + """The defect in one assertion. + + The recorded PID is this very process, so it is unambiguously alive — + and the lease is still not live, because the task stopped heartbeating. + """ + record = self.write_lock( + pid=os.getpid(), + heartbeat_delta=timedelta(hours=4), + expires_delta=timedelta(hours=4), + ) + fresh = issue_lock_store.assess_lock_freshness(record, now=self.now) + self.assertTrue(fresh["pid_alive"]) + self.assertFalse(fresh["live"]) + self.assertEqual( + fresh["status"], issue_lock_store.STATUS_STALE_MISSED_HEARTBEAT + ) + + def test_dead_pid_still_marks_stale_for_issue_753(self): + """The opposite asymmetry: dead-PID corroboration is preserved.""" + record = self.write_lock(pid=DEAD_PID, heartbeat_delta=timedelta(minutes=1)) + fresh = issue_lock_store.assess_lock_freshness(record, now=self.now) + self.assertEqual(fresh["status"], issue_lock_store.STATUS_STALE) + self.assertFalse(fresh["live"]) + self.assertIn("not alive", fresh["reason"]) + + def test_absolute_cap_requires_readoption(self): + record = self.write_lock( + created_delta=timedelta(hours=9), heartbeat_delta=timedelta(minutes=1) + ) + fresh = issue_lock_store.assess_lock_freshness(record, now=self.now) + self.assertEqual(fresh["status"], issue_lock_store.STATUS_STALE_ABSOLUTE_CAP) + self.assertIn("re-adoption", fresh["reason"]) + + def test_heartbeat_lifecycle_without_a_heartbeat_fails_closed(self): + record = self.write_lock() + del record["work_lease"]["last_heartbeat_at"] + issue_lock_store.save_lock_file(self._path(), record) + fresh = issue_lock_store.assess_lock_freshness(record, now=self.now) + self.assertEqual( + fresh["status"], issue_lock_store.STATUS_STALE_MISSED_HEARTBEAT + ) + self.assertIn("fail closed", fresh["reason"]) + + def test_absent_lock(self): + fresh = issue_lock_store.assess_lock_freshness(None) + self.assertEqual(fresh["status"], issue_lock_store.STATUS_ABSENT) + self.assertFalse(fresh["stale"]) + + +class TestLegacyLocksStayProtected(_LockFixture): + """AC-N8: deployment must not retroactively shorten an existing claim.""" + + def test_legacy_lock_with_a_stale_heartbeat_remains_live(self): + """The deployment-safety case. + + A four-hour legacy lease minted three hours ago has not heartbeated + once. Under the new grace it would be long gone; under its preserved + absolute expiry it is still live, and must stay that way. + """ + record = self.write_lock( + lifecycle=None, + created_delta=timedelta(hours=3), + heartbeat_delta=timedelta(hours=3), + expires_delta=timedelta(hours=1), + ) + fresh = issue_lock_store.assess_lock_freshness(record, now=self.now) + self.assertEqual(fresh["status"], issue_lock_store.STATUS_LIVE) + self.assertTrue(fresh["live"]) + self.assertTrue(fresh["legacy_lease"]) + self.assertTrue(fresh["legacy_expiry_preserved"]) + + def test_legacy_lock_past_its_absolute_expiry_is_expired_as_before(self): + record = self.write_lock( + lifecycle=None, + created_delta=timedelta(hours=5), + heartbeat_delta=timedelta(hours=5), + expires_delta=timedelta(hours=-1), + ) + fresh = issue_lock_store.assess_lock_freshness(record, now=self.now) + self.assertEqual(fresh["status"], issue_lock_store.STATUS_EXPIRED) + + def test_legacy_lock_is_never_reclaimed_by_the_heartbeat_band(self): + record = self.write_lock( + lifecycle=None, + created_delta=timedelta(hours=3), + heartbeat_delta=timedelta(hours=3), + expires_delta=timedelta(hours=1), + ) + reclaim = issue_lock_store.assess_expired_lock_reclaim(record, now=self.now) + self.assertFalse(reclaim["reclaim_allowed"]) + + +class TestReclaimAfterMissedHeartbeat(_LockFixture): + def test_missed_heartbeat_makes_ownership_reclaimable(self): + record = self.write_lock( + pid=os.getpid(), + heartbeat_delta=timedelta(minutes=15), + expires_delta=timedelta(hours=3), + ) + reclaim = issue_lock_store.assess_expired_lock_reclaim(record, now=self.now) + self.assertTrue(reclaim["reclaim_allowed"]) + self.assertIn("stale_missed_heartbeat", reclaim["reasons"][0]) + + def test_live_lease_is_never_reclaimable(self): + record = self.write_lock(heartbeat_delta=timedelta(minutes=1)) + reclaim = issue_lock_store.assess_expired_lock_reclaim(record, now=self.now) + self.assertFalse(reclaim["reclaim_allowed"]) + + def test_dead_pid_reclaim_path_is_unchanged(self): + """#753 must keep working through its original conditions.""" + record = self.write_lock(pid=DEAD_PID, heartbeat_delta=timedelta(minutes=1)) + reclaim = issue_lock_store.assess_expired_lock_reclaim(record, now=self.now) + self.assertTrue(reclaim["reclaim_allowed"]) + self.assertTrue(reclaim["owner_pid_dead"]) + + +class TestHeartbeatWriter(_LockFixture): + """A4: flock + CAS + exact verification, and no revival path.""" + + def _heartbeat(self, **kwargs): + params = { + "remote": REMOTE, + "org": ORG, + "repo": REPO, + "issue_number": ISSUE, + "branch_name": BRANCH, + "worktree_path": self.worktree, + "identity": IDENTITY, + "profile": PROFILE, + "task_session_id": "author_issue_work-aaaabbbbccccdddd", + "lock_dir": self.lock_dir.name, + "now": self.now, + } + params.update(kwargs) + return issue_lock_store.heartbeat_session_lock(**params) + + def test_heartbeat_slides_expiry_and_advances_generation(self): + self.write_lock(heartbeat_delta=timedelta(minutes=4), generation=5) + result = self._heartbeat() + self.assertTrue(result["success"], result) + self.assertEqual(result["prior_generation"], 5) + self.assertEqual(result["lock_generation"], 6) + self.assertEqual(result["heartbeat_count"], 1) + self.assertEqual(result["last_heartbeat_at"], _ts(self.now)) + self.assertEqual(result["expires_at"], _ts(self.now + timedelta(minutes=10))) + self.assertTrue(result["freshness"]["live"]) + + def test_heartbeat_is_durable_and_repeatable(self): + self.write_lock(heartbeat_delta=timedelta(minutes=4)) + self._heartbeat() + second = self._heartbeat(now=self.now + timedelta(minutes=1)) + self.assertTrue(second["success"], second) + self.assertEqual(second["heartbeat_count"], 2) + written = issue_lock_store.read_lock_file(self._path()) + self.assertEqual(written["work_lease"]["heartbeat_count"], 2) + + def test_stale_generation_is_refused(self): + self.write_lock(generation=5) + result = self._heartbeat(expected_generation=4) + self.assertFalse(result["success"]) + self.assertIn("generation changed", result["reasons"][0]) + + def test_foreign_session_is_refused(self): + self.write_lock() + result = self._heartbeat(task_session_id="author_issue_work-ffffffffffffffff") + self.assertFalse(result["success"]) + self.assertIn("task_session_id does not match", " ".join(result["reasons"])) + + def test_missing_session_id_is_refused(self): + self.write_lock() + result = self._heartbeat(task_session_id="") + self.assertFalse(result["success"]) + + def test_foreign_claimant_is_refused(self): + self.write_lock() + for field, value in ( + ("identity", "someone-else"), + ("profile", "other-profile"), + ): + with self.subTest(field=field): + result = self._heartbeat(**{field: value}) + self.assertFalse(result["success"]) + + def test_branch_and_worktree_mismatch_are_refused(self): + self.write_lock() + wrong_branch = self._heartbeat(branch_name=f"fix/issue-{ISSUE}-other") + self.assertFalse(wrong_branch["success"]) + wrong_worktree = self._heartbeat(worktree_path="/tmp/not-the-worktree") + self.assertFalse(wrong_worktree["success"]) + + def test_lapsed_lease_cannot_be_heartbeated_back_to_life(self): + """No revival path (A4). + + A session that stopped proving liveness must reclaim under a fresh + generation, not restore ownership retroactively. + """ + self.write_lock( + heartbeat_delta=timedelta(minutes=30), expires_delta=timedelta(hours=1) + ) + result = self._heartbeat() + self.assertFalse(result["success"]) + self.assertIn("reclaimed", " ".join(result["reasons"])) + + def test_absent_lock_cannot_be_created_by_heartbeat(self): + result = self._heartbeat() + self.assertFalse(result["success"]) + self.assertIn("no durable lock", result["reasons"][0]) + + def test_legacy_lock_is_refused_until_rebound(self): + self.write_lock(lifecycle=None) + result = self._heartbeat() + self.assertFalse(result["success"]) + self.assertTrue(result["legacy_lease"]) + self.assertIn("rebound", " ".join(result["reasons"])) + + +class TestLegacyRebind(_LockFixture): + """AC-N8 exit route: canonical exact-owner rebinding.""" + + def _rebind(self, **kwargs): + params = { + "remote": REMOTE, + "org": ORG, + "repo": REPO, + "issue_number": ISSUE, + "branch_name": BRANCH, + "worktree_path": self.worktree, + "identity": IDENTITY, + "profile": PROFILE, + "lock_dir": self.lock_dir.name, + "now": self.now, + } + params.update(kwargs) + return issue_lock_store.rebind_legacy_lock(**params) + + def test_rebind_mints_a_session_and_a_genuine_first_heartbeat(self): + self.write_lock( + lifecycle=None, + created_delta=timedelta(hours=3), + heartbeat_delta=timedelta(hours=3), + expires_delta=timedelta(hours=1), + generation=2, + ) + result = self._rebind() + self.assertTrue(result["success"], result) + self.assertTrue(result["task_session_id"]) + self.assertEqual(result["lock_generation"], 3) + + written = issue_lock_store.read_lock_file(self._path()) + lease = written["work_lease"] + self.assertEqual( + lease["lifecycle_version"], lease_policy.LIFECYCLE_HEARTBEAT_V1 + ) + self.assertEqual(lease["last_heartbeat_at"], _ts(self.now)) + self.assertEqual(lease["expires_at"], _ts(self.now + timedelta(minutes=10))) + self.assertFalse(issue_lock_store.is_legacy_lease(written)) + # The original claim is preserved for audit rather than overwritten. + origin = written["legacy_rebind"]["legacy_origin"] + self.assertTrue(origin["created_at"]) + self.assertEqual(origin["lifecycle"], lease_policy.LIFECYCLE_LEGACY) + + def test_rebound_lock_can_then_heartbeat(self): + self.write_lock( + lifecycle=None, + created_delta=timedelta(hours=3), + heartbeat_delta=timedelta(hours=3), + expires_delta=timedelta(hours=1), + ) + rebound = self._rebind() + beat = issue_lock_store.heartbeat_session_lock( + remote=REMOTE, + org=ORG, + repo=REPO, + issue_number=ISSUE, + branch_name=BRANCH, + worktree_path=self.worktree, + identity=IDENTITY, + profile=PROFILE, + task_session_id=rebound["task_session_id"], + lock_dir=self.lock_dir.name, + now=self.now + timedelta(minutes=1), + ) + self.assertTrue(beat["success"], beat) + + def test_rebind_refuses_a_foreign_owner(self): + self.write_lock(lifecycle=None, expires_delta=timedelta(hours=1)) + result = self._rebind(identity="someone-else") + self.assertFalse(result["success"]) + + def test_rebind_refuses_a_lock_already_on_the_lifecycle(self): + self.write_lock() + result = self._rebind() + self.assertFalse(result["success"]) + self.assertFalse(result["legacy_lease"]) + + def test_rebind_is_not_a_recovery_path_for_a_lapsed_legacy_lease(self): + """An expired legacy lease belongs to #760 renewal or #601 reclaim.""" + self.write_lock( + lifecycle=None, + created_delta=timedelta(hours=5), + heartbeat_delta=timedelta(hours=5), + expires_delta=timedelta(hours=-1), + ) + result = self._rebind() + self.assertFalse(result["success"]) + self.assertIn("not a recovery path", " ".join(result["reasons"])) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_issue_794_conflict_gate_reclaim.py b/tests/test_issue_794_conflict_gate_reclaim.py new file mode 100644 index 0000000..16f000c --- /dev/null +++ b/tests/test_issue_794_conflict_gate_reclaim.py @@ -0,0 +1,193 @@ +"""Conflict gate honors heartbeat-lifecycle non-live reclaim bands (#790 review #502/#516). + +``assess_expired_lock_reclaim`` already permits reclaim for the heartbeat-lifecycle +stale bands (``stale_missed_heartbeat``, ``stale_absolute_cap``) without a dead PID. +But ``assess_same_issue_lease_conflict`` used to enter that reclaim branch only under +``is_lease_expired`` (``expires_at <= now``). For a heartbeat-lifecycle lease that is +non-live yet whose ``expires_at`` is still in the future, the acquisition gate fell +through to the foreign "already has an active lease" block and never consulted the +reclaim assessor — so the load-bearing heartbeat was not load-bearing for foreign +reclaim, the exact abandonment scenario #790 exists to fix. + +Two future-``expires_at`` non-live shapes are reachable: + +* ``stale_absolute_cap`` — a session that keeps heartbeating past the 8h absolute cap + has ``expires_at = last_heartbeat + TTL`` in the future (default policy). +* ``stale_missed_heartbeat`` — under an independent TTL>grace policy the heartbeat + grace lapses while ``expires_at`` is still ahead. + +These tests pin: both reclaim from a foreign acquirer; a live lease still blocks a +foreign acquirer; the same owner may reclaim its own abandoned heartbeat lease; and a +legacy (pre-lifecycle) lock keeps its absolute-``expires_at`` clock — a non-expired +legacy lock with a dead PID is *not* reclaimable through this path. +""" + +from __future__ import annotations + +import os +import sys +import unittest +from datetime import datetime, timedelta, timezone +from unittest import mock + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import issue_lock_store as ils # noqa: E402 +import lease_policy # noqa: E402 + +ISSUE = 790 +OWNER_BRANCH = f"fix/issue-{ISSUE}-slice-a-heartbeat-policy" +OWNER_WORKTREE = "/tmp/wt-790-owner" +FOREIGN_BRANCH = f"fix/issue-{ISSUE}-foreign-attempt" +FOREIGN_WORKTREE = "/tmp/wt-790-foreign" + + +def _ts(moment: datetime) -> str: + return ( + moment.astimezone(timezone.utc) + .replace(microsecond=0) + .isoformat() + .replace("+00:00", "Z") + ) + + +class _ConflictGateBase(unittest.TestCase): + def setUp(self): + self.now = datetime(2026, 7, 23, 12, 0, 0, tzinfo=timezone.utc) + + def _lock( + self, + *, + lifecycle: str | None, + created_ago: timedelta, + heartbeat_ago: timedelta, + expires_in: timedelta, + pid: int, + ) -> dict: + lease: dict = { + "operation_type": ils.AUTHOR_ISSUE_WORK_LEASE, + "issue_number": ISSUE, + "branch": OWNER_BRANCH, + "worktree_path": OWNER_WORKTREE, + "created_at": _ts(self.now - created_ago), + "last_heartbeat_at": _ts(self.now - heartbeat_ago), + "expires_at": _ts(self.now + expires_in), + } + if lifecycle is not None: + lease["lifecycle_version"] = lifecycle + lease["task_session_id"] = "author_issue_work-deadbeefdeadbeef" + return { + "issue_number": ISSUE, + "branch_name": OWNER_BRANCH, + "remote": "prgs", + "org": "Scaled-Tech-Consulting", + "repo": "Gitea-Tools", + "worktree_path": OWNER_WORKTREE, + "session_pid": pid, + "pid": pid, + "work_lease": lease, + } + + def _foreign_conflict(self, existing: dict) -> str | None: + return ils.assess_same_issue_lease_conflict( + existing, + issue_number=ISSUE, + branch_name=FOREIGN_BRANCH, + worktree_path=FOREIGN_WORKTREE, + now=self.now, + ) + + def _same_owner_conflict(self, existing: dict) -> str | None: + return ils.assess_same_issue_lease_conflict( + existing, + issue_number=ISSUE, + branch_name=OWNER_BRANCH, + worktree_path=OWNER_WORKTREE, + now=self.now, + ) + + +class TestHeartbeatNonLiveFutureExpiresReclaim(_ConflictGateBase): + def test_stale_absolute_cap_future_expires_allows_foreign_reclaim(self): + # Created >8h ago, heartbeated one minute ago, expires 9 min in the FUTURE, + # owner PID alive: freshness = stale_absolute_cap, live=False, not expired. + existing = self._lock( + lifecycle=lease_policy.LIFECYCLE_HEARTBEAT_V1, + created_ago=timedelta(hours=9), + heartbeat_ago=timedelta(minutes=1), + expires_in=timedelta(minutes=9), + pid=os.getpid(), + ) + freshness = ils.assess_lock_freshness(existing, now=self.now) + self.assertEqual(freshness["status"], ils.STATUS_STALE_ABSOLUTE_CAP) + self.assertFalse(freshness["live"]) + self.assertFalse(ils.is_lease_expired(existing, now=self.now)) + with mock.patch.object(ils, "is_process_alive", return_value=True): + self.assertIsNone(self._foreign_conflict(existing)) + + def test_missed_heartbeat_ttl_gt_grace_future_expires_allows_foreign_reclaim(self): + # Heartbeat grace (default 10 min) lapsed 5 min ago, but a TTL>grace policy + # leaves expires_at 20 min in the FUTURE: stale_missed_heartbeat, not expired. + existing = self._lock( + lifecycle=lease_policy.LIFECYCLE_HEARTBEAT_V1, + created_ago=timedelta(minutes=30), + heartbeat_ago=timedelta(minutes=15), + expires_in=timedelta(minutes=20), + pid=os.getpid(), + ) + freshness = ils.assess_lock_freshness(existing, now=self.now) + self.assertEqual(freshness["status"], ils.STATUS_STALE_MISSED_HEARTBEAT) + self.assertFalse(freshness["live"]) + self.assertFalse(ils.is_lease_expired(existing, now=self.now)) + with mock.patch.object(ils, "is_process_alive", return_value=True): + self.assertIsNone(self._foreign_conflict(existing)) + + def test_same_owner_may_reclaim_its_own_abandoned_heartbeat_lease(self): + existing = self._lock( + lifecycle=lease_policy.LIFECYCLE_HEARTBEAT_V1, + created_ago=timedelta(hours=9), + heartbeat_ago=timedelta(minutes=1), + expires_in=timedelta(minutes=9), + pid=os.getpid(), + ) + with mock.patch.object(ils, "is_process_alive", return_value=True): + self.assertIsNone(self._same_owner_conflict(existing)) + + +class TestLiveAndLegacyStillBlockForeign(_ConflictGateBase): + def test_live_heartbeat_lease_still_blocks_foreign(self): + existing = self._lock( + lifecycle=lease_policy.LIFECYCLE_HEARTBEAT_V1, + created_ago=timedelta(minutes=5), + heartbeat_ago=timedelta(minutes=1), + expires_in=timedelta(minutes=9), + pid=os.getpid(), + ) + freshness = ils.assess_lock_freshness(existing, now=self.now) + self.assertTrue(freshness["live"]) + with mock.patch.object(ils, "is_process_alive", return_value=True): + block = self._foreign_conflict(existing) + self.assertIn("already has an active", block or "") + + def test_legacy_lease_future_expires_dead_pid_is_not_reclaimed_here(self): + # AC-N8: a legacy lock keeps its absolute expires_at clock. Non-expired + + # dead PID is non-live, but the widened band excludes legacy, so the + # foreign acquirer is still blocked rather than silently reclaiming. + existing = self._lock( + lifecycle=None, + created_ago=timedelta(hours=9), + heartbeat_ago=timedelta(hours=9), + expires_in=timedelta(hours=2), + pid=4_194_304, # far above any live pid on a test host + ) + with mock.patch.object(ils, "is_process_alive", return_value=False): + freshness = ils.assess_lock_freshness(existing, now=self.now) + self.assertTrue(freshness["legacy_lease"]) + self.assertFalse(freshness["live"]) + self.assertFalse(ils.is_lease_expired(existing, now=self.now)) + block = self._foreign_conflict(existing) + self.assertIn("already has an active", block or "") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_task_capability_role_invariants.py b/tests/test_task_capability_role_invariants.py index fbba796..25e89a8 100644 --- a/tests/test_task_capability_role_invariants.py +++ b/tests/test_task_capability_role_invariants.py @@ -139,6 +139,8 @@ EXPECTED_ROLE_EXCLUSIVE_TASKS = frozenset( "gitea_release_merger_pr_lease", "create_branch", "push_branch", + "bootstrap_author_issue_worktree", + "gitea_bootstrap_author_issue_worktree", # #812 AC20: publishing an unpublished local head is author-only for the # same reason every other push is — it writes a branch to the remote. "publish_unpublished_branch", diff --git a/tests/test_webui_analytics.py b/tests/test_webui_analytics.py new file mode 100644 index 0000000..bf1a419 --- /dev/null +++ b/tests/test_webui_analytics.py @@ -0,0 +1,252 @@ +"""Unit and integration tests for Model Usage & Performance Analytics (#651).""" + +from __future__ import annotations + +import os +import tempfile +import unittest +from starlette.testclient import TestClient + +import control_plane_db +from webui.analytics_loader import ( + ANALYTICS_SCHEMA_VERSION, + compute_percentile, + load_analytics, + record_usage, +) +from webui.app import create_app +from webui import console_redaction + + +class AnalyticsLoaderTest(unittest.TestCase): + + def setUp(self) -> None: + self.temp_dir = tempfile.TemporaryDirectory() + self.db_path = os.path.join(self.temp_dir.name, "test_control_plane.sqlite3") + os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path + self.db = control_plane_db.ControlPlaneDB(db_path=self.db_path) + + def tearDown(self) -> None: + self.temp_dir.cleanup() + + def test_compute_percentile(self) -> None: + self.assertIsNone(compute_percentile([], 50.0)) + self.assertEqual(compute_percentile([100], 50.0), 100.0) + + # 2 elements: [100, 200] + self.assertEqual(compute_percentile([100, 200], 50.0), 150.0) + + # 100 elements: 1..100 + vals = list(range(1, 101)) + self.assertEqual(compute_percentile(vals, 50.0), 50.5) + self.assertAlmostEqual(compute_percentile(vals, 90.0), 90.1) + + def test_record_and_aggregate_usage(self) -> None: + # Record event 1 (complete data) + u1 = record_usage( + db_path=self.db_path, + remote="dadeschools", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + role="author", + model="gemini-3.6-flash", + issue_number=651, + stage="implementation", + input_tokens=1000, + output_tokens=500, + estimated_cost_usd=0.0015, + latency_ms=200, + duration_ms=3000, + metadata={"secret_key": "secret123", "note": "token=secret123"}, + ) + self.assertGreater(u1, 0) + + # Record event 2 (missing tokens and cost -> unknown) + u2 = record_usage( + db_path=self.db_path, + remote="dadeschools", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + role="reviewer", + model="claude-3-5-sonnet", + pr_number=846, + stage="review", + latency_ms=500, + duration_ms=6000, + ) + self.assertGreater(u2, u1) + + snapshot = load_analytics( + db_path=self.db_path, + remote="dadeschools", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + ) + + self.assertTrue(snapshot.ok) + self.assertEqual(snapshot.schema_version, ANALYTICS_SCHEMA_VERSION) + self.assertEqual(snapshot.total_events, 2) + + # Verify overall summary + summary = snapshot.overall_summary + self.assertEqual(summary.total_events, 2) + self.assertEqual(summary.events_with_tokens, 1) + self.assertEqual(summary.total_tokens, 1500) + self.assertEqual(summary.events_with_cost, 1) + self.assertEqual(summary.estimated_cost_usd, 0.0015) + self.assertEqual(summary.events_with_latency, 2) + self.assertEqual(summary.latency_p50_ms, 350.0) + + # Verify missing data handling (AC 3: not zero-fabricated) + reviewer_model = snapshot.by_model.get("claude-3-5-sonnet") + self.assertIsNotNone(reviewer_model) + self.assertEqual(reviewer_model.total_events, 1) + self.assertEqual(reviewer_model.events_with_tokens, 0) + self.assertIsNone(reviewer_model.total_tokens) + self.assertEqual(reviewer_model.display_tokens, "Unknown") + self.assertEqual(reviewer_model.events_with_cost, 0) + self.assertIsNone(reviewer_model.estimated_cost_usd) + self.assertEqual(reviewer_model.display_cost, "Unknown") + + # Verify redaction (AC 4) + e1 = [e for e in snapshot.events if e.usage_id == u1][0] + self.assertIsNotNone(e1.metadata) + self.assertNotIn("secret123", e1.metadata) + self.assertIn("[REDACTED]", e1.metadata) + + def test_missing_db_fail_soft(self) -> None: + invalid_path = "/nonexistent_path_dir/db.sqlite3" + snapshot = load_analytics(db_path=invalid_path) + self.assertFalse(snapshot.ok) + self.assertIn("control_plane_db_unavailable", snapshot.reason) + self.assertEqual(snapshot.overall_summary.display_tokens, "Unknown") + + +class AnalyticsWebUITest(unittest.TestCase): + + def setUp(self) -> None: + self.temp_dir = tempfile.TemporaryDirectory() + self.db_path = os.path.join(self.temp_dir.name, "test_webui.sqlite3") + os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path + self.app = create_app() + self.client = TestClient(self.app) + + record_usage( + db_path=self.db_path, + remote="dadeschools", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + role="author", + model="gemini-3.6-flash", + issue_number=651, + stage="implementation", + input_tokens=2000, + output_tokens=1000, + estimated_cost_usd=0.003, + latency_ms=150, + duration_ms=2500, + ) + + def tearDown(self) -> None: + self.temp_dir.cleanup() + + def test_analytics_html_route(self) -> None: + response = self.client.get("/analytics") + self.assertEqual(response.status_code, 200) + self.assertIn("Model Usage & Performance Analytics", response.text) + self.assertIn("gemini-3.6-flash", response.text) + self.assertIn("3,000", response.text) + + def test_analytics_api_route(self) -> None: + response = self.client.get("/api/v1/analytics") + self.assertEqual(response.status_code, 200) + data = response.json() + self.assertTrue(data["ok"]) + self.assertEqual(data["total_events"], 1) + self.assertIn("gemini-3.6-flash", data["by_model"]) + + def test_analytics_ingest_unauthorized_denied(self) -> None: + """F2: unauthenticated POST must not write the control-plane DB.""" + payload = { + "remote": "dadeschools", + "org": "Scaled-Tech-Consulting", + "repo": "Gitea-Tools", + "role": "reviewer", + "model": "claude-3-5-sonnet", + "pr_number": 846, + "stage": "review", + "input_tokens": 500, + "output_tokens": 100, + "latency_ms": 400, + "metadata": "Review note token=secret456", + } + response = self.client.post("/api/v1/analytics/usage", json=payload) + self.assertEqual(response.status_code, 403) + res_json = response.json() + self.assertFalse(res_json.get("ok", True)) + self.assertEqual(res_json.get("error"), "unauthorized") + authorization = res_json.get("authorization") or {} + self.assertFalse(authorization.get("allowed")) + self.assertFalse(authorization.get("execution_enabled")) + + # No new row written + res2 = self.client.get("/api/v1/analytics") + self.assertEqual(res2.status_code, 200) + self.assertEqual(res2.json()["total_events"], 1) + + def test_html_escapes_script_bearing_model_role_stage(self) -> None: + """F1: stored XSS — dynamic model/role/stage must render escaped.""" + xss = '' + record_usage( + db_path=self.db_path, + remote="dadeschools", + org="Scaled-Tech-Consulting", + repo="Gitea-Tools", + role=xss, + model=xss, + stage=xss, + issue_number=999, + status="success", + ) + response = self.client.get("/analytics") + self.assertEqual(response.status_code, 200) + # Raw tag must not appear; escaped form must. + self.assertNotIn("", response.text) + self.assertIn("<script>alert(1)</script>", response.text) + + def test_load_analytics_coerces_none_scope(self) -> None: + """F4: None remote/org/repo become empty strings, never None.""" + snapshot = load_analytics(db_path=self.db_path, remote=None, org=None, repo=None) + self.assertIsInstance(snapshot.remote, str) + self.assertIsInstance(snapshot.org, str) + self.assertIsInstance(snapshot.repo, str) + self.assertEqual(snapshot.remote, "") + self.assertEqual(snapshot.org, "") + self.assertEqual(snapshot.repo, "") + + def test_usage_events_retention_max_rows(self) -> None: + """F3: record_usage_event enforces USAGE_EVENTS_MAX_ROWS.""" + db = control_plane_db.ControlPlaneDB(db_path=self.db_path) + original_max = db.USAGE_EVENTS_MAX_ROWS + try: + db.USAGE_EVENTS_MAX_ROWS = 3 + for i in range(5): + db.record_usage_event( + remote="dadeschools", + org="org", + repo="repo", + role="author", + model=f"model-{i}", + stage="test", + ) + rows = db.query_usage_events(limit=100) + self.assertLessEqual(len(rows), 3) + # Newest three retained + models = {r["model"] for r in rows} + self.assertEqual(models, {"model-2", "model-3", "model-4"}) + finally: + db.USAGE_EVENTS_MAX_ROWS = original_max + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_webui_sanctioned_restart.py b/tests/test_webui_sanctioned_restart.py new file mode 100644 index 0000000..a495935 --- /dev/null +++ b/tests/test_webui_sanctioned_restart.py @@ -0,0 +1,502 @@ +"""Sanctioned restart / graceful reload control tests (#642). + +Acceptance criteria under test: + +1. The sanctioned restart path is implemented behind gates (capability, + confirmation, operator authorization, host hook). +2. Manual ``pkill`` stays forbidden and is classified as contamination. +3. Post-restart mutations require clean health/session proof. +4. Authorized restart preview, unauthorized deny, contamination classification. +5. No entry point exposes a raw kill. +""" + +import json +import os +import tempfile +import unittest + +import mcp_namespace_health +import runtime_recovery_guard +from task_capability_map import TASK_CAPABILITY_MAP +from webui import console_audit, console_authz, gated_actions, sanctioned_restart + +NAMESPACE = "gitea-author" + +# An operator-authorized, hook-configured host. Passed explicitly so no test +# depends on (or mutates) the real process environment. +READY_ENV = { + sanctioned_restart.RESTART_HOOK_ENV: "launchd:cc.prgs.gitea-author", + runtime_recovery_guard.OPERATOR_AUTHORIZATION_ENV: "ops-ticket-4821", +} + + +def admin(subject: str = "admin@example.test") -> console_authz.Principal: + return console_authz.Principal( + subject=subject, + role=console_authz.ADMIN, + identity_source=console_authz.IDENTITY_ACCESS_PROXY, + authenticated=True, + ) + + +def viewer() -> console_authz.Principal: + return console_authz.Principal( + subject="viewer@example.test", + role=console_authz.VIEWER, + identity_source=console_authz.IDENTITY_ACCESS_PROXY, + authenticated=True, + ) + + +class TestCapabilityWiring(unittest.TestCase): + """AC1: authority is declared, not invented by the console.""" + + def test_actions_resolve_through_the_capability_map(self): + for action_id in ( + sanctioned_restart.ACTION_RESTART_NAMESPACE, + sanctioned_restart.ACTION_RELOAD_NAMESPACE, + ): + with self.subTest(action=action_id): + action = console_authz.get_action(action_id) + self.assertIsNotNone(action) + self.assertIn(action.task_key, TASK_CAPABILITY_MAP) + self.assertEqual( + action.mcp_permission, + TASK_CAPABILITY_MAP[action.task_key]["permission"], + ) + + def test_restart_permission_is_not_a_gitea_operation(self): + """No configured Gitea profile should satisfy a host restart.""" + permission = TASK_CAPABILITY_MAP["restart_namespace"]["permission"] + self.assertFalse(permission.startswith("gitea.")) + + def test_restart_is_destructive_dual_control_break_glass(self): + action = console_authz.get_action( + sanctioned_restart.ACTION_RESTART_NAMESPACE + ) + self.assertEqual(action.action_class, console_authz.CLASS_DESTRUCTIVE) + self.assertEqual(action.minimum_role, console_authz.ADMIN) + self.assertTrue(action.dual_control) + self.assertTrue(action.break_glass) + self.assertTrue(action.requires_confirmation) + + def test_reload_is_privileged_but_not_destructive(self): + action = console_authz.get_action( + sanctioned_restart.ACTION_RELOAD_NAMESPACE + ) + self.assertEqual(action.action_class, console_authz.CLASS_PRIVILEGED) + self.assertTrue(action.requires_confirmation) + + +class TestPreview(unittest.TestCase): + """AC4: an authorized preview renders the plan without executing it.""" + + def test_preview_lists_the_mutation_ledger(self): + preview = sanctioned_restart.build_restart_preview( + NAMESPACE, principal=admin(), env=READY_ENV + ) + steps = [entry["step"] for entry in preview["mutation_ledger"]] + self.assertEqual( + steps, ["quiesce", "host_restart_hook", "health_recheck", "audit"] + ) + self.assertTrue(preview["scope_valid"]) + self.assertTrue(preview["post_restart_verification_required"]) + + def test_reload_preview_drains_instead_of_restarting(self): + preview = sanctioned_restart.build_restart_preview( + NAMESPACE, sanctioned_restart.MODE_RELOAD, + principal=admin(), env=READY_ENV, + ) + steps = [entry["step"] for entry in preview["mutation_ledger"]] + self.assertIn("host_graceful_reload", steps) + self.assertNotIn("host_restart_hook", steps) + + def test_preview_never_enables_execution(self): + preview = sanctioned_restart.build_restart_preview( + NAMESPACE, principal=admin(), env=READY_ENV + ) + self.assertFalse(preview["execution_enabled"]) + self.assertFalse(preview["authorization"]["execution_enabled"]) + + def test_confirmation_phrase_binds_the_namespace(self): + self.assertTrue( + sanctioned_restart.confirmation_matches( + NAMESPACE, sanctioned_restart.MODE_RESTART, + "restart gitea-author", + ) + ) + # A phrase typed for one namespace must not authorize another. + self.assertFalse( + sanctioned_restart.confirmation_matches( + "gitea-merger", sanctioned_restart.MODE_RESTART, + "restart gitea-author", + ) + ) + + +class TestGates(unittest.TestCase): + """AC1/AC4: every gate denies with a stable reason code.""" + + def _assess(self, **kwargs): + params = { + "principal": admin(), + "confirmation": f"restart {NAMESPACE}", + "env": READY_ENV, + } + params.update(kwargs) + namespace = params.pop("namespace", NAMESPACE) + mode = params.pop("mode", sanctioned_restart.MODE_RESTART) + return sanctioned_restart.assess_restart_request( + namespace, mode, **params + ) + + def test_authorized_confirmed_request_passes_every_gate(self): + result = self._assess() + self.assertTrue(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.ALLOW_HOST_ACTION_REQUIRED + ) + + def test_passing_every_gate_is_not_an_execution_grant(self): + """An allowed request still never lets the console touch the process.""" + result = self._assess() + self.assertTrue(result["allowed"]) + self.assertFalse(result["execution_enabled"]) + self.assertFalse(result["console_executes"]) + + def test_unauthorized_principal_is_denied(self): + result = self._assess(principal=viewer()) + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.DENY_UNAUTHORIZED + ) + + def test_anonymous_principal_is_denied(self): + result = self._assess(principal=None) + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.DENY_UNAUTHORIZED + ) + + def test_missing_confirmation_is_denied(self): + result = self._assess(confirmation=None) + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.DENY_CONFIRMATION_MISSING + ) + + def test_confirmation_for_another_namespace_is_denied(self): + result = self._assess(confirmation="restart gitea-merger") + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.DENY_CONFIRMATION_MISMATCH + ) + + def test_missing_operator_authorization_is_denied(self): + env = {sanctioned_restart.RESTART_HOOK_ENV: "launchd:cc.prgs.author"} + result = self._assess(env=env) + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], + sanctioned_restart.DENY_OPERATOR_AUTHORIZATION, + ) + + def test_missing_host_hook_is_denied_without_kill_fallback(self): + env = { + runtime_recovery_guard.OPERATOR_AUTHORIZATION_ENV: "ops-ticket-1", + } + result = self._assess(env=env) + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.DENY_HOOK_NOT_CONFIGURED + ) + + def test_fleet_scope_is_refused(self): + for scope in ("all", "*", "fleet"): + with self.subTest(scope=scope): + result = self._assess( + namespace=scope, confirmation=f"restart {scope}" + ) + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.DENY_FLEET_SCOPE + ) + + def test_unknown_namespace_is_refused(self): + result = self._assess( + namespace="gitea-nope", confirmation="restart gitea-nope" + ) + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.DENY_UNKNOWN_NAMESPACE + ) + + def test_unknown_mode_is_refused(self): + result = self._assess(mode="obliterate") + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.DENY_UNKNOWN_MODE + ) + + def test_live_contamination_marker_blocks_restart(self): + marker = runtime_recovery_guard.build_contamination_record( + reason_class=runtime_recovery_guard.REASON_MANUAL_DAEMON_KILL, + command_redacted="pkill -f mcp_server.py", + ) + result = self._assess(contamination_marker=marker) + self.assertFalse(result["allowed"]) + self.assertEqual( + result["reason_code"], sanctioned_restart.DENY_CONTAMINATED_RUNTIME + ) + + def test_reconciler_cleared_marker_no_longer_blocks(self): + marker = runtime_recovery_guard.build_contamination_record( + reason_class=runtime_recovery_guard.REASON_MANUAL_DAEMON_KILL, + command_redacted="pkill -f mcp_server.py", + ) + marker = dict(marker, cleared_by_reconciler=True) + result = self._assess(contamination_marker=marker) + self.assertTrue(result["allowed"]) + + +class TestExecutionNeverKills(unittest.TestCase): + """AC5: no path exposes or runs a raw process kill.""" + + def test_authorized_execution_defers_to_the_host_supervisor(self): + result = sanctioned_restart.execute_restart( + NAMESPACE, + principal=admin(), + confirmation=f"restart {NAMESPACE}", + env=READY_ENV, + ) + self.assertTrue(result["allowed"]) + self.assertFalse(result["success"]) + self.assertFalse(result["process_kill_executed"]) + self.assertEqual( + result["outcome"], sanctioned_restart.ALLOW_HOST_ACTION_REQUIRED + ) + + def test_denied_execution_reports_the_refusing_gate(self): + result = sanctioned_restart.execute_restart( + NAMESPACE, principal=viewer(), confirmation=f"restart {NAMESPACE}", + env=READY_ENV, + ) + self.assertFalse(result["allowed"]) + self.assertEqual( + result["outcome"], sanctioned_restart.DENY_UNAUTHORIZED + ) + self.assertFalse(result["process_kill_executed"]) + + def test_module_never_spawns_a_process(self): + path = os.path.join( + os.path.dirname(os.path.dirname(os.path.abspath(__file__))), + "webui", "sanctioned_restart.py", + ) + with open(path, encoding="utf-8") as handle: + source = handle.read() + for forbidden in ( + "import subprocess", "import signal", "os.kill", "os.system", + "popen", + ): + with self.subTest(forbidden=forbidden): + self.assertNotIn(forbidden, source.lower()) + + def test_no_surface_returns_a_kill_command(self): + payloads = [ + sanctioned_restart.build_restart_preview( + NAMESPACE, principal=admin(), env=READY_ENV + ), + sanctioned_restart.restart_policy(), + sanctioned_restart.execute_restart( + NAMESPACE, principal=admin(), + confirmation=f"restart {NAMESPACE}", env=READY_ENV, + ), + ] + for payload in payloads: + rendered = json.dumps(payload, default=str).lower() + self.assertNotIn("kill -9", rendered) + self.assertNotIn("pkill -f", rendered) + + def test_policy_declares_no_raw_kill_and_no_silent_restart(self): + policy = sanctioned_restart.restart_policy() + self.assertFalse(policy["raw_kill_exposed"]) + self.assertFalse(policy["console_executes_process_kill"]) + self.assertFalse(policy["fleet_scope_permitted"]) + self.assertFalse(policy["silent_auto_restart_permitted"]) + self.assertTrue(policy["audit_required"]) + + +class TestContaminationClassification(unittest.TestCase): + """AC2: manual pkill is contamination, and it blocks clean claims.""" + + def test_manual_daemon_pkill_is_contamination(self): + result = sanctioned_restart.classify_restart_command( + "pkill -f mcp_server.py" + ) + self.assertTrue(result["contamination"]) + self.assertFalse(result["clean_claim_allowed"]) + self.assertIsNotNone(result["contamination_marker"]) + self.assertEqual( + result["sanctioned_alternative"], + sanctioned_restart.ACTION_RESTART_NAMESPACE, + ) + + def test_broad_process_kill_is_contamination(self): + result = sanctioned_restart.classify_restart_command("killall -9 Python") + self.assertTrue(result["contamination"]) + self.assertFalse(result["clean_claim_allowed"]) + + def test_marker_names_the_sanctioned_alternative(self): + result = sanctioned_restart.classify_restart_command( + "pkill -f mcp_server.py" + ) + marker = result["contamination_marker"] + self.assertIn( + sanctioned_restart.ACTION_RESTART_NAMESPACE, marker["detail"] + ) + + def test_benign_command_is_not_contamination(self): + result = sanctioned_restart.classify_restart_command("git status") + self.assertFalse(result["contamination"]) + self.assertTrue(result["clean_claim_allowed"]) + + def test_no_command_is_not_contamination(self): + result = sanctioned_restart.classify_restart_command(None) + self.assertFalse(result["contamination"]) + self.assertTrue(result["clean_claim_allowed"]) + + +class TestPostRestartHealth(unittest.TestCase): + """AC3: a clean post-restart claim needs live client-namespace proof.""" + + def test_live_client_probe_clears_the_session(self): + result = sanctioned_restart.verify_post_restart_health( + NAMESPACE, + probe_result={"success": True}, + probe_source=mcp_namespace_health.PROBE_SOURCE_CLIENT, + registered_tools=["gitea_whoami"], + required_tool="gitea_whoami", + ) + self.assertEqual(result["status"], sanctioned_restart.HEALTH_CLEAN) + self.assertTrue(result["clean_claim_allowed"]) + self.assertTrue(result["mutations_allowed"]) + + def test_offline_probe_does_not_clear_the_session(self): + result = sanctioned_restart.verify_post_restart_health( + NAMESPACE, + probe_result={"success": True}, + probe_source=mcp_namespace_health.PROBE_SOURCE_OFFLINE, + registered_tools=["gitea_whoami"], + required_tool="gitea_whoami", + ) + self.assertFalse(result["clean_claim_allowed"]) + self.assertFalse(result["mutations_allowed"]) + + def test_failed_probe_is_unhealthy(self): + result = sanctioned_restart.verify_post_restart_health( + NAMESPACE, + probe_result={"success": False, "error": "client is closing: EOF"}, + probe_source=mcp_namespace_health.PROBE_SOURCE_CLIENT, + registered_tools=["gitea_whoami"], + required_tool="gitea_whoami", + ) + self.assertEqual(result["status"], sanctioned_restart.HEALTH_UNHEALTHY) + self.assertFalse(result["clean_claim_allowed"]) + + def test_static_registration_alone_never_clears_the_session(self): + result = sanctioned_restart.verify_post_restart_health( + NAMESPACE, + registered_tools=["gitea_whoami"], + required_tool="gitea_whoami", + ) + self.assertFalse(result["clean_claim_allowed"]) + + +class TestAuditEmission(unittest.TestCase): + """Every restart attempt is audited with actor, target, and result.""" + + def _run(self, principal, sink): + prior = os.environ.get(console_audit.AUDIT_LOG_ENV) + os.environ[console_audit.AUDIT_LOG_ENV] = sink + try: + return sanctioned_restart.execute_restart( + NAMESPACE, + principal=principal, + confirmation=f"restart {NAMESPACE}", + env=READY_ENV, + request_id="req-642", + ) + finally: + if prior is None: + os.environ.pop(console_audit.AUDIT_LOG_ENV, None) + else: + os.environ[console_audit.AUDIT_LOG_ENV] = prior + + def test_allowed_attempt_is_written_with_actor_and_target(self): + with tempfile.TemporaryDirectory() as tmp: + sink = os.path.join(tmp, "audit.jsonl") + result = self._run(admin(), sink) + self.assertTrue(result["audit"]["written"]) + with open(sink, encoding="utf-8") as handle: + record = json.loads(handle.read().strip()) + self.assertEqual( + record["action"], sanctioned_restart.ACTION_RESTART_NAMESPACE + ) + self.assertEqual(record["target"]["namespace"], NAMESPACE) + self.assertEqual(record["target"]["mode"], "restart") + self.assertEqual(record["result"], console_audit.RESULT_ALLOWED) + self.assertEqual(record["actor"]["subject"], "admin@example.test") + self.assertFalse(record["metadata"]["process_kill_executed"]) + + def test_denied_attempt_is_audited_too(self): + with tempfile.TemporaryDirectory() as tmp: + sink = os.path.join(tmp, "audit.jsonl") + self._run(viewer(), sink) + with open(sink, encoding="utf-8") as handle: + record = json.loads(handle.read().strip()) + self.assertEqual(record["result"], console_audit.RESULT_DENIED) + self.assertEqual( + record["reason_code"], sanctioned_restart.DENY_UNAUTHORIZED + ) + + def test_restart_audit_uses_break_glass_retention(self): + action = console_authz.get_action( + sanctioned_restart.ACTION_RESTART_NAMESPACE + ) + self.assertEqual( + console_audit.retention_class_for(action), + console_audit.RETENTION_BREAK_GLASS, + ) + + +class TestRegistrySurface(unittest.TestCase): + """AC5: the console surfaces the control, still disabled, with no kill.""" + + def test_registry_exposes_both_actions_disabled(self): + registry = gated_actions.load_action_registry() + for action_id in ( + sanctioned_restart.ACTION_RESTART_NAMESPACE, + sanctioned_restart.ACTION_RELOAD_NAMESPACE, + ): + with self.subTest(action=action_id): + action = registry.get(action_id) + self.assertIsNotNone(action) + self.assertFalse(action.enabled) + + def test_registry_preview_names_the_namespace_target(self): + preview = gated_actions.preview_action( + sanctioned_restart.ACTION_RESTART_NAMESPACE, namespace=NAMESPACE + ) + target = preview["mutation_ledger"][0]["target"] + self.assertIn(NAMESPACE, target) + self.assertFalse(preview["enabled"]) + + def test_registry_attempt_fails_closed(self): + result = gated_actions.attempt_action( + sanctioned_restart.ACTION_RESTART_NAMESPACE, namespace=NAMESPACE + ) + self.assertFalse(result["success"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_workspace_guard_alignment.py b/tests/test_workspace_guard_alignment.py index 3d41291..b65b7da 100644 --- a/tests/test_workspace_guard_alignment.py +++ b/tests/test_workspace_guard_alignment.py @@ -35,8 +35,10 @@ class TestCanonicalRepoRoot(unittest.TestCase): self.assertEqual(root, CONTROL_ROOT) def test_falls_back_when_git_unavailable(self): + # When fallback is a path under /branches/, recover + # via commonpath ancestry (never string-split on "/branches/"). root = amw.resolve_canonical_repo_root("/missing/path", MCP_PROCESS_ROOT) - self.assertEqual(root, os.path.realpath(MCP_PROCESS_ROOT)) + self.assertEqual(root, os.path.realpath(CONTROL_ROOT)) class TestWorkspaceRepoMembership(unittest.TestCase): diff --git a/webui/analytics_loader.py b/webui/analytics_loader.py new file mode 100644 index 0000000..15a6138 --- /dev/null +++ b/webui/analytics_loader.py @@ -0,0 +1,433 @@ +"""Model usage, token cost, latency, and workflow-performance analytics (#651, Phase 4). + +Ingests session instrumentation metrics, aggregates usage/cost/latency percentiles +by project, role, model, issue/PR, and stage, enforcing secret redaction and +explicitly rendering missing metrics as "Unknown" without zero-fabrication. +""" + +from __future__ import annotations + +import math +from dataclasses import asdict, dataclass +from typing import Any, Sequence + +import control_plane_db +from webui import console_redaction + +ANALYTICS_SCHEMA_VERSION = 1 + + +@dataclass(frozen=True) +class UsageEvent: + usage_id: int + session_id: str | None + remote: str + org: str + repo: str + project_id: str | None + role: str + model: str + issue_number: int | None + pr_number: int | None + stage: str + input_tokens: int | None + output_tokens: int | None + total_tokens: int | None + estimated_cost_usd: float | None + latency_ms: int | None + duration_ms: int | None + status: str + metadata: str | None + created_at: str + + def to_dict(self) -> dict[str, Any]: + d = asdict(self) + if d["metadata"]: + d["metadata"] = console_redaction.redact_text(d["metadata"]) + return d + + +@dataclass(frozen=True) +class GroupMetrics: + name: str + total_events: int + events_with_tokens: int + input_tokens: int | None + output_tokens: int | None + total_tokens: int | None + events_with_cost: int + estimated_cost_usd: float | None + events_with_latency: int + latency_p50_ms: float | None + latency_p90_ms: float | None + latency_p95_ms: float | None + latency_p99_ms: float | None + latency_avg_ms: float | None + events_with_duration: int + duration_avg_ms: float | None + display_tokens: str + display_cost: str + display_latency_p50: str + display_latency_p90: str + display_duration_avg: str + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +@dataclass(frozen=True) +class AnalyticsSnapshot: + ok: bool + reason: str + schema_version: int + remote: str + org: str + repo: str + total_events: int + overall_summary: GroupMetrics + by_project: dict[str, GroupMetrics] + by_role: dict[str, GroupMetrics] + by_model: dict[str, GroupMetrics] + by_work_item: dict[str, GroupMetrics] + by_stage: dict[str, GroupMetrics] + events: tuple[UsageEvent, ...] + + def to_dict(self) -> dict[str, Any]: + return { + "ok": self.ok, + "reason": self.reason, + "schema_version": self.schema_version, + "remote": self.remote, + "org": self.org, + "repo": self.repo, + "total_events": self.total_events, + "overall_summary": self.overall_summary.to_dict(), + "by_project": {k: v.to_dict() for k, v in self.by_project.items()}, + "by_role": {k: v.to_dict() for k, v in self.by_role.items()}, + "by_model": {k: v.to_dict() for k, v in self.by_model.items()}, + "by_work_item": {k: v.to_dict() for k, v in self.by_work_item.items()}, + "by_stage": {k: v.to_dict() for k, v in self.by_stage.items()}, + "events": [e.to_dict() for e in self.events], + } + + +def compute_percentile(values: Sequence[float | int], percentile: float) -> float | None: + if not values: + return None + sorted_vals = sorted(values) + n = len(sorted_vals) + if n == 1: + return float(sorted_vals[0]) + k = (n - 1) * (percentile / 100.0) + f = math.floor(k) + c = math.ceil(k) + if f == c: + return float(sorted_vals[int(f)]) + d0 = sorted_vals[int(f)] * (c - k) + d1 = sorted_vals[int(c)] * (k - f) + return float(d0 + d1) + + +def aggregate_events(group_name: str, events: Sequence[UsageEvent]) -> GroupMetrics: + total_events = len(events) + if total_events == 0: + return GroupMetrics( + name=group_name, + total_events=0, + events_with_tokens=0, + input_tokens=None, + output_tokens=None, + total_tokens=None, + events_with_cost=0, + estimated_cost_usd=None, + events_with_latency=0, + latency_p50_ms=None, + latency_p90_ms=None, + latency_p95_ms=None, + latency_p99_ms=None, + latency_avg_ms=None, + events_with_duration=0, + duration_avg_ms=None, + display_tokens="Unknown", + display_cost="Unknown", + display_latency_p50="Unknown", + display_latency_p90="Unknown", + display_duration_avg="Unknown", + ) + + token_events = [ + e for e in events + if e.total_tokens is not None or e.input_tokens is not None or e.output_tokens is not None + ] + events_with_tokens = len(token_events) + if events_with_tokens > 0: + input_tokens = sum(e.input_tokens or 0 for e in token_events) + output_tokens = sum(e.output_tokens or 0 for e in token_events) + total_tokens = sum( + e.total_tokens if e.total_tokens is not None else ((e.input_tokens or 0) + (e.output_tokens or 0)) + for e in token_events + ) + display_tokens = f"{total_tokens:,}" + else: + input_tokens = None + output_tokens = None + total_tokens = None + display_tokens = "Unknown" + + cost_events = [e for e in events if e.estimated_cost_usd is not None] + events_with_cost = len(cost_events) + if events_with_cost > 0: + estimated_cost_usd = round(sum(e.estimated_cost_usd for e in cost_events), 6) + display_cost = f"${estimated_cost_usd:.4f}" + else: + estimated_cost_usd = None + display_cost = "Unknown" + + latency_vals = [e.latency_ms for e in events if e.latency_ms is not None] + events_with_latency = len(latency_vals) + if events_with_latency > 0: + latency_p50_ms = compute_percentile(latency_vals, 50.0) + latency_p90_ms = compute_percentile(latency_vals, 90.0) + latency_p95_ms = compute_percentile(latency_vals, 95.0) + latency_p99_ms = compute_percentile(latency_vals, 99.0) + latency_avg_ms = round(sum(latency_vals) / events_with_latency, 2) + display_latency_p50 = f"{round(latency_p50_ms, 1)} ms" if latency_p50_ms is not None else "Unknown" + display_latency_p90 = f"{round(latency_p90_ms, 1)} ms" if latency_p90_ms is not None else "Unknown" + else: + latency_p50_ms = None + latency_p90_ms = None + latency_p95_ms = None + latency_p99_ms = None + latency_avg_ms = None + display_latency_p50 = "Unknown" + display_latency_p90 = "Unknown" + + duration_vals = [e.duration_ms for e in events if e.duration_ms is not None] + events_with_duration = len(duration_vals) + if events_with_duration > 0: + duration_avg_ms = round(sum(duration_vals) / events_with_duration, 2) + display_duration_avg = f"{round(duration_avg_ms / 1000.0, 2)} s" if duration_avg_ms >= 1000 else f"{round(duration_avg_ms, 1)} ms" + else: + duration_avg_ms = None + display_duration_avg = "Unknown" + + return GroupMetrics( + name=group_name, + total_events=total_events, + events_with_tokens=events_with_tokens, + input_tokens=input_tokens, + output_tokens=output_tokens, + total_tokens=total_tokens, + events_with_cost=events_with_cost, + estimated_cost_usd=estimated_cost_usd, + events_with_latency=events_with_latency, + latency_p50_ms=latency_p50_ms, + latency_p90_ms=latency_p90_ms, + latency_p95_ms=latency_p95_ms, + latency_p99_ms=latency_p99_ms, + latency_avg_ms=latency_avg_ms, + events_with_duration=events_with_duration, + duration_avg_ms=duration_avg_ms, + display_tokens=display_tokens, + display_cost=display_cost, + display_latency_p50=display_latency_p50, + display_latency_p90=display_latency_p90, + display_duration_avg=display_duration_avg, + ) + + +def record_usage( + *, + db_path: str | None = None, + session_id: str | None = None, + remote: str = "dadeschools", + org: str = "", + repo: str = "", + project_id: str | None = None, + role: str = "unknown", + model: str = "unknown", + issue_number: int | None = None, + pr_number: int | None = None, + stage: str = "unknown", + input_tokens: int | None = None, + output_tokens: int | None = None, + total_tokens: int | None = None, + estimated_cost_usd: float | None = None, + latency_ms: int | None = None, + duration_ms: int | None = None, + status: str = "success", + metadata: str | dict[str, Any] | None = None, + created_at: str | None = None, +) -> int: + """Ingest/record a single usage event with optional metrics.""" + db = control_plane_db.ControlPlaneDB(db_path=db_path) + return db.record_usage_event( + session_id=session_id, + remote=remote, + org=org, + repo=repo, + project_id=project_id, + role=role, + model=model, + issue_number=issue_number, + pr_number=pr_number, + stage=stage, + input_tokens=input_tokens, + output_tokens=output_tokens, + total_tokens=total_tokens, + estimated_cost_usd=estimated_cost_usd, + latency_ms=latency_ms, + duration_ms=duration_ms, + status=status, + metadata=metadata, + created_at=created_at, + ) + + +def load_analytics( + *, + db_path: str | None = None, + remote: str | None = None, + org: str | None = None, + repo: str | None = None, + project_id: str | None = None, + role: str | None = None, + model: str | None = None, + stage: str | None = None, + issue_number: int | None = None, + pr_number: int | None = None, + limit: int = 500, +) -> AnalyticsSnapshot: + """Load analytics snapshot aggregated by project, role, model, issue/PR, and stage.""" + remote_filter = (remote or "").strip() or None + org_filter = (org or "").strip() or None + repo_filter = (repo or "").strip() or None + role_filter = (role or "").strip() or None + model_filter = (model or "").strip() or None + stage_filter = (stage or "").strip() or None + + # F4: coerce optional scope filters to str so AnalyticsSnapshot never holds None. + scope_remote = (remote or "").strip() + scope_org = (org or "").strip() + scope_repo = (repo or "").strip() + + try: + db = control_plane_db.ControlPlaneDB(db_path=db_path) + rows = db.query_usage_events( + remote=remote_filter, + org=org_filter, + repo=repo_filter, + project_id=project_id, + role=role_filter, + model=model_filter, + stage=stage_filter, + issue_number=issue_number, + pr_number=pr_number, + limit=limit, + ) + except Exception as exc: + empty_summary = aggregate_events("Overall", []) + return AnalyticsSnapshot( + ok=False, + reason=f"control_plane_db_unavailable: {exc}", + schema_version=ANALYTICS_SCHEMA_VERSION, + remote=scope_remote, + org=scope_org, + repo=scope_repo, + total_events=0, + overall_summary=empty_summary, + by_project={}, + by_role={}, + by_model={}, + by_work_item={}, + by_stage={}, + events=(), + ) + + parsed_events: list[UsageEvent] = [] + for r in rows: + meta = console_redaction.redact_text(r.get("metadata")) if r.get("metadata") else None + parsed_events.append( + UsageEvent( + usage_id=r["usage_id"], + session_id=r.get("session_id"), + remote=r.get("remote") or scope_remote, + org=r.get("org") or scope_org, + repo=r.get("repo") or scope_repo, + project_id=r.get("project_id"), + role=r.get("role") or "unknown", + model=r.get("model") or "unknown", + issue_number=r.get("issue_number"), + pr_number=r.get("pr_number"), + stage=r.get("stage") or "unknown", + input_tokens=r.get("input_tokens"), + output_tokens=r.get("output_tokens"), + total_tokens=r.get("total_tokens"), + estimated_cost_usd=r.get("estimated_cost_usd"), + latency_ms=r.get("latency_ms"), + duration_ms=r.get("duration_ms"), + status=r.get("status") or "success", + metadata=meta, + created_at=r.get("created_at") or "", + ) + ) + + overall_summary = aggregate_events("Overall", parsed_events) + + # Group by project + groups_by_project: dict[str, list[UsageEvent]] = {} + for e in parsed_events: + key = e.project_id or (f"{e.org}/{e.repo}" if e.org and e.repo else "default") + groups_by_project.setdefault(key, []).append(e) + by_project = {k: aggregate_events(k, v) for k, v in groups_by_project.items()} + + # Group by role + groups_by_role: dict[str, list[UsageEvent]] = {} + for e in parsed_events: + groups_by_role.setdefault(e.role, []).append(e) + by_role = {k: aggregate_events(k, v) for k, v in groups_by_role.items()} + + # Group by model + groups_by_model: dict[str, list[UsageEvent]] = {} + for e in parsed_events: + groups_by_model.setdefault(e.model, []).append(e) + by_model = {k: aggregate_events(k, v) for k, v in groups_by_model.items()} + + # Group by work item + groups_by_work_item: dict[str, list[UsageEvent]] = {} + for e in parsed_events: + if e.issue_number: + key = f"issue #{e.issue_number}" + elif e.pr_number: + key = f"pr #{e.pr_number}" + else: + key = "unlinked" + groups_by_work_item.setdefault(key, []).append(e) + by_work_item = {k: aggregate_events(k, v) for k, v in groups_by_work_item.items()} + + # Group by stage + groups_by_stage: dict[str, list[UsageEvent]] = {} + for e in parsed_events: + groups_by_stage.setdefault(e.stage, []).append(e) + by_stage = {k: aggregate_events(k, v) for k, v in groups_by_stage.items()} + + return AnalyticsSnapshot( + ok=True, + reason="ok", + schema_version=ANALYTICS_SCHEMA_VERSION, + remote=scope_remote, + org=scope_org, + repo=scope_repo, + total_events=len(parsed_events), + overall_summary=overall_summary, + by_project=by_project, + by_role=by_role, + by_model=by_model, + by_work_item=by_work_item, + by_stage=by_stage, + events=tuple(parsed_events), + ) + + +def snapshot_to_dict(snapshot: AnalyticsSnapshot) -> dict[str, Any]: + return snapshot.to_dict() diff --git a/webui/analytics_views.py b/webui/analytics_views.py new file mode 100644 index 0000000..506aa96 --- /dev/null +++ b/webui/analytics_views.py @@ -0,0 +1,248 @@ +"""HTML views for the Model Usage & Performance Analytics console (#651).""" + +from __future__ import annotations + +import html + +from webui.analytics_loader import AnalyticsSnapshot, GroupMetrics, UsageEvent +from webui.layout import render_page + + +def _escape(text: object) -> str: + """HTML-escape dynamic analytics fields (mirrors audit_views / project_views).""" + return html.escape(str(text), quote=True) + + +def _render_badge(text: str, badge_type: str = "muted") -> str: + return f'{_escape(text)}' + + +def _render_group_table(title: str, groups: dict[str, GroupMetrics], key_header: str = "Group") -> str: + if not groups: + return ( + f"

{_escape(title)}

" + '

No telemetry events recorded for this dimension.

' + ) + + rows = [] + for key, g in sorted(groups.items(), key=lambda x: x[1].total_events, reverse=True): + cost_cell = ( + f'{_escape(g.display_cost)}' + if g.events_with_cost > 0 + else _render_badge("Unknown") + ) + tokens_cell = ( + _escape(g.display_tokens) + if g.events_with_tokens > 0 + else _render_badge("Unknown") + ) + lat_p50 = ( + _escape(g.display_latency_p50) + if g.events_with_latency > 0 + else _render_badge("Unknown") + ) + lat_p90 = ( + _escape(g.display_latency_p90) + if g.events_with_latency > 0 + else _render_badge("Unknown") + ) + dur_avg = ( + _escape(g.display_duration_avg) + if g.events_with_duration > 0 + else _render_badge("Unknown") + ) + + rows.append( + "" + f"{_escape(key)}" + f"{g.total_events}" + f"{tokens_cell}" + f"{cost_cell}" + f"{lat_p50}" + f"{lat_p90}" + f"{dur_avg}" + "" + ) + + rows_html = "".join(rows) + return f""" +

{_escape(title)}

+
+ + + + + + + + + + + + + + {rows_html} + +
{_escape(key_header)}EventsTotal TokensEst. CostLatency (p50)Latency (p90)Avg Stage Duration
+
+ """ + + +def _render_events_table(events: tuple[UsageEvent, ...]) -> str: + if not events: + return ( + "

Recent Usage & Instrumentation Events

" + '

No individual telemetry events recorded yet. Opt-in instrumentation via session logging or authorized POST /api/v1/analytics/usage.

' + ) + + rows = [] + for e in list(events)[-50:]: # Display latest 50 + if e.issue_number is not None: + work_item = f"issue #{e.issue_number}" + elif e.pr_number is not None: + work_item = f"pr #{e.pr_number}" + else: + work_item = "unlinked" + tokens = ( + _escape(f"{e.total_tokens:,}") + if e.total_tokens is not None + else _render_badge("Unknown") + ) + cost = ( + _escape(f"${e.estimated_cost_usd:.4f}") + if e.estimated_cost_usd is not None + else _render_badge("Unknown") + ) + latency = ( + _escape(f"{e.latency_ms} ms") + if e.latency_ms is not None + else _render_badge("Unknown") + ) + duration = ( + _escape(f"{e.duration_ms} ms") + if e.duration_ms is not None + else _render_badge("Unknown") + ) + status_badge = _render_badge( + e.status, "success" if e.status == "success" else "danger" + ) + + rows.append( + "" + f"#{e.usage_id}" + f"{_escape(e.created_at)}" + f"{_escape(e.role)}" + f"{_escape(e.model)}" + f"{_escape(e.stage)}" + f"{_escape(work_item)}" + f"{tokens}" + f"{cost}" + f"{latency}" + f"{duration}" + f"{status_badge}" + "" + ) + + rows_html = "".join(rows) + return f""" +

Recent Telemetry Events

+
+ + + + + + + + + + + + + + + + + + {rows_html} + +
IDTimestampRoleModelStageWork ItemTokensCostLatencyDurationStatus
+
+ """ + + +def render_analytics_page(snapshot: AnalyticsSnapshot) -> str: + """Render the main Model Usage & Performance Analytics console page.""" + summary = snapshot.overall_summary + + kpi_tokens = ( + _escape(summary.display_tokens) + if summary.events_with_tokens > 0 + else _render_badge("Unknown") + ) + kpi_cost = ( + _escape(summary.display_cost) + if summary.events_with_cost > 0 + else _render_badge("Unknown") + ) + kpi_lat_p50 = ( + _escape(summary.display_latency_p50) + if summary.events_with_latency > 0 + else _render_badge("Unknown") + ) + kpi_dur_avg = ( + _escape(summary.display_duration_avg) + if summary.events_with_duration > 0 + else _render_badge("Unknown") + ) + + status_notice = "" + if not snapshot.ok: + status_notice = ( + f'
Degraded Data Source: ' + f'{_escape(snapshot.reason)}
' + ) + + body_html = f""" +

Model Usage & Performance Analytics (Phase 4)

+

+ Durable console analytics for model usage, token cost, latency percentiles, and workflow-stage performance correlated to issues, PRs, and worker roles. +

+ + {status_notice} + +
+ Note on telemetry fidelity: Missing data or untracked metrics are explicitly labeled as Unknown. No token costs or latency metrics are zero-fabricated. +
+ +
+
+ Total Events +

{summary.total_events}

+
+
+ Total Tokens +

{kpi_tokens}

+
+
+ Est. Token Cost +

{kpi_cost}

+
+
+ Latency (p50) +

{kpi_lat_p50}

+
+
+ Avg Stage Duration +

{kpi_dur_avg}

+
+
+ + {_render_group_table("Usage & Cost by Model", snapshot.by_model, "Model")} + {_render_group_table("Performance by Workflow Stage", snapshot.by_stage, "Stage")} + {_render_group_table("Usage & Cost by Role", snapshot.by_role, "Role")} + {_render_group_table("Work Item Analytics", snapshot.by_work_item, "Work Item")} + {_render_events_table(snapshot.events)} + """ + + return render_page(title="Model Usage & Performance Analytics", body_html=body_html) diff --git a/webui/app.py b/webui/app.py index f5147b5..f7648eb 100644 --- a/webui/app.py +++ b/webui/app.py @@ -47,6 +47,12 @@ from webui.worktree_views import render_worktrees_page from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict from webui.runtime_views import render_runtime_page from webui.timeline import load_timeline, snapshot_to_dict as timeline_snapshot_to_dict +from webui.analytics_loader import ( + load_analytics, + record_usage, + snapshot_to_dict as analytics_snapshot_to_dict, +) +from webui.analytics_views import render_analytics_page from webui.system_health import ( API_PATH as SYSTEM_HEALTH_API_PATH, load_system_health, @@ -567,6 +573,114 @@ async def api_v1_timeline(request: Request) -> JSONResponse: return JSONResponse(timeline_snapshot_to_dict(snapshot), status_code=status_code) +async def analytics(request: Request) -> HTMLResponse: + """Read-only model usage, token cost, latency, and performance analytics HTML view (#651).""" + snapshot = load_analytics( + remote=request.query_params.get("remote"), + org=request.query_params.get("org"), + repo=request.query_params.get("repo"), + role=request.query_params.get("role"), + model=request.query_params.get("model"), + stage=request.query_params.get("stage"), + issue_number=_query_int(request, "issue"), + pr_number=_query_int(request, "pr"), + limit=_query_int(request, "limit") or 200, + ) + return HTMLResponse(render_analytics_page(snapshot)) + + +async def api_v1_analytics(request: Request) -> JSONResponse: + """Read-only model usage, token cost, latency, and performance analytics API (#651).""" + snapshot = load_analytics( + remote=request.query_params.get("remote"), + org=request.query_params.get("org"), + repo=request.query_params.get("repo"), + role=request.query_params.get("role"), + model=request.query_params.get("model"), + stage=request.query_params.get("stage"), + issue_number=_query_int(request, "issue"), + pr_number=_query_int(request, "pr"), + limit=_query_int(request, "limit") or 500, + ) + status_code = 200 if snapshot.ok else 500 + return JSONResponse(analytics_snapshot_to_dict(snapshot), status_code=status_code) + + +async def api_v1_analytics_ingest(request: Request) -> JSONResponse: + """Optional session instrumentation ingestion endpoint (#651). + + Fail-closed write: every request is authorized through console_authz + (``record_analytics_usage``) before any control-plane DB mutation. Phase 1 + keeps ``execution_enabled=False`` and denies unauthenticated callers, so + this route cannot be used as an unauthenticated write or XSS injection + vector (PR #876 F2). + """ + try: + body = await request.json() + except Exception: + body = {} + if not isinstance(body, dict): + body = {} + + principal = resolve_principal(headers=dict(request.headers)) + decision = authorize( + "record_analytics_usage", principal, for_execution=True + ) + allowed = bool(decision.allowed and decision.execution_enabled) + console_audit.record_event( + action_id="record_analytics_usage", + result=( + console_audit.RESULT_ALLOWED + if allowed + else console_audit.RESULT_DENIED + ), + decision=decision, + principal=principal, + target=_audit_target("record_analytics_usage", body), + request_id=_request_id(), + detail=decision.detail, + ) + authorization = decision.to_dict() + if not allowed: + return JSONResponse( + { + "ok": False, + "error": "unauthorized", + "detail": ( + "POST /api/v1/analytics/usage requires an authenticated " + "principal with record_analytics_usage execution enabled" + ), + "authorization": authorization, + }, + status_code=403, + ) + + usage_id = record_usage( + session_id=body.get("session_id"), + remote=body.get("remote", "dadeschools"), + org=body.get("org", ""), + repo=body.get("repo", ""), + project_id=body.get("project_id"), + role=body.get("role", "unknown"), + model=body.get("model", "unknown"), + issue_number=body.get("issue_number") or body.get("issue"), + pr_number=body.get("pr_number") or body.get("pr"), + stage=body.get("stage", "unknown"), + input_tokens=body.get("input_tokens"), + output_tokens=body.get("output_tokens"), + total_tokens=body.get("total_tokens"), + estimated_cost_usd=body.get("estimated_cost_usd"), + latency_ms=body.get("latency_ms"), + duration_ms=body.get("duration_ms"), + status=body.get("status", "success"), + metadata=body.get("metadata"), + ) + return JSONResponse( + {"ok": True, "usage_id": usage_id, "authorization": authorization}, + status_code=201, + ) + + async def method_not_allowed(request: Request, _exc: Exception) -> Response: path = request.url.path if path in _AUDIT_MUTATION_PATHS and request.method == "POST": @@ -608,6 +722,10 @@ def create_app(*, bind_host: str | None = None) -> Starlette: Route("/runtime", runtime, methods=["GET"]), Route("/api/runtime", api_runtime, methods=["GET"]), Route("/api/v1/timeline", api_v1_timeline, methods=["GET"]), + Route("/analytics", analytics, methods=["GET"]), + Route("/api/analytics", api_v1_analytics, methods=["GET"]), + Route("/api/v1/analytics", api_v1_analytics, methods=["GET"]), + Route("/api/v1/analytics/usage", api_v1_analytics_ingest, methods=["POST"]), Route("/audit", audit, methods=["GET", "POST"]), Route("/api/audit", api_audit, methods=["GET", "POST"]), Route("/worktrees", worktrees, methods=["GET"]), diff --git a/webui/console_authz.py b/webui/console_authz.py index 2e64b2c..0c51037 100644 --- a/webui/console_authz.py +++ b/webui/console_authz.py @@ -236,6 +236,47 @@ _ACTION_SPECS: tuple[ConsoleAction, ...] = ( phase=3, summary="Remove a remote feature branch.", ), + # #651 analytics ingest: local control-plane write, not a Gitea mutation. + # Phase 2 gated write so Phase 1 (ACTIVE_PHASE=1) fails closed on execution. + ConsoleAction( + action_id="record_analytics_usage", + task_key="record_analytics_usage", + action_class=CLASS_WRITE, + minimum_role=OPERATOR, + requires_confirmation=True, + dual_control=False, + break_glass=False, + phase=2, + summary="Ingest a model-usage / latency analytics event into the control-plane DB.", + ), + # #642: sanctioned daemon lifecycle. These exist so operators have an + # audited path off `pkill -f mcp_server.py` (#630). Restart drops every + # in-flight request on a namespace, so it carries the same dual-control and + # break-glass weight as a merge; reload drains first and is privileged but + # not destructive. Neither ever exposes a raw kill: execution is handed to + # a host supervisor by ``webui.sanctioned_restart``. + ConsoleAction( + action_id="system.reload_namespace", + task_key="reload_namespace", + action_class=CLASS_PRIVILEGED, + minimum_role=CONTROLLER, + requires_confirmation=True, + dual_control=False, + break_glass=False, + phase=2, + summary="Gracefully reload one MCP namespace via the host supervisor.", + ), + ConsoleAction( + action_id="system.restart_namespace", + task_key="restart_namespace", + action_class=CLASS_DESTRUCTIVE, + minimum_role=ADMIN, + requires_confirmation=True, + dual_control=True, + break_glass=True, + phase=2, + summary="Restart one MCP namespace via the host supervisor.", + ), ) ACTIONS: dict[str, ConsoleAction] = {a.action_id: a for a in _ACTION_SPECS} diff --git a/webui/gated_actions.py b/webui/gated_actions.py index baefab8..d914874 100644 --- a/webui/gated_actions.py +++ b/webui/gated_actions.py @@ -110,6 +110,8 @@ def _format_target(action_id: str, params: dict[str, Any]) -> str: ) if action_id == "create_issue": return f"issue {params.get('title', '?')!r}" + if action_id in {"system.restart_namespace", "system.reload_namespace"}: + return f"MCP namespace {params.get('namespace', '?')!r}" return "unspecified" @@ -165,6 +167,17 @@ def build_action_registry() -> ActionRegistry: "gitea_create_issue_comment", "Post a PR review thread comment."), ("close_pr", "Close PR", "close_pr", "gitea_edit_pr", "Close a pull request without merge."), + # #642: the sanctioned replacement for the forbidden manual daemon-kill + # recovery path (#630). The "tool" is a host supervisor hook, not an MCP + # call — the console never signals a process. Preview and gating live in + # ``webui.sanctioned_restart``; these stay disabled like every other + # registry entry. + ("system.reload_namespace", "Reload MCP namespace", "reload_namespace", + "host.supervisor_reload", + "Gracefully reload one MCP namespace via the host supervisor."), + ("system.restart_namespace", "Restart MCP namespace", + "restart_namespace", "host.supervisor_restart", + "Restart one MCP namespace via the host supervisor."), ) actions = tuple( GatedAction( diff --git a/webui/nav.py b/webui/nav.py index c24643d..45d1212 100644 --- a/webui/nav.py +++ b/webui/nav.py @@ -65,6 +65,7 @@ NAV_GROUPS: tuple[NavGroup, ...] = ( )), NavGroup("Insights", ( NavItem("/insights", "Insights", "stub"), + NavItem("/analytics", "Analytics"), NavItem("/audit", "Audit"), )), ) diff --git a/webui/sanctioned_restart.py b/webui/sanctioned_restart.py new file mode 100644 index 0000000..f413829 --- /dev/null +++ b/webui/sanctioned_restart.py @@ -0,0 +1,613 @@ +"""Sanctioned MCP restart and graceful reload controls (#642, Phase 2). + +Sessions have historically recovered MCP connectivity by killing the host +daemon (``pkill -f mcp_server.py``, #630). That path stays forbidden: it kills +every namespace on the host, contaminates the surviving session, and leaves no +audit trail. This module is the sanctioned replacement. + +A restart is modelled as a *gated action*, never as a command: + +1. **Capability** — the console action resolves through ``console_authz`` + against ``task_capability_map``, so the console cannot invent an authority + the MCP layer does not already define. +2. **Preview** — :func:`build_restart_preview` renders a mutation ledger and + the exact confirmation phrase. It never returns a shell command. +3. **Confirmation** — the operator echoes a phrase naming the exact namespace + and mode. A phrase for one namespace never authorizes another. +4. **Operator authorization** — host daemon maintenance is authorized out of + band through the environment (#630, and #710 finding F1: a worker session + cannot set an env var for an already-running daemon, so this cannot be + self-asserted the way a tool argument could). +5. **Execution** — :func:`execute_restart` never spawns a process. Once every + gate passes it hands the request to the configured host-managed restart + hook; with no hook configured it fails closed. +6. **Health recheck** — :func:`verify_post_restart_health` requires live + client-namespace probe evidence before any post-restart clean claim. + +Manual ``pkill`` remains forbidden and is classified as contamination by +:func:`classify_restart_command`, which blocks clean claims (#630 AC3). + +This module performs no I/O beyond reading its own environment configuration, +imports no MCP client, and holds no credential. +""" + +from __future__ import annotations + +import os +from dataclasses import asdict, dataclass +from typing import Any + +import mcp_namespace_health +import runtime_recovery_guard +from webui import console_audit, console_authz + +# --- Operations ------------------------------------------------------------- + +MODE_RESTART = "restart" +MODE_RELOAD = "reload" +MODES: tuple[str, ...] = (MODE_RESTART, MODE_RELOAD) + +ACTION_RESTART_NAMESPACE = "system.restart_namespace" +ACTION_RELOAD_NAMESPACE = "system.reload_namespace" + +ACTION_FOR_MODE: dict[str, str] = { + MODE_RESTART: ACTION_RESTART_NAMESPACE, + MODE_RELOAD: ACTION_RELOAD_NAMESPACE, +} + +# Namespaces the console may target. An unlisted name fails closed rather than +# being passed through to a host hook. +KNOWN_NAMESPACES: tuple[str, ...] = tuple( + sorted( + set(mcp_namespace_health.DEFAULT_NAMESPACES) + | {"gitea-author", "gitea-reviewer", "gitea-merger", + "gitea-reconciler", "gitea-controller"} + ) +) + +# Scope tokens that would mean "everything at once". Explicit non-goal: the +# console never offers a fleet-wide restart, because that is the blast radius +# `pkill -f mcp_server.py` already had. +_FLEET_TOKENS = frozenset({"*", "all", "fleet", "any", ""}) + +# --- Environment configuration ---------------------------------------------- +# Read server-side only; the value is an opaque host hook reference (e.g. a +# launchd label), never a command line, and is never rendered to a client. +RESTART_HOOK_ENV = "GITEA_SANCTIONED_RESTART_HOOK" + +# --- Reason codes ----------------------------------------------------------- + +DENY_UNKNOWN_MODE = "unknown_mode" +DENY_UNKNOWN_NAMESPACE = "unknown_namespace" +DENY_FLEET_SCOPE = "fleet_scope_not_permitted" +DENY_UNAUTHORIZED = "unauthorized" +DENY_CONFIRMATION_MISSING = "confirmation_required" +DENY_CONFIRMATION_MISMATCH = "confirmation_mismatch" +DENY_OPERATOR_AUTHORIZATION = "operator_authorization_missing" +DENY_HOOK_NOT_CONFIGURED = "restart_hook_not_configured" +DENY_CONTAMINATED_RUNTIME = "contaminated_runtime" + +ALLOW_HOST_ACTION_REQUIRED = "host_action_required" + +# Post-restart verification outcomes. +HEALTH_CLEAN = "clean" +HEALTH_UNPROVEN = "unproven" +HEALTH_UNHEALTHY = "unhealthy" + + +def _clean(value: Any) -> str: + return str(value or "").strip() + + +# --- Mutation ledger -------------------------------------------------------- + + +@dataclass(frozen=True) +class RestartLedgerEntry: + """One planned step, shown before anything is asked of the host.""" + + sequence: int + step: str + summary: str + executes_process_kill: bool = False + + +def _mutation_ledger(namespace: str, mode: str) -> tuple[RestartLedgerEntry, ...]: + if mode == MODE_RELOAD: + middle = RestartLedgerEntry( + sequence=2, + step="host_graceful_reload", + summary=( + f"Ask the configured host supervisor to reload {namespace} " + "in place, draining in-flight requests. The console does not " + "signal the process itself." + ), + ) + else: + middle = RestartLedgerEntry( + sequence=2, + step="host_restart_hook", + summary=( + f"Ask the configured host supervisor to restart {namespace}. " + "The console never sends a signal and never runs a kill." + ), + ) + return ( + RestartLedgerEntry( + sequence=1, + step="quiesce", + summary=( + f"Stop admitting new gated mutations for {namespace} and " + "record the intent before anything restarts." + ), + ), + middle, + RestartLedgerEntry( + sequence=3, + step="health_recheck", + summary=( + f"Re-probe {namespace} through the live client namespace and " + "prove the required tool is callable again." + ), + ), + RestartLedgerEntry( + sequence=4, + step="audit", + summary=( + "Append actor, target namespace, mode, and result to the " + "console audit log." + ), + ), + ) + + +# --- Confirmation ----------------------------------------------------------- + + +def confirmation_phrase(namespace: str, mode: str) -> str: + """Exact phrase an operator must echo, naming the namespace and mode. + + Binding the namespace into the phrase is the point: a confirmation typed + for ``gitea-author`` cannot be replayed against ``gitea-merger``. + """ + return f"{_clean(mode)} {_clean(namespace)}" + + +def confirmation_matches( + namespace: str, mode: str, confirmation: str | None +) -> bool: + """Compare *confirmation* to the required phrase (exact, whitespace-trimmed).""" + return _clean(confirmation) == confirmation_phrase(namespace, mode) + + +# --- Scope validation ------------------------------------------------------- + + +def _validate_scope(namespace: str, mode: str) -> tuple[str, str] | None: + """Return ``(reason_code, detail)`` when the scope is refused.""" + ns = _clean(namespace) + md = _clean(mode) + + if md not in MODES: + return ( + DENY_UNKNOWN_MODE, + f"Mode {md!r} is not one of {', '.join(MODES)}.", + ) + if ns.lower() in _FLEET_TOKENS: + return ( + DENY_FLEET_SCOPE, + ( + "Fleet-wide restart is an explicit non-goal: it reproduces the " + "blast radius of `pkill -f mcp_server.py` (#630). Restart one " + "namespace at a time." + ), + ) + if ns not in KNOWN_NAMESPACES: + return ( + DENY_UNKNOWN_NAMESPACE, + f"Namespace {ns!r} is not a known MCP namespace.", + ) + return None + + +# --- Host hook -------------------------------------------------------------- + + +def restart_hook(env: dict[str, str] | None = None) -> dict[str, Any]: + """Report the configured host-managed restart hook. + + The hook is a reference the *host* resolves (a supervisor label), not a + command this process runs. ``configured=False`` fails restart closed. + """ + source = env if env is not None else os.environ + reference = _clean(source.get(RESTART_HOOK_ENV)) + return { + "configured": bool(reference), + "reference": reference or None, + "source": RESTART_HOOK_ENV if reference else None, + "self_assertable": False, + "console_executes_process": False, + } + + +# --- Preview ---------------------------------------------------------------- + + +def build_restart_preview( + namespace: str, + mode: str = MODE_RESTART, + *, + principal: console_authz.Principal | None = None, + env: dict[str, str] | None = None, +) -> dict[str, Any]: + """Render the dry-run preview for a restart/reload request. + + Read-only: no authorization is granted, no host is contacted, and the + result never contains a shell command. + """ + ns = _clean(namespace) + md = _clean(mode) + action_id = ACTION_FOR_MODE.get(md, ACTION_RESTART_NAMESPACE) + action = console_authz.get_action(action_id) + decision = console_authz.authorize(action_id, principal) + scope_error = _validate_scope(ns, md) + hook = restart_hook(env) + operator = runtime_recovery_guard.operator_authorization(env) + + return { + "action_id": action_id, + "namespace": ns, + "mode": md, + "scope_valid": scope_error is None, + "scope_reason_code": scope_error[0] if scope_error else None, + "scope_detail": scope_error[1] if scope_error else None, + "required_role": action.minimum_role if action else None, + "required_permission": action.mcp_permission if action else None, + "action_class": action.action_class if action else None, + "dual_control": action.dual_control if action else True, + "break_glass": action.break_glass if action else True, + "requires_confirmation": True, + "confirmation_phrase": confirmation_phrase(ns, md), + "mutation_ledger": [asdict(entry) for entry in _mutation_ledger(ns, md)], + "authorization": decision.to_dict(), + "operator_authorization": operator, + "restart_hook": hook, + "execution_enabled": False, + "raw_process_kill_exposed": False, + "known_namespaces": list(KNOWN_NAMESPACES), + "post_restart_verification_required": True, + } + + +# --- Gate ------------------------------------------------------------------- + + +def assess_restart_request( + namespace: str, + mode: str = MODE_RESTART, + *, + principal: console_authz.Principal | None = None, + confirmation: str | None = None, + contamination_marker: dict[str, Any] | None = None, + env: dict[str, str] | None = None, +) -> dict[str, Any]: + """Decide whether a restart request may proceed to the host hook. + + Every gate must pass. The first failure wins and is reported with a stable + reason code; a pass never means "restarted", only "may be handed to the + configured host hook". + """ + ns = _clean(namespace) + md = _clean(mode) + action_id = ACTION_FOR_MODE.get(md, ACTION_RESTART_NAMESPACE) + preview = build_restart_preview(ns, md, principal=principal, env=env) + + def refuse(reason_code: str, detail: str) -> dict[str, Any]: + return { + "allowed": False, + "gates_passed": False, + "reason_code": reason_code, + "detail": detail, + "action_id": action_id, + "namespace": ns, + "mode": md, + "preview": preview, + "execution_enabled": False, + } + + scope_error = _validate_scope(ns, md) + if scope_error is not None: + return refuse(*scope_error) + + # Authority is checked as an authorization decision, not an execution + # grant. ``for_execution=True`` asks "may the console perform this write?", + # and the answer here is permanently no: step 2 of the ledger is a request + # to the host supervisor, so the console's Phase 2 execution gate is not + # the relevant gate. Every branch below keeps ``execution_enabled`` False + # and :func:`execute_restart` never touches a process. + decision = console_authz.authorize(action_id, principal) + if not decision.allowed: + return refuse(DENY_UNAUTHORIZED, decision.detail) + + if not _clean(confirmation): + return refuse( + DENY_CONFIRMATION_MISSING, + ( + "Type the confirmation phrase " + f"{preview['confirmation_phrase']!r} to proceed." + ), + ) + if not confirmation_matches(ns, md, confirmation): + return refuse( + DENY_CONFIRMATION_MISMATCH, + ( + "Confirmation does not name this namespace and mode; expected " + f"{preview['confirmation_phrase']!r}." + ), + ) + + operator = preview["operator_authorization"] + if not operator["authorized"]: + return refuse( + DENY_OPERATOR_AUTHORIZATION, + ( + "Host daemon maintenance requires out-of-band operator " + "authorization via " + f"{runtime_recovery_guard.OPERATOR_AUTHORIZATION_ENV}." + ), + ) + + # #630's task-scoped gate deliberately lets a contaminated worker keep + # commenting and handing off. Restart is stricter and unconditional: a + # runtime already contaminated by a manual kill must be reconciled before + # it is restarted again, or the restart just launders the contamination. + if contamination_marker and not contamination_marker.get( + "cleared_by_reconciler" + ): + return refuse( + DENY_CONTAMINATED_RUNTIME, + ( + "A live contamination marker is present; clear it through the " + "reconciler path before restarting." + ), + ) + + hook = preview["restart_hook"] + if not hook["configured"]: + return refuse( + DENY_HOOK_NOT_CONFIGURED, + ( + "No host-managed restart hook is configured " + f"({RESTART_HOOK_ENV}). The console will not fall back to a " + "process kill." + ), + ) + + return { + "allowed": True, + "gates_passed": True, + "reason_code": ALLOW_HOST_ACTION_REQUIRED, + "detail": ( + "Every gate passed. The restart must be performed by the " + "configured host supervisor; the console does not signal the " + "process." + ), + "action_id": action_id, + "namespace": ns, + "mode": md, + "preview": preview, + "execution_enabled": False, + "console_executes": False, + "console_active_phase": console_authz.ACTIVE_PHASE, + } + + +# --- Execution -------------------------------------------------------------- + + +def execute_restart( + namespace: str, + mode: str = MODE_RESTART, + *, + principal: console_authz.Principal | None = None, + confirmation: str | None = None, + contamination_marker: dict[str, Any] | None = None, + env: dict[str, str] | None = None, + request_id: str | None = None, + session_id: str | None = None, +) -> dict[str, Any]: + """Run every gate, audit the outcome, and hand off to the host. + + This function never spawns a process, never sends a signal, and never + builds a command line. ``success`` is False in both directions: a refused + request is refused, and an authorized request still requires the host + supervisor to act. + """ + assessment = assess_restart_request( + namespace, + mode, + principal=principal, + confirmation=confirmation, + contamination_marker=contamination_marker, + env=env, + ) + action_id = assessment["action_id"] + allowed = assessment["allowed"] + + audit = console_audit.record_event( + action_id=action_id, + result=( + console_audit.RESULT_ALLOWED if allowed + else console_audit.RESULT_DENIED + ), + principal=principal, + target={"namespace": assessment["namespace"], "mode": assessment["mode"]}, + reason_code=assessment["reason_code"], + detail=assessment["detail"], + request_id=request_id, + session_id=session_id, + metadata={ + "gates_passed": assessment["gates_passed"], + "process_kill_executed": False, + "post_restart_verification_required": True, + }, + ) + + return { + "success": False, + "outcome": ( + ALLOW_HOST_ACTION_REQUIRED if allowed else assessment["reason_code"] + ), + "allowed": allowed, + "detail": assessment["detail"], + "namespace": assessment["namespace"], + "mode": assessment["mode"], + "action_id": action_id, + "process_kill_executed": False, + "host_hook": assessment["preview"]["restart_hook"], + "next_action": ( + "Have the host supervisor perform the restart, then call " + "verify_post_restart_health with live client-namespace evidence " + "before claiming a clean session." + if allowed + else assessment["detail"] + ), + "assessment": assessment, + "audit": audit, + } + + +# --- Contamination classification ------------------------------------------- + + +def classify_restart_command( + command: str | None, + *, + mcp_pids: list[Any] | tuple[Any, ...] | None = None, + session_id: str | None = None, + remote: str | None = None, + role: str | None = None, +) -> dict[str, Any]: + """Classify an operator-proposed recovery command (#630 AC2). + + A manual ``pkill``/``kill`` of the MCP daemon is contamination, not a + restart. When contaminating, a durable marker is returned so downstream + gated mutations and clean claims fail closed. + """ + classification = runtime_recovery_guard.classify_recovery_command( + command, mcp_pids=mcp_pids + ) + contaminating = bool(classification.get("contamination")) + + marker = None + if contaminating: + marker = runtime_recovery_guard.build_contamination_record( + reason_class=( + classification.get("reason_class") + or runtime_recovery_guard.REASON_MANUAL_DAEMON_KILL + ), + command_redacted=classification.get("redacted_command"), + session_id=session_id, + remote=remote, + role=role, + detail=( + "Manual daemon kill is forbidden; use the sanctioned " + f"{ACTION_RESTART_NAMESPACE} gated action instead." + ), + ) + + return { + "contamination": contaminating, + "sanctioned": not contaminating and not classification.get("process_kill"), + "clean_claim_allowed": not contaminating, + "reason_class": classification.get("reason_class"), + "redacted_command": classification.get("redacted_command"), + "classification": classification, + "contamination_marker": marker, + "sanctioned_alternative": ACTION_RESTART_NAMESPACE, + } + + +# --- Post-restart health verification --------------------------------------- + + +def verify_post_restart_health( + namespace: str, + *, + probe_result: dict[str, Any] | None = None, + probe_source: str | None = None, + registered_tools: list[str] | tuple[str, ...] | None = None, + required_tool: str | None = None, + profile: str | None = None, +) -> dict[str, Any]: + """Require live proof a namespace is callable before any clean claim (AC3). + + Static registration is not proof and neither is an offline subprocess + probe: only ``probe_source=client_namespace`` evidence can clear a + post-restart session for mutations. + """ + ns = _clean(namespace) + health = mcp_namespace_health.classify_namespace_probe( + ns, + required_tool=required_tool, + registered_tools=registered_tools, + probe_result=probe_result, + profile=profile, + probe_source=probe_source, + ) + healthy = bool(health.get("healthy")) + proven = bool(health.get("ide_namespace_proven")) + + if healthy and proven: + status = HEALTH_CLEAN + elif healthy: + status = HEALTH_UNPROVEN + else: + status = HEALTH_UNHEALTHY + + reasons = list(health.get("reasons") or []) + if status == HEALTH_UNPROVEN: + reasons.append( + "Namespace reported healthy without live client-namespace " + "evidence; a post-restart clean claim requires " + f"probe_source={mcp_namespace_health.PROBE_SOURCE_CLIENT!r}." + ) + + return { + "namespace": ns, + "status": status, + "healthy": healthy, + "ide_namespace_proven": proven, + "clean_claim_allowed": status == HEALTH_CLEAN, + "mutations_allowed": status == HEALTH_CLEAN, + "reasons": reasons, + "health": health, + } + + +# --- Policy surface --------------------------------------------------------- + + +def restart_policy() -> dict[str, Any]: + """Machine-readable description of the sanctioned restart contract.""" + return { + "policy_version": 1, + "modes": list(MODES), + "actions": [ACTION_RESTART_NAMESPACE, ACTION_RELOAD_NAMESPACE], + "known_namespaces": list(KNOWN_NAMESPACES), + "fleet_scope_permitted": False, + "console_executes_process_kill": False, + "raw_kill_exposed": False, + "requires_confirmation": True, + "confirmation_binds_namespace": True, + "operator_authorization_env": ( + runtime_recovery_guard.OPERATOR_AUTHORIZATION_ENV + ), + "restart_hook_env": RESTART_HOOK_ENV, + "manual_kill_classified_as": runtime_recovery_guard.CONTAMINATION_KIND, + "post_restart_clean_claim_requires": ( + mcp_namespace_health.PROBE_SOURCE_CLIENT + ), + "audit_required": True, + "silent_auto_restart_permitted": False, + }