"""Compose runtime health + inventory into a sessions/runtime view (#641). Phase 1 is read-only. It correlates namespaces, sessions, capabilities, worktree bindings, lease ownership, stale flags, and contamination markers when they are detectable on disk (#630 / #671). It never restarts, kills, or takes over a session. Sources: * :mod:`webui.runtime_health` — profile, role, stale runtime, shell health. * :mod:`webui.inventory` — sessions, leases, locks, worktrees, namespaces and collision signals from the control-plane DB + filesystem. * :mod:`mcp_session_state` — durable contamination markers (inspect only). Secrets are never read. Absolute paths are collapsed by inventory redaction. """ from __future__ import annotations from dataclasses import dataclass, field from typing import Any, Callable import mcp_session_state from webui.inventory import ( InventorySnapshot, load_inventory_snapshot, snapshot_to_dict as inventory_snapshot_to_dict, ) from webui.runtime_health import ( RuntimeSnapshot, load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict, ) # Sanctioned recovery pointers only — never pkill / killall (#630). SANCTIONED_RECOVERY_DOCS: tuple[dict[str, str], ...] = ( { "label": "MCP namespace EOF recovery (reconnect only)", "path": "docs/mcp-namespace-eof-recovery.md", "note": "IDE/client reconnect or operator-owned restart; never kill daemons.", }, { "label": "MCP namespace health", "path": "docs/mcp-namespace-health.md", "note": "client_namespace probe proves namespace health.", }, { "label": "Restart path inventory", "path": "docs/mcp-restart-path-inventory.md", "note": "Catalog of sanctioned reconnect/restart paths.", }, { "label": "Local web UI recovery", "path": "docs/webui-local-dev.md", "note": "Operator console start and documented recovery sequence.", }, ) _CONTAMINATION_KINDS: tuple[str, ...] = ( mcp_session_state.KIND_RUNTIME_RECOVERY_CONTAMINATION, mcp_session_state.KIND_STABLE_BRANCH_CONTAMINATION, ) @dataclass(frozen=True) class ContaminationMarker: """A detectable durable contamination marker (audit-safe summary).""" kind: str on_disk: bool has_payload: bool summary: str reason_class: str | None = None session_id: str | None = None role: str | None = None command_summary: str | None = None cleared: bool = False def to_dict(self) -> dict[str, Any]: return { "kind": self.kind, "on_disk": self.on_disk, "has_payload": self.has_payload, "summary": self.summary, "reason_class": self.reason_class, "session_id": self.session_id, "role": self.role, "command_summary": self.command_summary, "cleared": self.cleared, "active": self.on_disk and self.has_payload and not self.cleared, } @dataclass(frozen=True) class SessionRow: """One correlated session row for the sessions table.""" session_id: str role: str | None profile: str | None namespace: str | None pid: int | None pid_alive: bool | None status: str | None started_at: str | None last_heartbeat_at: str | None lease_ids: tuple[str, ...] = () work_refs: tuple[str, ...] = () worktree_paths: tuple[str, ...] = () stale_flags: tuple[str, ...] = () contamination_flags: tuple[str, ...] = () def to_dict(self) -> dict[str, Any]: return { "session_id": self.session_id, "role": self.role, "profile": self.profile, "namespace": self.namespace, "pid": self.pid, "pid_alive": self.pid_alive, "status": self.status, "started_at": self.started_at, "last_heartbeat_at": self.last_heartbeat_at, "lease_ids": list(self.lease_ids), "work_refs": list(self.work_refs), "worktree_paths": list(self.worktree_paths), "stale_flags": list(self.stale_flags), "contamination_flags": list(self.contamination_flags), "is_stale": bool(self.stale_flags), "is_contaminated": bool(self.contamination_flags), } @dataclass(frozen=True) class SessionViewSnapshot: """Composed runtime + session inventory view (#641).""" runtime: RuntimeSnapshot inventory: InventorySnapshot sessions: tuple[SessionRow, ...] contamination_markers: tuple[ContaminationMarker, ...] recovery_docs: tuple[dict[str, str], ...] = SANCTIONED_RECOVERY_DOCS fetch_error: str | None = None @property def stale_session_count(self) -> int: return sum(1 for row in self.sessions if row.stale_flags) @property def contaminated_session_count(self) -> int: return sum(1 for row in self.sessions if row.contamination_flags) @property def active_contamination(self) -> tuple[ContaminationMarker, ...]: return tuple(m for m in self.contamination_markers if m.to_dict()["active"]) def _inspect_contamination( kind: str, *, remote: str | None, inspect: Callable[..., dict[str, Any]] | None = None, load: Callable[..., dict[str, Any] | None] | None = None, ) -> ContaminationMarker: """Inspect one contamination kind; never raises into the page render path.""" inspect_fn = inspect or mcp_session_state.inspect_state_envelope load_fn = load or mcp_session_state.load_state try: envelope = inspect_fn(kind=kind, remote=remote) except Exception as exc: # noqa: BLE001 — fail soft for the dashboard return ContaminationMarker( kind=kind, on_disk=False, has_payload=False, summary=f"contamination inspect failed: {type(exc).__name__}", ) reason_class = None session_id = None role = None command_summary = None cleared = False summary = str(envelope.get("summary") or "") if envelope.get("on_disk") and envelope.get("has_payload"): try: payload = load_fn(kind=kind, remote=remote) or {} except Exception: # noqa: BLE001 payload = {} if isinstance(payload, dict): reason_class = payload.get("reason_class") session_id = payload.get("session_id") role = payload.get("role") command_summary = payload.get("command_summary") or payload.get("detail") cleared = bool(payload.get("cleared_by_reconciler")) if not summary: summary = ( f"{kind}: {reason_class or 'present'}" + (" (cleared)" if cleared else "") ) return ContaminationMarker( kind=kind, on_disk=bool(envelope.get("on_disk")), has_payload=bool(envelope.get("has_payload")), summary=summary or f"{kind}: not present", reason_class=str(reason_class) if reason_class else None, session_id=str(session_id) if session_id else None, role=str(role) if role else None, command_summary=str(command_summary) if command_summary else None, cleared=cleared, ) def _load_contamination_markers( *, remote: str | None, inspect: Callable[..., dict[str, Any]] | None = None, load: Callable[..., dict[str, Any] | None] | None = None, ) -> tuple[ContaminationMarker, ...]: return tuple( _inspect_contamination(kind, remote=remote, inspect=inspect, load=load) for kind in _CONTAMINATION_KINDS ) def _build_session_rows( inventory: InventorySnapshot, contamination: tuple[ContaminationMarker, ...], ) -> tuple[SessionRow, ...]: sessions_section = inventory.section("sessions") leases_section = inventory.section("leases") locks_section = inventory.section("locks") leases_by_session: dict[str, list[dict[str, Any]]] = {} for lease in (leases_section.items if leases_section else ()): sid = str(lease.get("session_id") or "") if sid: leases_by_session.setdefault(sid, []).append(lease) locks_by_session_hint: dict[str, list[dict[str, Any]]] = {} for lock in (locks_section.items if locks_section else ()): # Locks carry claimant profile/username, not control-plane session ids. # Correlate later by matching lease work_number ↔ lock issue_number. pass active_markers = [m for m in contamination if m.to_dict()["active"]] marker_session_ids = { m.session_id for m in active_markers if m.session_id } rows: list[SessionRow] = [] for raw in sessions_section.items if sessions_section else (): sid = str(raw.get("session_id") or "") if not sid: continue session_leases = leases_by_session.get(sid, []) lease_ids = tuple( str(lease["lease_id"]) for lease in session_leases if lease.get("lease_id") ) work_refs: list[str] = [] work_numbers: list[int] = [] for lease in session_leases: kind = lease.get("work_kind") number = lease.get("work_number") if kind and number is not None: work_refs.append(f"{kind}#{number}") try: work_numbers.append(int(number)) except (TypeError, ValueError): pass worktree_paths: list[str] = [] for lock in (locks_section.items if locks_section else ()): try: issue_no = int(lock.get("issue_number")) except (TypeError, ValueError): continue if issue_no in work_numbers and lock.get("worktree_path"): worktree_paths.append(str(lock["worktree_path"])) stale_flags: list[str] = [] if raw.get("pid_alive") is False: stale_flags.append("pid-dead") status = str(raw.get("status") or "").lower() if status and status not in {"active", "alive", "running", "ok"}: stale_flags.append(f"status:{status}") for lease in session_leases: if lease.get("expired") is True: stale_flags.append("lease-expired") if str(lease.get("status") or "").lower() == "active" and lease.get( "expired" ) is True: stale_flags.append("active-lease-past-expiry") contamination_flags: list[str] = [] if sid in marker_session_ids: for marker in active_markers: if marker.session_id == sid: contamination_flags.append(marker.kind) # Process-wide contamination with no session binding still surfaces # against every live session so it cannot be silent (#630). for marker in active_markers: if not marker.session_id and marker.kind not in contamination_flags: contamination_flags.append(f"{marker.kind}:process-wide") rows.append( SessionRow( session_id=sid, role=raw.get("role"), profile=raw.get("profile"), namespace=raw.get("namespace"), pid=raw.get("pid") if isinstance(raw.get("pid"), int) else None, pid_alive=raw.get("pid_alive") if isinstance(raw.get("pid_alive"), bool) else None, status=raw.get("status"), started_at=raw.get("started_at"), last_heartbeat_at=raw.get("last_heartbeat_at"), lease_ids=lease_ids, work_refs=tuple(work_refs), worktree_paths=tuple(worktree_paths), stale_flags=tuple(dict.fromkeys(stale_flags)), contamination_flags=tuple(dict.fromkeys(contamination_flags)), ) ) # Silence unused variable for the locks_by_session_hint placeholder path. _ = locks_by_session_hint return tuple(rows) def load_session_view_snapshot( *, load_runtime: Callable[..., RuntimeSnapshot] | None = None, load_inventory: Callable[..., InventorySnapshot] | None = None, inspect_contamination: Callable[..., dict[str, Any]] | None = None, load_contamination_payload: Callable[..., dict[str, Any] | None] | None = None, ) -> SessionViewSnapshot: """Build the composed sessions/runtime view. Fail-soft on partial sources.""" runtime_loader = load_runtime or load_runtime_snapshot inventory_loader = load_inventory or load_inventory_snapshot fetch_error: str | None = None try: runtime = runtime_loader() except Exception as exc: # noqa: BLE001 fetch_error = f"runtime snapshot failed: {type(exc).__name__}: {exc}" # Minimal placeholder so the page still renders inventory + recovery. from webui.runtime_health import RuntimeSnapshot as _RS runtime = _RS( project_id="unknown", repo_root="", remote="", host="", profile_name="unknown", role_kind="unknown", config_model="unknown", profile_mode="unknown", profile_source="unknown", authenticated_username=None, identity_error=str(exc), repo_sha=None, remote_master_sha=None, commits_behind_master=None, stale_runtime_warning=None, shell_health={}, workflow_hashes=(), schema_hashes=(), restart_guidance="docs/mcp-namespace-eof-recovery.md", fetch_error=str(exc), ) try: inventory = inventory_loader() except Exception as exc: # noqa: BLE001 msg = f"inventory snapshot failed: {type(exc).__name__}: {exc}" fetch_error = f"{fetch_error}; {msg}" if fetch_error else msg inventory = load_inventory_snapshot( db_path="/nonexistent-for-fail-soft", lock_dir="/nonexistent-for-fail-soft", ) remote = getattr(runtime, "remote", None) contamination = _load_contamination_markers( remote=remote, inspect=inspect_contamination, load=load_contamination_payload, ) sessions = _build_session_rows(inventory, contamination) return SessionViewSnapshot( runtime=runtime, inventory=inventory, sessions=sessions, contamination_markers=contamination, fetch_error=fetch_error, ) def snapshot_to_dict(snapshot: SessionViewSnapshot) -> dict[str, Any]: """JSON export for ``/api/sessions`` (read-only).""" return { "api_version": "v1", "view": "runtime-sessions", "issue": 641, "fetch_error": snapshot.fetch_error, "runtime": runtime_snapshot_to_dict(snapshot.runtime), "inventory": inventory_snapshot_to_dict(snapshot.inventory), "sessions": [row.to_dict() for row in snapshot.sessions], "session_counts": { "total": len(snapshot.sessions), "stale": snapshot.stale_session_count, "contaminated": snapshot.contaminated_session_count, }, "contamination_markers": [ marker.to_dict() for marker in snapshot.contamination_markers ], "active_contamination": [ marker.to_dict() for marker in snapshot.active_contamination ], "recovery_docs": [dict(doc) for doc in snapshot.recovery_docs], "read_only": True, "phase": 1, "mutations": [], }