"""Compose runtime health + inventory into a sessions/runtime view (#641). Phase 1 is read-only. It correlates namespaces, sessions, capabilities, worktree bindings, lease ownership, stale flags, and contamination markers when they are detectable on disk (#630 / #671). It never restarts, kills, or takes over a session. Sources: * :mod:`webui.runtime_health` — profile, role, stale runtime, shell health. * :mod:`webui.inventory` — sessions, leases, locks, worktrees, namespaces and collision signals from the control-plane DB + filesystem. * :mod:`mcp_session_state` — durable contamination markers (inspect only). Secrets are never read. Inventory-sourced values arrive already redacted by :func:`webui.inventory.scrub`; free-form marker text this module loads itself is put through :func:`webui.inventory.scrub_text`, which collapses ``$HOME`` and redacts credential-shaped tokens *inside* a string rather than only at its start. Ownership columns are authority-aware. A lease or lock section that could not be read renders as ``unknown``, never as ``none`` or ``unbound``: a lease the reader could not load is not an absent lease (see :attr:`webui.inventory.InventorySnapshot.ownership_authority_complete`). """ from __future__ import annotations from dataclasses import dataclass from typing import Any, Callable import mcp_session_state from webui.inventory import ( OWNERSHIP_SECTIONS, STATUS_OK, InventorySection, InventorySnapshot, load_inventory_snapshot, scrub_text, snapshot_to_dict as inventory_snapshot_to_dict, ) from webui.runtime_health import ( RuntimeSnapshot, load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict, ) # Sanctioned recovery pointers only — never pkill / killall (#630). SANCTIONED_RECOVERY_DOCS: tuple[dict[str, str], ...] = ( { "label": "MCP namespace EOF recovery (reconnect only)", "path": "docs/mcp-namespace-eof-recovery.md", "note": "IDE/client reconnect or operator-owned restart; never kill daemons.", }, { "label": "MCP namespace health", "path": "docs/mcp-namespace-health.md", "note": "client_namespace probe proves namespace health.", }, { "label": "Restart path inventory", "path": "docs/mcp-restart-path-inventory.md", "note": "Catalog of sanctioned reconnect/restart paths.", }, { "label": "Local web UI recovery", "path": "docs/webui-local-dev.md", "note": "Operator console start and documented recovery sequence.", }, ) _CONTAMINATION_KINDS: tuple[str, ...] = ( mcp_session_state.KIND_RUNTIME_RECOVERY_CONTAMINATION, mcp_session_state.KIND_STABLE_BRANCH_CONTAMINATION, ) #: Status recorded on a row when the backing inventory section is absent #: entirely — distinct from a section that reported itself degraded. AUTHORITY_MISSING = "missing" def _section_status(section: InventorySection | None) -> str: """Status of an ownership section, treating an absent section as missing.""" if section is None: return AUTHORITY_MISSING return section.status def _combined_authority(*statuses: str) -> str: """Worst status of the sections a derived column depends on. A column proved from two sections is only trustworthy when *both* read cleanly, so the first non-``ok`` status wins. """ for status in statuses: if status != STATUS_OK: return status return STATUS_OK @dataclass(frozen=True) class ContaminationMarker: """A detectable durable contamination marker (audit-safe summary).""" kind: str on_disk: bool has_payload: bool summary: str reason_class: str | None = None session_id: str | None = None role: str | None = None command_summary: str | None = None cleared: bool = False def to_dict(self) -> dict[str, Any]: return { "kind": self.kind, "on_disk": self.on_disk, "has_payload": self.has_payload, "summary": self.summary, "reason_class": self.reason_class, "session_id": self.session_id, "role": self.role, "command_summary": self.command_summary, "cleared": self.cleared, "active": self.on_disk and self.has_payload and not self.cleared, } @dataclass(frozen=True) class SessionRow: """One correlated session row for the sessions table.""" session_id: str role: str | None profile: str | None namespace: str | None pid: int | None pid_alive: bool | None status: str | None started_at: str | None last_heartbeat_at: str | None lease_ids: tuple[str, ...] = () work_refs: tuple[str, ...] = () worktree_paths: tuple[str, ...] = () stale_flags: tuple[str, ...] = () contamination_flags: tuple[str, ...] = () #: Status of the section backing ``lease_ids``/``work_refs``. While this is #: not ``ok`` those tuples mean "could not be read", never "none held". lease_authority: str = STATUS_OK #: Worst status across the sections backing ``worktree_paths`` (locks are #: correlated through lease work numbers, so both must read cleanly). worktree_authority: str = STATUS_OK @property def ownership_authority_complete(self) -> bool: """True only when this row's ownership columns are provable.""" return self.lease_authority == STATUS_OK and self.worktree_authority == STATUS_OK def to_dict(self) -> dict[str, Any]: return { "session_id": self.session_id, "role": self.role, "profile": self.profile, "namespace": self.namespace, "pid": self.pid, "pid_alive": self.pid_alive, "status": self.status, "started_at": self.started_at, "last_heartbeat_at": self.last_heartbeat_at, "lease_ids": list(self.lease_ids), "work_refs": list(self.work_refs), "worktree_paths": list(self.worktree_paths), "stale_flags": list(self.stale_flags), "contamination_flags": list(self.contamination_flags), "is_stale": bool(self.stale_flags), "is_contaminated": bool(self.contamination_flags), "lease_authority": self.lease_authority, "worktree_authority": self.worktree_authority, "lease_authority_complete": self.lease_authority == STATUS_OK, "worktree_authority_complete": self.worktree_authority == STATUS_OK, "ownership_authority_complete": self.ownership_authority_complete, "ownership_note": ( "Lease and worktree columns are proved from sections that read " "cleanly." if self.ownership_authority_complete else "An ownership source could not be read; empty lease_ids or " "worktree_paths on this row mean unknown, not unowned." ), } @dataclass(frozen=True) class SessionViewSnapshot: """Composed runtime + session inventory view (#641).""" runtime: RuntimeSnapshot inventory: InventorySnapshot sessions: tuple[SessionRow, ...] contamination_markers: tuple[ContaminationMarker, ...] recovery_docs: tuple[dict[str, str], ...] = SANCTIONED_RECOVERY_DOCS fetch_error: str | None = None @property def stale_session_count(self) -> int: return sum(1 for row in self.sessions if row.stale_flags) @property def contaminated_session_count(self) -> int: return sum(1 for row in self.sessions if row.contamination_flags) @property def active_contamination(self) -> tuple[ContaminationMarker, ...]: return tuple(m for m in self.contamination_markers if m.to_dict()["active"]) @property def ownership_authority_complete(self) -> bool: """False while any ownership section is degraded, absent, or unavailable.""" return self.inventory.ownership_authority_complete @property def ownership_section_status(self) -> dict[str, str]: """Per-section status for the three ownership-bearing sections.""" return { name: _section_status(self.inventory.section(name)) for name in OWNERSHIP_SECTIONS } def _inspect_contamination( kind: str, *, remote: str | None, inspect: Callable[..., dict[str, Any]] | None = None, load: Callable[..., dict[str, Any] | None] | None = None, ) -> ContaminationMarker: """Inspect one contamination kind; never raises into the page render path.""" inspect_fn = inspect or mcp_session_state.inspect_state_envelope load_fn = load or mcp_session_state.load_state try: envelope = inspect_fn(kind=kind, remote=remote) except Exception as exc: # noqa: BLE001 — fail soft for the dashboard return ContaminationMarker( kind=kind, on_disk=False, has_payload=False, summary=f"contamination inspect failed: {type(exc).__name__}", ) reason_class = None session_id = None role = None command_summary = None cleared = False summary = scrub_text(str(envelope.get("summary") or "")) if envelope.get("on_disk") and envelope.get("has_payload"): try: payload = load_fn(kind=kind, remote=remote) or {} except Exception: # noqa: BLE001 payload = {} if isinstance(payload, dict): reason_class = payload.get("reason_class") session_id = payload.get("session_id") role = payload.get("role") command_summary = payload.get("command_summary") or payload.get("detail") cleared = bool(payload.get("cleared_by_reconciler")) if not summary: summary = scrub_text( f"{kind}: {reason_class or 'present'}" + (" (cleared)" if cleared else "") ) # Marker payloads are operator-supplied free text and are the one thing on # this page that does not arrive through inventory scrubbing. The write-time # redactor is a narrow denylist, so redact again at the display boundary: # it leaves $HOME paths, `-H 'X-Api-Key: …'`, `--password …`, and # `PRIVATE_KEY=…` intact. The field itself stays — it is #630 evidence. return ContaminationMarker( kind=kind, on_disk=bool(envelope.get("on_disk")), has_payload=bool(envelope.get("has_payload")), summary=summary or f"{kind}: not present", reason_class=scrub_text(str(reason_class)) if reason_class else None, session_id=scrub_text(str(session_id)) if session_id else None, role=scrub_text(str(role)) if role else None, command_summary=scrub_text(str(command_summary)) if command_summary else None, cleared=cleared, ) def _load_contamination_markers( *, remote: str | None, inspect: Callable[..., dict[str, Any]] | None = None, load: Callable[..., dict[str, Any] | None] | None = None, ) -> tuple[ContaminationMarker, ...]: return tuple( _inspect_contamination(kind, remote=remote, inspect=inspect, load=load) for kind in _CONTAMINATION_KINDS ) def _build_session_rows( inventory: InventorySnapshot, contamination: tuple[ContaminationMarker, ...], ) -> tuple[SessionRow, ...]: sessions_section = inventory.section("sessions") leases_section = inventory.section("leases") locks_section = inventory.section("locks") # Ownership columns may only assert absence when their source read cleanly. lease_authority = _section_status(leases_section) # Locks carry claimant profile/username, not control-plane session ids, so a # worktree binding is correlated through lease work numbers: it depends on # the locks *and* the leases section. worktree_authority = _combined_authority( lease_authority, _section_status(locks_section) ) leases_by_session: dict[str, list[dict[str, Any]]] = {} for lease in (leases_section.items if leases_section else ()): sid = str(lease.get("session_id") or "") if sid: leases_by_session.setdefault(sid, []).append(lease) active_markers = [m for m in contamination if m.to_dict()["active"]] marker_session_ids = { m.session_id for m in active_markers if m.session_id } rows: list[SessionRow] = [] for raw in sessions_section.items if sessions_section else (): sid = str(raw.get("session_id") or "") if not sid: continue session_leases = leases_by_session.get(sid, []) lease_ids = tuple( str(lease["lease_id"]) for lease in session_leases if lease.get("lease_id") ) work_refs: list[str] = [] work_numbers: list[int] = [] for lease in session_leases: kind = lease.get("work_kind") number = lease.get("work_number") if kind and number is not None: work_refs.append(f"{kind}#{number}") try: work_numbers.append(int(number)) except (TypeError, ValueError): pass worktree_paths: list[str] = [] for lock in (locks_section.items if locks_section else ()): try: issue_no = int(lock.get("issue_number")) except (TypeError, ValueError): continue if issue_no in work_numbers and lock.get("worktree_path"): worktree_paths.append(str(lock["worktree_path"])) stale_flags: list[str] = [] if raw.get("pid_alive") is False: stale_flags.append("pid-dead") status = str(raw.get("status") or "").lower() if status and status not in {"active", "alive", "running", "ok"}: stale_flags.append(f"status:{status}") for lease in session_leases: if lease.get("expired") is True: stale_flags.append("lease-expired") if str(lease.get("status") or "").lower() == "active" and lease.get( "expired" ) is True: stale_flags.append("active-lease-past-expiry") contamination_flags: list[str] = [] if sid in marker_session_ids: for marker in active_markers: if marker.session_id == sid: contamination_flags.append(marker.kind) # Process-wide contamination with no session binding still surfaces # against every live session so it cannot be silent (#630). for marker in active_markers: if not marker.session_id and marker.kind not in contamination_flags: contamination_flags.append(f"{marker.kind}:process-wide") rows.append( SessionRow( session_id=sid, role=raw.get("role"), profile=raw.get("profile"), namespace=raw.get("namespace"), pid=raw.get("pid") if isinstance(raw.get("pid"), int) else None, pid_alive=raw.get("pid_alive") if isinstance(raw.get("pid_alive"), bool) else None, status=raw.get("status"), started_at=raw.get("started_at"), last_heartbeat_at=raw.get("last_heartbeat_at"), lease_ids=lease_ids, work_refs=tuple(work_refs), worktree_paths=tuple(worktree_paths), stale_flags=tuple(dict.fromkeys(stale_flags)), contamination_flags=tuple(dict.fromkeys(contamination_flags)), lease_authority=lease_authority, worktree_authority=worktree_authority, ) ) return tuple(rows) def load_session_view_snapshot( *, load_runtime: Callable[..., RuntimeSnapshot] | None = None, load_inventory: Callable[..., InventorySnapshot] | None = None, inspect_contamination: Callable[..., dict[str, Any]] | None = None, load_contamination_payload: Callable[..., dict[str, Any] | None] | None = None, ) -> SessionViewSnapshot: """Build the composed sessions/runtime view. Fail-soft on partial sources.""" runtime_loader = load_runtime or load_runtime_snapshot inventory_loader = load_inventory or load_inventory_snapshot fetch_error: str | None = None try: runtime = runtime_loader() except Exception as exc: # noqa: BLE001 fetch_error = f"runtime snapshot failed: {type(exc).__name__}: {exc}" # Minimal placeholder so the page still renders inventory + recovery. from webui.runtime_health import RuntimeSnapshot as _RS runtime = _RS( project_id="unknown", repo_root="", remote="", host="", profile_name="unknown", role_kind="unknown", config_model="unknown", profile_mode="unknown", profile_source="unknown", authenticated_username=None, identity_error=str(exc), repo_sha=None, remote_master_sha=None, commits_behind_master=None, stale_runtime_warning=None, shell_health={}, workflow_hashes=(), schema_hashes=(), restart_guidance="docs/mcp-namespace-eof-recovery.md", fetch_error=str(exc), ) try: inventory = inventory_loader() except Exception as exc: # noqa: BLE001 msg = f"inventory snapshot failed: {type(exc).__name__}: {exc}" fetch_error = f"{fetch_error}; {msg}" if fetch_error else msg inventory = load_inventory_snapshot( db_path="/nonexistent-for-fail-soft", lock_dir="/nonexistent-for-fail-soft", ) remote = getattr(runtime, "remote", None) contamination = _load_contamination_markers( remote=remote, inspect=inspect_contamination, load=load_contamination_payload, ) sessions = _build_session_rows(inventory, contamination) return SessionViewSnapshot( runtime=runtime, inventory=inventory, sessions=sessions, contamination_markers=contamination, fetch_error=fetch_error, ) def snapshot_to_dict(snapshot: SessionViewSnapshot) -> dict[str, Any]: """JSON export for ``/api/sessions`` (read-only).""" return { "api_version": "v1", "view": "runtime-sessions", "issue": 641, "fetch_error": snapshot.fetch_error, "runtime": runtime_snapshot_to_dict(snapshot.runtime), "inventory": inventory_snapshot_to_dict(snapshot.inventory), "sessions": [row.to_dict() for row in snapshot.sessions], "session_counts": { "total": len(snapshot.sessions), "stale": snapshot.stale_session_count, "contaminated": snapshot.contaminated_session_count, }, "ownership_authority_complete": snapshot.ownership_authority_complete, "ownership_section_status": snapshot.ownership_section_status, "ownership_note": ( "Every ownership source read cleanly; a session with no lease and no " "worktree path genuinely holds neither." if snapshot.ownership_authority_complete else "An ownership source is degraded or unavailable. Empty lease_ids " "and worktree_paths mean unknown, not unowned; consult each row's " "lease_authority and worktree_authority." ), "contamination_markers": [ marker.to_dict() for marker in snapshot.contamination_markers ], "active_contamination": [ marker.to_dict() for marker in snapshot.active_contamination ], "recovery_docs": [dict(doc) for doc in snapshot.recovery_docs], "read_only": True, "phase": 1, "mutations": [], }