Compare commits

..
31 changed files with 1408 additions and 5634 deletions
File diff suppressed because it is too large Load Diff
+37 -80
View File
@@ -40,88 +40,22 @@ def _normalize_path(path: str) -> str:
return (path or "").replace("\\", "/").rstrip("/") return (path or "").replace("\\", "/").rstrip("/")
def get_canonical_branches_root(project_root: str | None = None) -> str:
"""Return the absolute path of the canonical branches directory for *project_root*."""
root = os.path.realpath(project_root) if project_root else os.path.realpath(os.getcwd())
canonical_repo_root = resolve_canonical_repo_root(root, root)
return os.path.realpath(os.path.join(canonical_repo_root, "branches"))
def is_path_under_branches(path: str, project_root: str | None = None) -> bool: def is_path_under_branches(path: str, project_root: str | None = None) -> bool:
"""True when *path* resolves inside a canonical ``branches/`` directory.""" """True when *path* resolves inside ``<project_root>/branches/``."""
if not path or not str(path).strip(): normalized = _normalize_path(path)
if not normalized:
return False return False
try: if "/branches/" in f"{normalized}/":
real_path = os.path.realpath(os.path.abspath(str(path).strip())) return True
except Exception: if normalized.endswith("/branches"):
return False return True
if project_root:
branches_root = get_canonical_branches_root(project_root or real_path) root = _normalize_path(os.path.realpath(project_root))
try: real = _normalize_path(os.path.realpath(path))
common = os.path.commonpath([branches_root, real_path]) if real.startswith(f"{root}/"):
except Exception: rel = real[len(root) + 1 :]
return False return rel == "branches" or rel.startswith("branches/")
return False
if common != branches_root:
return False
rel = os.path.relpath(real_path, branches_root)
return rel != "." and not rel.startswith("..")
def resolve_canonical_repo_root(workspace_path: str, fallback_project_root: str) -> str:
"""Return the stable repository root for *workspace_path* via git metadata (#460)."""
p = (workspace_path or "").strip()
if p:
try:
res = subprocess.run(
["git", "-C", p, "rev-parse", "--git-common-dir"],
capture_output=True,
text=True,
check=True,
)
common = _realpath_git_common_dir(p, res.stdout)
if common.endswith(f"{os.sep}.git") or os.path.basename(common) == ".git":
candidate_root = os.path.dirname(common)
real_p = os.path.realpath(p)
try:
if os.path.commonpath([candidate_root, real_p]) == candidate_root:
return candidate_root
except Exception:
pass
except Exception:
pass
# Fallback when git metadata is unavailable. Never string-split on
# "/branches/" (review #531 F2 / #551): recover the repo root only via
# resolved-path commonpath ancestry. Do **not** require on-disk isdir —
# MCP may launch with project_root = branches/<wt> before that path
# exists, and #274 path-shaped worktree-as-project-root must still resolve.
fallback = os.path.realpath(fallback_project_root or workspace_path or ".")
cur = fallback
for _ in range(64):
parent = os.path.dirname(cur)
if parent == cur:
break
branches_dir = os.path.realpath(os.path.join(parent, "branches"))
try:
# Path-shaped: fallback is under parent/branches/ (commonpath).
if os.path.commonpath([branches_dir, fallback]) == branches_dir:
return parent
except ValueError:
pass
# Fallback path itself is the branches directory.
if os.path.basename(os.path.realpath(cur)) == "branches":
try:
if os.path.commonpath([os.path.realpath(cur), fallback]) == os.path.realpath(
cur
):
return parent
except ValueError:
pass
cur = parent
return fallback
def resolve_mutation_workspace( def resolve_mutation_workspace(
@@ -153,6 +87,29 @@ def _realpath_git_common_dir(workspace_path: str, common_dir: str) -> str:
return os.path.realpath(os.path.join(workspace_path, raw)) return os.path.realpath(os.path.join(workspace_path, raw))
def resolve_canonical_repo_root(workspace_path: str, fallback_project_root: str) -> str:
"""Return the stable repository root for *workspace_path* via git metadata (#460)."""
path = (workspace_path or "").strip()
fallback = os.path.realpath(fallback_project_root)
if not path:
return fallback
try:
res = subprocess.run(
["git", "-C", path, "rev-parse", "--git-common-dir"],
capture_output=True,
text=True,
check=True,
)
common = _realpath_git_common_dir(path, res.stdout)
except Exception:
return fallback
if common.endswith(f"{os.sep}.git"):
return os.path.dirname(common)
if os.path.basename(common) == ".git":
return os.path.dirname(common)
return fallback
def resolve_author_mutation_context( def resolve_author_mutation_context(
worktree_path: str | None, worktree_path: str | None,
process_project_root: str, process_project_root: str,
+198 -1
View File
@@ -31,7 +31,7 @@ from typing import Any, Iterator, Sequence
import dependency_graph import dependency_graph
SCHEMA_VERSION = 4 SCHEMA_VERSION = 5
# Assignable work kinds only — raw monitoring incidents are never work items. # Assignable work kinds only — raw monitoring incidents are never work items.
WORK_KINDS = frozenset({"issue", "pr"}) WORK_KINDS = frozenset({"issue", "pr"})
@@ -186,6 +186,34 @@ CREATE INDEX IF NOT EXISTS idx_dependency_edges_target
ON dependency_edges(remote, org, repo, target_kind, target_number); ON dependency_edges(remote, org, repo, target_kind, target_number);
CREATE INDEX IF NOT EXISTS idx_assignments_session ON assignments(session_id, status); CREATE INDEX IF NOT EXISTS idx_assignments_session ON assignments(session_id, status);
CREATE INDEX IF NOT EXISTS idx_incident_gitea ON incident_links(gitea_org, gitea_repo, gitea_issue_number); CREATE INDEX IF NOT EXISTS idx_incident_gitea ON incident_links(gitea_org, gitea_repo, gitea_issue_number);
-- Model usage, token cost, latency, and performance events (#651)
CREATE TABLE IF NOT EXISTS usage_events (
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
session_id TEXT,
remote TEXT NOT NULL DEFAULT 'dadeschools',
org TEXT NOT NULL DEFAULT '',
repo TEXT NOT NULL DEFAULT '',
project_id TEXT,
role TEXT NOT NULL DEFAULT 'unknown',
model TEXT NOT NULL DEFAULT 'unknown',
issue_number INTEGER,
pr_number INTEGER,
stage TEXT NOT NULL DEFAULT 'unknown',
input_tokens INTEGER,
output_tokens INTEGER,
total_tokens INTEGER,
estimated_cost_usd REAL,
latency_ms INTEGER,
duration_ms INTEGER,
status TEXT NOT NULL DEFAULT 'success',
metadata TEXT,
created_at TEXT NOT NULL
);
CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo);
CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model);
CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage);
""" """
@@ -339,6 +367,7 @@ class ControlPlaneDB:
self._migrate_incident_links_null_scope(conn) self._migrate_incident_links_null_scope(conn)
self._migrate_lease_lifecycle_columns(conn) self._migrate_lease_lifecycle_columns(conn)
self._migrate_session_ownership_columns(conn) self._migrate_session_ownership_columns(conn)
self._migrate_usage_events_table(conn)
conn.execute( conn.execute(
"INSERT OR REPLACE INTO schema_meta(key, value) VALUES (?, ?)", "INSERT OR REPLACE INTO schema_meta(key, value) VALUES (?, ?)",
("schema_version", str(SCHEMA_VERSION)), ("schema_version", str(SCHEMA_VERSION)),
@@ -518,6 +547,174 @@ class ControlPlaneDB:
f"UPDATE incident_links SET {col} = '' WHERE {col} IS NULL" f"UPDATE incident_links SET {col} = '' WHERE {col} IS NULL"
) )
def _migrate_usage_events_table(self, conn: sqlite3.Connection) -> None:
"""Create usage_events table and indexes if they do not exist (#651)."""
conn.execute("""
CREATE TABLE IF NOT EXISTS usage_events (
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
session_id TEXT,
remote TEXT NOT NULL DEFAULT 'dadeschools',
org TEXT NOT NULL DEFAULT '',
repo TEXT NOT NULL DEFAULT '',
project_id TEXT,
role TEXT NOT NULL DEFAULT 'unknown',
model TEXT NOT NULL DEFAULT 'unknown',
issue_number INTEGER,
pr_number INTEGER,
stage TEXT NOT NULL DEFAULT 'unknown',
input_tokens INTEGER,
output_tokens INTEGER,
total_tokens INTEGER,
estimated_cost_usd REAL,
latency_ms INTEGER,
duration_ms INTEGER,
status TEXT NOT NULL DEFAULT 'success',
metadata TEXT,
created_at TEXT NOT NULL
);
""")
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo);")
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model);")
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage);")
def record_usage_event(
self,
*,
session_id: str | None = None,
remote: str = "dadeschools",
org: str = "",
repo: str = "",
project_id: str | None = None,
role: str = "unknown",
model: str = "unknown",
issue_number: int | None = None,
pr_number: int | None = None,
stage: str = "unknown",
input_tokens: int | None = None,
output_tokens: int | None = None,
total_tokens: int | None = None,
estimated_cost_usd: float | None = None,
latency_ms: int | None = None,
duration_ms: int | None = None,
status: str = "success",
metadata: str | dict[str, Any] | None = None,
created_at: str | None = None,
) -> int:
"""Record a model usage, token cost, latency, or stage performance event (#651)."""
ts = created_at or _ts()
meta_str: str | None = None
if metadata is not None:
from webui import console_redaction
redacted_meta = console_redaction.redact_payload(metadata)
if isinstance(redacted_meta, str):
meta_str = redacted_meta
else:
try:
meta_str = json.dumps(redacted_meta, default=str)
except Exception:
meta_str = str(redacted_meta)
if total_tokens is None and (input_tokens is not None or output_tokens is not None):
total_tokens = (input_tokens or 0) + (output_tokens or 0)
with self._tx(immediate=True) as conn:
cursor = conn.execute(
"""
INSERT INTO usage_events (
session_id, remote, org, repo, project_id, role, model,
issue_number, pr_number, stage, input_tokens, output_tokens,
total_tokens, estimated_cost_usd, latency_ms, duration_ms,
status, metadata, created_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
""",
(
session_id,
remote,
org,
repo,
project_id,
role,
model,
issue_number,
pr_number,
stage,
input_tokens,
output_tokens,
total_tokens,
estimated_cost_usd,
latency_ms,
duration_ms,
status,
meta_str,
ts,
),
)
return cursor.lastrowid
def query_usage_events(
self,
*,
remote: str | None = None,
org: str | None = None,
repo: str | None = None,
project_id: str | None = None,
role: str | None = None,
model: str | None = None,
issue_number: int | None = None,
pr_number: int | None = None,
stage: str | None = None,
session_id: str | None = None,
limit: int = 500,
offset: int = 0,
) -> list[dict[str, Any]]:
"""Query stored usage events matching filters (#651)."""
conditions = []
params = []
if remote:
conditions.append("remote = ?")
params.append(remote)
if org:
conditions.append("org = ?")
params.append(org)
if repo:
conditions.append("repo = ?")
params.append(repo)
if project_id:
conditions.append("project_id = ?")
params.append(project_id)
if role:
conditions.append("role = ?")
params.append(role)
if model:
conditions.append("model = ?")
params.append(model)
if issue_number is not None:
conditions.append("issue_number = ?")
params.append(issue_number)
if pr_number is not None:
conditions.append("pr_number = ?")
params.append(pr_number)
if stage:
conditions.append("stage = ?")
params.append(stage)
if session_id:
conditions.append("session_id = ?")
params.append(session_id)
where_clause = f"WHERE {' AND '.join(conditions)}" if conditions else ""
sql = f"""
SELECT * FROM usage_events
{where_clause}
ORDER BY usage_id ASC
LIMIT ? OFFSET ?
"""
params.extend([limit, offset])
with self._tx(immediate=False) as conn:
cursor = conn.execute(sql, params)
rows = cursor.fetchall()
return [dict(row) for row in rows]
# ── sessions ────────────────────────────────────────────────────────── # ── sessions ──────────────────────────────────────────────────────────
def upsert_session( def upsert_session(
+1 -2
View File
@@ -247,8 +247,7 @@ def bootstrap_permits_control_checkout(
""" """
if not isinstance(assessment, dict): if not isinstance(assessment, dict):
return False return False
import author_issue_bootstrap if not is_create_issue_task(task):
if not is_create_issue_task(task) and not author_issue_bootstrap.is_author_issue_bootstrap_task(task):
return False return False
# Positive proof: the assessment must affirmatively allow, with no # Positive proof: the assessment must affirmatively allow, with no
-1
View File
@@ -69,7 +69,6 @@ that gates each call, not which tools exist.
- `gitea_audit_worktree_cleanup` - `gitea_audit_worktree_cleanup`
- `gitea_authorize_reconciliation_cleanup_phase` - `gitea_authorize_reconciliation_cleanup_phase`
- `gitea_authorize_review_correction` - `gitea_authorize_review_correction`
- `gitea_bootstrap_author_issue_worktree`
- `gitea_capability_stop_terminal_report` - `gitea_capability_stop_terminal_report`
- `gitea_capture_branches_worktree_snapshot` - `gitea_capture_branches_worktree_snapshot`
- `gitea_check_pr_eligibility` - `gitea_check_pr_eligibility`
@@ -0,0 +1,125 @@
# Model Usage, Token Cost, Latency, and Workflow Analytics (Phase 4)
- **Tracking Issue:** [#651](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
- **Console Surface:** `/analytics`, `/api/v1/analytics`, `/api/v1/analytics/usage`
## 1. Overview
The Web Console Analytics module provides durable, aggregate visibility into **model usage, token cost, latency percentiles, and workflow-stage performance** across projects, worker roles, AI models, issues, and PRs.
### Non-Goals
- No mandatory client-side telemetry that leaks prompts or secret keys.
- No third-party payment provider or billing integration.
- No automatic model routing changes without controller policy (#647).
---
## 2. Event Schema (`usage_events`)
Usage metrics are stored in the control-plane database under table `usage_events`.
| Column | Type | Description |
|---|---|---|
| `usage_id` | `INTEGER` | Primary key (autoincrement) |
| `session_id` | `TEXT` | Optional active session identifier |
| `remote` | `TEXT` | Known Gitea instance (`dadeschools` or `prgs`) |
| `org` | `TEXT` | Repository owner / organization |
| `repo` | `TEXT` | Repository name |
| `project_id` | `TEXT` | Optional project identifier |
| `role` | `TEXT` | Active worker role (`author`, `reviewer`, `merger`, `reconciler`, `controller`) |
| `model` | `TEXT` | LLM model identifier (e.g. `gemini-3.6-flash`, `claude-3-5-sonnet`) |
| `issue_number` | `INTEGER` | Correlated Gitea issue number (optional) |
| `pr_number` | `INTEGER` | Correlated Gitea PR number (optional) |
| `stage` | `TEXT` | Workflow stage (`preflight`, `implementation`, `review`, `merge`, `reconciliation`) |
| `input_tokens` | `INTEGER` | Input token count (optional / nullable) |
| `output_tokens` | `INTEGER` | Output token count (optional / nullable) |
| `total_tokens` | `INTEGER` | Total token count (optional / nullable) |
| `estimated_cost_usd` | `REAL` | Estimated USD cost (optional / nullable) |
| `latency_ms` | `INTEGER` | Request latency in milliseconds (optional / nullable) |
| `duration_ms` | `INTEGER` | Stage execution duration in milliseconds (optional / nullable) |
| `status` | `TEXT` | Outcome status (`success`, `failure`, `timeout`) |
| `metadata` | `TEXT` | Redacted metadata or summary string |
| `created_at` | `TEXT` | ISO 8601 UTC timestamp |
---
## 3. Handling of Missing Data ("Unknown" vs. Zero Fabrication)
To ensure operational metrics accurately reflect evidence:
- **Untracked or missing metrics are displayed as `Unknown`**, never zero-fabricated.
- If an event omits `estimated_cost_usd`, `latency_ms`, or token counts, the aggregator marks those fields as missing (`None`) rather than defaulting to `0` or `$0.00`.
- Summary tables and KPI cards explicitly indicate when data is unmeasured or partially reported.
---
## 4. Redaction & Security Rules
Per `#633` security policy:
- Free-text fields (`metadata`, `prompt_summary`, `session_id`) are run through `console_redaction.redact_text` before persistence and output serialization.
- Secret tokens, keychain commands, authorization headers, passwords, and JWTs are stripped automatically.
---
## 5. Opt-in Instrumentation Guide
Applications, MCP servers, and background sessions can report usage metrics through either Python API or HTTP ingestion.
### Python Ingestion
```python
from webui.analytics_loader import record_usage
record_usage(
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
role="author",
model="gemini-3.6-flash",
issue_number=651,
stage="implementation",
input_tokens=1420,
output_tokens=380,
total_tokens=1800,
estimated_cost_usd=0.00045,
latency_ms=320,
duration_ms=4500,
status="success",
metadata={"note": "Implementation of analytics module"},
)
```
### HTTP Ingestion API
```http
POST /api/v1/analytics/usage HTTP/1.1
Content-Type: application/json
{
"remote": "dadeschools",
"org": "Scaled-Tech-Consulting",
"repo": "Gitea-Tools",
"role": "author",
"model": "gemini-3.6-flash",
"issue_number": 651,
"stage": "implementation",
"input_tokens": 1420,
"output_tokens": 380,
"total_tokens": 1800,
"estimated_cost_usd": 0.00045,
"latency_ms": 320,
"duration_ms": 4500,
"status": "success",
"metadata": "Analytics schema landed"
}
```
---
## 6. Querying Analytics API
```http
GET /api/v1/analytics?role=author&stage=implementation HTTP/1.1
```
Returns `AnalyticsSnapshot` JSON containing aggregations (`by_model`, `by_stage`, `by_role`, `by_work_item`, `by_project`) and latency percentiles (`p50`, `p90`, `p95`, `p99`).
-122
View File
@@ -1,122 +0,0 @@
# Sanctioned restart and graceful reload controls (#642)
Sessions used to recover MCP connectivity by killing the host daemon
(`pkill -f mcp_server.py`, #630). That is forbidden and stays forbidden: it
kills every namespace on the host, contaminates whichever session survives, and
leaves no audit trail. This document describes the sanctioned replacement,
implemented in `webui/sanctioned_restart.py`.
## What the console will and will not do
The console **never** restarts anything. It authorizes an intent, records it,
and hands off to a host supervisor. There is no code path in which the console
sends a signal, spawns a process, or renders a kill command — a regression test
asserts the module contains no `subprocess`, `signal`, `os.kill`, `os.system`,
or `popen` reference, and that no returned payload contains a kill command.
## Operations
| Mode | Action | Minimum role | Behaviour |
|------|--------|--------------|-----------|
| `reload` | `system.reload_namespace` | controller | Host supervisor reloads the namespace in place, draining in-flight requests. |
| `restart` | `system.restart_namespace` | admin | Host supervisor restarts the namespace. In-flight requests are lost. |
Scope is always exactly one namespace. A fleet-wide restart is an explicit
non-goal: `all`, `*`, `fleet`, and an empty scope are refused with
`fleet_scope_not_permitted`, because that is precisely the blast radius the
forbidden kill already had. An unrecognised namespace is refused rather than
passed through to the host.
## The gate sequence
`assess_restart_request()` applies every gate in order and reports the first
failure with a stable reason code:
| Order | Gate | Reason code on failure |
|-------|------|------------------------|
| 1 | Mode is `restart` or `reload` | `unknown_mode` |
| 2 | Scope is a single known namespace | `fleet_scope_not_permitted`, `unknown_namespace` |
| 3 | Principal holds the required console role | `unauthorized` |
| 4 | Confirmation phrase supplied | `confirmation_required` |
| 5 | Confirmation names this namespace and mode | `confirmation_mismatch` |
| 6 | Out-of-band operator authorization present | `operator_authorization_missing` |
| 7 | Runtime is not contaminated | `contaminated_runtime` |
| 8 | Host restart hook configured | `restart_hook_not_configured` |
Passing every gate yields `host_action_required`, never "restarted".
### Confirmation binds the namespace
The required phrase is `"<mode> <namespace>"` — for example
`restart gitea-author`. Binding the namespace into the phrase is the point: a
confirmation typed for one namespace cannot be replayed against another.
### Operator authorization is not self-assertable
Host daemon maintenance is authorized out of band through
`GITEA_OPERATOR_DAEMON_MAINTENANCE_AUTHORIZATION`, read from the process
environment and nowhere else (#630; #710 finding F1). A worker session cannot
set an environment variable for an already-running daemon, so this cannot be
faked the way a tool argument could.
### The host hook
`GITEA_SANCTIONED_RESTART_HOOK` holds an opaque reference the *host* resolves —
a supervisor label such as a launchd job name, never a command line. With no
hook configured the request is refused; the console does not fall back to a
process kill. The value is read server-side and never rendered to a client.
## Manual kill remains contamination
`classify_restart_command()` classifies an operator-proposed recovery command.
A manual `pkill`/`kill`/`killall` of the MCP daemon is contamination, not a
restart: it returns `clean_claim_allowed: false` and builds a durable
contamination marker (redacted command only, never secrets) naming
`system.restart_namespace` as the sanctioned alternative.
A live, uncleared contamination marker also blocks a restart. This is stricter
than #630's task-scoped gate, which deliberately lets a contaminated worker keep
commenting and handing off: restarting a contaminated runtime would launder the
contamination rather than resolve it. Clear the marker through the reconciler
path first.
## Post-restart health verification
After the host supervisor acts, `verify_post_restart_health()` decides whether
the session may claim to be clean:
| Status | Meaning | Clean claim |
|--------|---------|-------------|
| `clean` | Required tool callable, proven through the live client namespace | Allowed |
| `unproven` | Reported healthy without live client-namespace evidence | Refused |
| `unhealthy` | Probe failed | Refused |
Only `probe_source=client_namespace` evidence clears a session. Static tool
registration is not proof, and neither is an offline subprocess probe — an IDE
client can hold a registered tool list while live calls fail with
`client is closing: EOF` (see
[`mcp-namespace-health.md`](mcp-namespace-health.md)).
## Audit
Every attempt — allowed or denied — is recorded through
`webui.console_audit` with actor, target namespace, mode, result, and reason
code, and is redacted before it is persisted. `system.restart_namespace` is
break-glass, so its records are retained for 730 days. Records carry
`process_kill_executed: false`, which is a fact about the code path rather than
a claim: no such path exists.
## Environment variables
| Variable | Purpose |
|----------|---------|
| `GITEA_SANCTIONED_RESTART_HOOK` | Host supervisor reference; absent means restart is refused. |
| `GITEA_OPERATOR_DAEMON_MAINTENANCE_AUTHORIZATION` | Out-of-band operator authorization reference. |
| `WEBUI_AUDIT_LOG` | Console audit sink; absent means records are built but not persisted. |
## Non-goals
* No unrestricted `kill` from the UI, in any role, in any phase.
* No fleet-wide restart.
* No silent auto-restart loop: every attempt is confirmed and audited.
* This does not implement the Phase 1 health API (#634).
+2 -11
View File
@@ -91,8 +91,6 @@ already define, and a regression test asserts each mapping matches.
| `close_pr` | controller | privileged | `gitea.pr.close` | Yes | No | No | 3 | | `close_pr` | controller | privileged | `gitea.pr.close` | Yes | No | No | 3 |
| `merge_pr` | controller | privileged | `gitea.pr.merge` | Yes | **Yes** | **Yes** | 3 | | `merge_pr` | controller | privileged | `gitea.pr.merge` | Yes | **Yes** | **Yes** | 3 |
| `delete_branch` | admin | destructive | `gitea.branch.delete` | Yes | **Yes** | **Yes** | 3 | | `delete_branch` | admin | destructive | `gitea.branch.delete` | Yes | **Yes** | **Yes** | 3 |
| `system.reload_namespace` | controller | privileged | `runtime.reload_namespace` | Yes | No | No | 2 |
| `system.restart_namespace` | admin | destructive | `runtime.restart_namespace` | Yes | **Yes** | **Yes** | 2 |
**Dual control** means the acting principal may not be the sole authority: a **Dual control** means the acting principal may not be the sole authority: a
second distinct principal must confirm. **Break-glass** means the action is second distinct principal must confirm. **Break-glass** means the action is
@@ -104,13 +102,6 @@ honouring it.
`delete_branch` is admin-only rather than controller because it is the one `delete_branch` is admin-only rather than controller because it is the one
irreversible action in the set. irreversible action in the set.
`system.restart_namespace` is admin-only for the same reason: restarting a
namespace drops every in-flight request on it. `system.reload_namespace` drains
first, so it is privileged but not destructive. Neither action is ever executed
by the console — both hand off to a host supervisor, and neither exposes a raw
process kill. See
[`sanctioned-restart-controls.md`](sanctioned-restart-controls.md) (#642).
### Authorization decision ### Authorization decision
`authorize(action_id, principal, for_execution=False)` returns a decision `authorize(action_id, principal, for_execution=False)` returns a decision
@@ -208,8 +199,8 @@ breaking the request it describes.
| Class | Applies to | Default | | Class | Applies to | Default |
|-------|-----------|---------| |-------|-----------|---------|
| `standard` | Routine gated writes | 90 days | | `standard` | Routine gated writes | 90 days |
| `privileged` | `review_pr`, `close_pr`, `system.reload_namespace`, and any unclassifiable action | 365 days | | `privileged` | `review_pr`, `close_pr`, and any unclassifiable action | 365 days |
| `break_glass` | `merge_pr`, `delete_branch`, `system.restart_namespace` | 730 days | | `break_glass` | `merge_pr`, `delete_branch` | 730 days |
Each record carries its own class, day count, and computed `expires_at`, so Each record carries its own class, day count, and computed `expires_at`, so
retention is auditable per record rather than inferred from file age. An retention is auditable per record rather than inferred from file age. An
-79
View File
@@ -54,7 +54,6 @@ status, onboarding checklist state, and the fail-closed error payloads (#635).
| `/` | Home / operator overview | | `/` | Home / operator overview |
| `/health` | JSON liveness (`status`, `service`, `mode`, `timestamp`, `uptime_seconds`) | | `/health` | JSON liveness (`status`, `service`, `mode`, `timestamp`, `uptime_seconds`) |
| `/api/v1/system/health` | Structured read-only system health (#634) | | `/api/v1/system/health` | Structured read-only system health (#634) |
| `/system-health` | System-health dashboard — readiness, version/uptime, dependencies, MCP namespaces, stale-runtime parity (#639) |
| `/queue` | Live PR and issue queue dashboard (#429) | | `/queue` | Live PR and issue queue dashboard (#429) |
| `/api/queue` | JSON queue export with pagination metadata | | `/api/queue` | JSON queue export with pagination metadata |
| `/projects` | Project registry list with status and onboarding progress (#427, #635) | | `/projects` | Project registry list with status and onboarding progress (#427, #635) |
@@ -259,37 +258,6 @@ Not-yet-implemented surfaces (`/sessions`, `/inventory`, `/timeline`,
surfaces are backed by #636). Mutating methods on stub routes still fail closed surfaces are backed by #636). Mutating methods on stub routes still fail closed
with `read-only-mvp`. with `read-only-mvp`.
## System-health dashboard (#639)
`/system-health` renders the same snapshot the `/api/v1/system/health` API
returns, so the page and the API can never disagree. Cards: overall readiness,
stale-runtime parity, version and uptime, dependency probes, MCP namespaces,
probe errors (only when present), and recovery pointers. `?deep=1` opts into
the network probe exactly as the API does; the plain page load stays cheap.
Field authority and honesty rules:
* `ready` and `readiness_complete` are shown separately. A snapshot whose
required probes never ran is not the same as one that ran them and passed,
and the page never collapses the two into an unproven green.
* A probe that did not run appears under **Not probed**, never as healthy.
* `stale_runtime.mutation_safe` is displayed verbatim from the API. When the
runtime is stale, or when parity is indeterminate, the page warns and does
not claim mutation safety.
* MCP namespaces are reported `unproven`: the web process runs outside the
IDE-managed MCP client and cannot prove that path (#543).
Redaction is split by field kind. Free text — probe details, readiness and
parity reasons, probe errors — passes through `system_health.redact`.
Structured fields — commit SHAs, probe names, statuses, timestamps — are
HTML-escaped only, because `redact`'s opaque-token rule matches any run of 32
or more characters and would otherwise blank every 40-character git SHA, which
is precisely the evidence the parity view exists to show.
The dashboard is read-only: no restart, reload, or process-kill control. Those
arrive in Phase 2 (#642). Recovery guidance points at the sanctioned client
reconnect / operator restart path — never a manual daemon kill (#630).
## Deployment boundary (#435) ## Deployment boundary (#435)
MVP serves on loopback by default. Binding `0.0.0.0` or `::` is **refused** MVP serves on loopback by default. Binding `0.0.0.0` or `::` is **refused**
@@ -349,53 +317,6 @@ health, workflow/schema SHA-256 hashes, and stale-runtime warnings when the
checkout is behind merged safety-gate changes. Restart guidance links to #420; checkout is behind merged safety-gate changes. Restart guidance links to #420;
no tokens or MCP restart actions are exposed. no tokens or MCP restart actions are exposed.
## Inventory API (#636)
`GET /api/v1/inventory` returns one versioned, read-only snapshot that unifies
what the lease (#433), worktree (#432), and runtime (#430) MVP views each show
separately, so traffic-control and recovery consumers read the same source.
`GET /api/v1/inventory/{section}` returns a single section under the identical
schema (`sessions`, `leases`, `locks`, `worktrees`, `namespaces`); an unknown
section is a `404` with `error: unknown_section`. Both routes are `GET`-only.
Each section carries its own `status` (`ok` / `degraded` / `unavailable`), a
`reason` when not `ok`, and a `scan_ms`. A subsystem that cannot be read
degrades to a reasoned section; it never raises and never emits an empty list
that would read as "nothing is there".
### Field authority
Every section names where its rows came from; authorities are never blended.
| Section | Authority | Source |
|---|---|---|
| `sessions` | `control_plane_db` | #613 control-plane DB (`mode=ro`), authoritative for exclusive ownership (#600/#601) |
| `leases` | `control_plane_db` | #613 control-plane DB; degrades if the `work_items` table is absent |
| `locks` | `filesystem` | durable per-issue lock files (`issue_lock_store`) |
| `worktrees` | `filesystem` | registered git worktrees via the #432 hygiene scanner |
| `namespaces` | `filesystem` | the active profile serving this web process (others are not enumerable) |
The payload restates this map under `field_authority` for machine consumers.
### Ownership safety
`ownership_authority_complete` is true only when every ownership-bearing section
(`sessions`, `leases`, `locks`) read cleanly. While it is false, nothing is
reported as unowned and no collision is asserted from a degraded source —
absence of evidence is reported as absence of evidence, never as free work.
`collisions` surfaces detectable conflicts, each with a `kind` and `severity`:
`lock-without-worktree`, `duplicate-live-lock`, `live-lock-dead-owner` (unexpired
lease, dead pid — a #753 recovery candidate that would read as live to a naive
timestamp check), `stale-lock-dead-owner`, `expired-lock-live-owner` (the
#635/#760 daemon-pid deadlock), `concurrent-active-lease`, `active-lease-past-expiry`,
and `orphan-lease`. Collisions are emitted only from sections that read cleanly.
The control-plane DB is opened through a `mode=ro` URI so a read never creates
or migrates it; paths are collapsed against `$HOME`, URLs lose userinfo and
query strings, and credential-shaped values are redacted at the boundary. Lease
steal/release and worktree deletion are Phase 2+ and have no representation here.
## Workflow-event timeline (#637) ## Workflow-event timeline (#637)
`GET /api/v1/timeline` is a read-only, versioned aggregation of workflow `GET /api/v1/timeline` is a read-only, versioned aggregation of workflow
+141 -268
View File
@@ -977,32 +977,6 @@ def _create_issue_bootstrap_assessment(
""" """
import create_issue_bootstrap as _cib import create_issue_bootstrap as _cib
import author_issue_bootstrap as _aib
if _aib.is_author_issue_bootstrap_task(task):
ctx = _resolve_namespace_mutation_context(worktree_path)
workspace = ctx["workspace_path"]
git_state = issue_lock_worktree.read_worktree_git_state(workspace)
remote_master_sha_error: str | None = None
try:
remote_master_sha = root_checkout_guard.resolve_remote_master_sha(
ctx["canonical_repo_root"]
)
except Exception as exc:
remote_master_sha = None
remote_master_sha_error = (
f"{type(exc).__name__}: {exc}".strip() or "resolver failed"
)
return _aib.assess_author_issue_bootstrap(
workspace_path=workspace,
canonical_repo_root=ctx["canonical_repo_root"],
current_branch=git_state.get("current_branch"),
head_sha=git_state.get("head_sha"),
porcelain_status=git_state.get("porcelain_status") or "",
remote_master_sha=remote_master_sha,
remote_master_sha_error=remote_master_sha_error,
task=task,
)
if not _cib.is_create_issue_task(task): if not _cib.is_create_issue_task(task):
return None return None
@@ -9606,7 +9580,6 @@ def gitea_commit_files(
host: str | None = None, host: str | None = None,
org: str | None = None, org: str | None = None,
repo: str | None = None, repo: str | None = None,
worktree_path: str | None = None,
) -> dict: ) -> dict:
"""Commit changes to multiple files in a Gitea repository in a single atomic commit. """Commit changes to multiple files in a Gitea repository in a single atomic commit.
@@ -9619,46 +9592,10 @@ def gitea_commit_files(
host: Override the Gitea host. host: Override the Gitea host.
org: Override the owner/organization. org: Override the owner/organization.
repo: Override the repository name. repo: Override the repository name.
worktree_path: Optional worktree path for author mutation context.
Returns: Returns:
dict with success status and commit/branch information. dict with success status and commit/branch information.
""" """
if worktree_path is None:
lock_data = issue_lock_store.read_session_issue_lock() or {}
worktree_path = lock_data.get("worktree_path")
if not worktree_path:
try:
prof = get_profile()
uname = prof.get("username") or prof.get("profile_name")
for path in issue_lock_store.iter_lock_files():
lk = issue_lock_store.read_lock_file(path) or {}
claimant = lk.get("claimant") or {}
if lk.get("remote") == remote and (claimant.get("username") == uname or lk.get("profile") == prof.get("profile_name")):
issue_lock_store.bind_session_lock(lk, renewal_sanctioned=True)
worktree_path = lk.get("worktree_path")
break
except Exception:
pass
if worktree_path is None and files:
for f in files:
p = f.get("workspace_path") or f.get("local_path") or ""
if p and os.path.isabs(p):
real_p = os.path.realpath(p)
real_root = os.path.realpath(PROJECT_ROOT)
branches_dir = os.path.join(real_root, "branches")
if real_p.startswith(branches_dir + os.sep):
rel_sub = os.path.relpath(real_p, branches_dir)
wt_folder = rel_sub.split(os.sep)[0]
if wt_folder and wt_folder != "..":
worktree_path = os.path.join(branches_dir, wt_folder)
break
if worktree_path:
os.environ["GITEA_AUTHOR_WORKTREE"] = worktree_path
os.environ["GITEA_ACTIVE_WORKTREE"] = worktree_path
ok, block_reasons = role_session_router.check_author_mutation_after_reviewer_stop( ok, block_reasons = role_session_router.check_author_mutation_after_reviewer_stop(
"commit_files" "commit_files"
) )
@@ -9671,7 +9608,7 @@ def gitea_commit_files(
"reasons": block_reasons, "reasons": block_reasons,
} }
blocked = _namespace_mutation_block( blocked = _namespace_mutation_block(
"commit_files", commit="", branch="", remote=remote, worktree_path=worktree_path "commit_files", commit="", branch="", remote=remote
) )
if blocked: if blocked:
return blocked return blocked
@@ -9699,7 +9636,7 @@ def gitea_commit_files(
) )
# #735: forward explicit org/repo into shared anti-stomp preflight. # #735: forward explicit org/repo into shared anti-stomp preflight.
verify_preflight_purity(remote=remote, worktree_path=worktree_path, task="commit_files", org=org, repo=repo) verify_preflight_purity(remote, task="commit_files", org=org, repo=repo)
processed_files, source_proofs = _prepare_commit_payload_files(files) processed_files, source_proofs = _prepare_commit_payload_files(files)
h, o, r = _resolve(remote, host, org, repo) h, o, r = _resolve(remote, host, org, repo)
@@ -9976,96 +9913,6 @@ def gitea_publish_unpublished_issue_branch(
} }
@mcp.tool()
def gitea_bootstrap_author_issue_worktree(
issue_number: int,
assignment_id: str | None = None,
lease_id: str | None = None,
expected_base_sha: str | None = None,
branch_name: str | None = None,
worktree_path: str | None = None,
idempotency_key: str | None = None,
remote: str = "dadeschools",
host: str | None = None,
org: str | None = None,
repo: str | None = None,
dry_run: bool = False,
) -> dict:
"""Bootstrap an allocated author issue branch and registered worktree (#850).
Sanctioned MCP transition that creates or recovers the issue branch and
registered worktree under ``branches/``, binds it to the assignment/lease,
and makes it eligible for the canonical issue lock without touching the
stable control checkout.
Args:
issue_number: Allocated issue number to bootstrap.
assignment_id: Optional allocation assignment ID.
lease_id: Optional workflow lease ID.
expected_base_sha: Authoritative expected base SHA / concurrency pin.
branch_name: Optional custom branch name (must match issue-<N> pattern).
worktree_path: Optional custom worktree path under branches/.
idempotency_key: Optional key for idempotent replay/resume.
remote: Known instance 'dadeschools' or 'prgs'.
host: Override Gitea host.
org: Override Org.
repo: Override Repo.
dry_run: Report planned transition without mutating repository.
"""
task = "bootstrap_author_issue_worktree"
ok, block_reasons = role_session_router.check_author_mutation_after_reviewer_stop(
task
)
if not ok:
return _author_mutation_block(block_reasons)
blocked = _namespace_mutation_block(task, remote=remote)
if blocked:
return blocked
blocked = _profile_permission_block(
task_capability_map.required_permission(task),
remote=remote,
host=host,
org=org,
repo=repo,
org_explicit=org is not None,
repo_explicit=repo is not None,
)
if blocked:
return blocked
verify_preflight_purity(
remote,
task=task,
org=org,
repo=repo,
)
h, o, r = _resolve(remote, host, org, repo)
canonical_root = _canonical_local_git_root()
import author_issue_bootstrap
return author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=issue_number,
canonical_repo_root=canonical_root,
assignment_id=assignment_id,
lease_id=lease_id,
expected_base_sha=expected_base_sha,
branch_name=branch_name,
worktree_path=worktree_path,
idempotency_key=idempotency_key,
remote=remote,
host=h,
org=o,
repo=r,
active_identity=_active_username(),
active_profile=_active_profile_name(),
owner_session=_current_session_id(),
dry_run=dry_run,
)
# Merge methods supported by the Gitea merge API. # Merge methods supported by the Gitea merge API.
_MERGE_METHODS = ("merge", "squash", "rebase") _MERGE_METHODS = ("merge", "squash", "rebase")
@@ -12167,127 +12014,163 @@ def gitea_reconcile_merged_cleanups(
if dry_run: if dry_run:
report["dry_run"] = True report["dry_run"] = True
report["executed"] = False report["executed"] = False
# #851: surface planned lifecycle order so dry-run matches execute.
report["planned_execution_orders"] = {
str(entry.get("pr_number")): entry.get("planned_execution_order") or []
for entry in (report.get("entries") or [])
}
return {"success": True, "performed": False, **report} return {"success": True, "performed": False, **report}
verify_preflight_purity( verify_preflight_purity(
remote, task="reconcile_merged_cleanups", org=org, repo=repo remote, task="reconcile_merged_cleanups", org=org, repo=repo
) )
actions: list[dict] = [] actions: list[dict] = []
project_root = _canonical_local_git_root()
def _ownership_records_for_branch(
head_branch: str, pr_num_int: int | None
) -> list[dict]:
ownership_bundle = _collect_branch_ownership_records(
remote=remote,
host=h,
org=o,
repo=r,
branch=head_branch,
pr_number=pr_num_int,
project_root=project_root,
auth=auth,
base_api=base,
)
ownership_records = list(ownership_bundle.get("records") or [])
if ownership_bundle.get("inventory_error"):
ownership_records.append(
{
"category": (
branch_cleanup_guard.OWNERSHIP_CATEGORY_INVENTORY_ERROR
),
"status": "unknown",
"remote": remote,
"host": h,
"org": o,
"repo": r,
"branch": head_branch,
"reclaim_allowed": False,
"role": "inventory",
}
)
return ownership_records
def _attempt_owned_remote_delete(
*,
head_branch: str,
pr_num_int: int | None,
after_worktree_removal: bool = False,
) -> dict:
"""Fail-closed remote delete with live ownership reassessment (#851)."""
import urllib.parse
ownership_records = _ownership_records_for_branch(head_branch, pr_num_int)
ownership = branch_cleanup_guard.assess_active_branch_ownership(
remote=remote,
org=o,
repo=r,
branch=head_branch,
host=h,
records=ownership_records,
)
if ownership.get("block"):
return {
"action": "delete_remote_branch",
"branch": head_branch,
"success": False,
"performed": False,
"delete_acknowledged": False,
"verified_absent": False,
"blocker_kind": "active_branch_ownership",
"reasons": ownership.get("reasons") or [],
"blocking_categories": ownership.get("blocking_categories") or [],
"after_worktree_removal": after_worktree_removal,
"ownership_reassessed": after_worktree_removal,
}
encoded = urllib.parse.quote(head_branch, safe="")
url = f"{base}/branches/{encoded}"
with _audited(
"delete_branch",
host=h,
remote=remote,
org=o,
repo=r,
target_branch=head_branch,
request_metadata={
"branch": head_branch,
"source": "reconcile_merged_cleanups",
"ownership_checked": True,
"after_worktree_removal": after_worktree_removal,
},
):
api_request("DELETE", url, auth)
readback = _probe_remote_branch(h, o, r, auth, head_branch)
readback_assessment = branch_cleanup_guard.assess_post_delete_readback(
readback
)
verified = bool(readback_assessment.get("verified_absent"))
return {
"action": "delete_remote_branch",
"branch": head_branch,
"success": bool(readback_assessment.get("ok")),
"performed": True,
"delete_acknowledged": True,
"verified_absent": verified,
"readback": readback_assessment.get("readback"),
"reasons": readback_assessment.get("reasons") or [],
"after_worktree_removal": after_worktree_removal,
"ownership_reassessed": after_worktree_removal,
}
for entry in report.get("entries") or []: for entry in report.get("entries") or []:
head_branch = entry.get("head_branch") or "" head_branch = entry.get("head_branch") or ""
remote_assessment = entry.get("remote_branch") or {} remote_assessment = entry.get("remote_branch") or {}
local_assessment = entry.get("local_worktree") or {} local_assessment = entry.get("local_worktree") or {}
pr_num = entry.get("pr_number")
try:
pr_num_int = int(pr_num) if pr_num is not None else None
except (TypeError, ValueError):
pr_num_int = None
if remote_assessment.get("safe_to_delete_remote"): # #851 lifecycle: when the target worktree is independently safe, remove
import urllib.parse # it first so worktree_binding ownership does not permanently strand
# both the worktree and the remote branch. Never skip worktree removal
pr_num = entry.get("pr_number") # merely because remote delete would be blocked by that binding.
try: # Ownership protection for remote delete remains fail-closed below.
pr_num_int = int(pr_num) if pr_num is not None else None worktree_removed = False
except (TypeError, ValueError):
pr_num_int = None
ownership_bundle = _collect_branch_ownership_records(
remote=remote,
host=h,
org=o,
repo=r,
branch=head_branch,
pr_number=pr_num_int,
project_root=_canonical_local_git_root(),
auth=auth,
base_api=base,
)
ownership_records = list(ownership_bundle.get("records") or [])
if ownership_bundle.get("inventory_error"):
ownership_records.append(
{
"category": (
branch_cleanup_guard.OWNERSHIP_CATEGORY_INVENTORY_ERROR
),
"status": "unknown",
"remote": remote,
"host": h,
"org": o,
"repo": r,
"branch": head_branch,
"reclaim_allowed": False,
"role": "inventory",
}
)
ownership = branch_cleanup_guard.assess_active_branch_ownership(
remote=remote,
org=o,
repo=r,
branch=head_branch,
host=h,
records=ownership_records,
)
if ownership.get("block"):
actions.append(
{
"action": "delete_remote_branch",
"branch": head_branch,
"success": False,
"performed": False,
"delete_acknowledged": False,
"verified_absent": False,
"blocker_kind": "active_branch_ownership",
"reasons": ownership.get("reasons") or [],
"blocking_categories": ownership.get(
"blocking_categories"
)
or [],
}
)
continue
encoded = urllib.parse.quote(head_branch, safe="")
url = f"{base}/branches/{encoded}"
with _audited(
"delete_branch",
host=h,
remote=remote,
org=o,
repo=r,
target_branch=head_branch,
request_metadata={
"branch": head_branch,
"source": "reconcile_merged_cleanups",
"ownership_checked": True,
},
):
api_request("DELETE", url, auth)
readback = _probe_remote_branch(h, o, r, auth, head_branch)
readback_assessment = branch_cleanup_guard.assess_post_delete_readback(
readback
)
verified = bool(readback_assessment.get("verified_absent"))
actions.append(
{
"action": "delete_remote_branch",
"branch": head_branch,
"success": bool(readback_assessment.get("ok")),
"performed": True,
"delete_acknowledged": True,
"verified_absent": verified,
"readback": readback_assessment.get("readback"),
"reasons": readback_assessment.get("reasons") or [],
}
)
if local_assessment.get("safe_to_remove_worktree"): if local_assessment.get("safe_to_remove_worktree"):
result = merged_cleanup_reconcile.remove_local_worktree( result = merged_cleanup_reconcile.remove_local_worktree(
_canonical_local_git_root(), project_root,
head_branch, head_branch,
worktree_path=local_assessment.get("worktree_path"), worktree_path=local_assessment.get("worktree_path"),
) )
actions.append({"action": "remove_local_worktree", **result}) actions.append({"action": "remove_local_worktree", **result})
# Idempotent resume: absent worktree is already gone.
msg = (result.get("message") or "").lower()
worktree_removed = bool(result.get("success")) or (
"not found" in msg
)
if remote_assessment.get("safe_to_delete_remote"):
actions.append(
_attempt_owned_remote_delete(
head_branch=head_branch,
pr_num_int=pr_num_int,
after_worktree_removal=worktree_removed,
)
)
for scratch in report.get("reviewer_scratch_entries") or []: for scratch in report.get("reviewer_scratch_entries") or []:
if not scratch.get("safe_to_remove_worktree"): if not scratch.get("safe_to_remove_worktree"):
continue continue
result = merged_cleanup_reconcile.remove_reviewer_scratch_worktree( result = merged_cleanup_reconcile.remove_reviewer_scratch_worktree(
_canonical_local_git_root(), scratch.get("worktree_path") or "" project_root, scratch.get("worktree_path") or ""
) )
actions.append({"action": "remove_reviewer_scratch_worktree", **result}) actions.append({"action": "remove_reviewer_scratch_worktree", **result})
@@ -20267,12 +20150,6 @@ def gitea_resolve_task_capability(
task: str, task: str,
remote: str = "dadeschools", remote: str = "dadeschools",
host: str | None = None, host: str | None = None,
org: str | None = None,
repo: str | None = None,
issue_number: int | None = None,
worktree_path: str | None = None,
pr_number: int | None = None,
**kwargs: Any,
) -> dict: ) -> dict:
"""Read-only / side-effect free: resolve capability, profile, and namespace for a task. """Read-only / side-effect free: resolve capability, profile, and namespace for a task.
@@ -20289,14 +20166,13 @@ def gitea_resolve_task_capability(
remote: Known remote instance name. remote: Known remote instance name.
host: Optional override for the Gitea host. host: Optional override for the Gitea host.
""" """
task_key = task_capability_map._canonical_preflight_task(task)
TASK_MAP = task_capability_map.TASK_CAPABILITY_MAP TASK_MAP = task_capability_map.TASK_CAPABILITY_MAP
# Every fresh attempt invalidates the previous task/role stamp before any # Every fresh attempt invalidates the previous task/role stamp before any
# fallible resolver work. Unknown/malformed tasks and unexpected failures # fallible resolver work. Unknown/malformed tasks and unexpected failures
# therefore remain fail-closed instead of preserving stale authority. # therefore remain fail-closed instead of preserving stale authority.
_clear_resolved_capability_stamp() _clear_resolved_capability_stamp()
if task_key not in TASK_MAP: if task not in TASK_MAP:
# #723: structured fail-closed unknown_task (never raise into internal_error). # #723: structured fail-closed unknown_task (never raise into internal_error).
profile = get_profile() profile = get_profile()
h = host or (REMOTES.get(remote, {}).get("host") if remote in REMOTES else None) h = host or (REMOTES.get(remote, {}).get("host") if remote in REMOTES else None)
@@ -20342,8 +20218,8 @@ def gitea_resolve_task_capability(
result["cleared_stale_denial"] = True result["cleared_stale_denial"] = True
return result return result
required_permission = task_capability_map.required_permission(task_key) required_permission = task_capability_map.required_permission(task)
required_role = task_capability_map.required_role(task_key) required_role = task_capability_map.required_role(task)
role_exclusive_tasks = task_capability_map.ROLE_EXCLUSIVE_TASKS role_exclusive_tasks = task_capability_map.ROLE_EXCLUSIVE_TASKS
infra_assessment = role_session_router.assess_infra_stop(PROJECT_ROOT) infra_assessment = role_session_router.assess_infra_stop(PROJECT_ROOT)
@@ -20529,10 +20405,7 @@ def gitea_resolve_task_capability(
available_in_session = allowed_in_current_session available_in_session = allowed_in_current_session
runtime_stale_blocker = False runtime_stale_blocker = False
if ( if "PYTEST_CURRENT_TEST" not in os.environ or "GITEA_FORCE_MCP_RUNTIME_CHECK" in os.environ:
"PYTEST_CURRENT_TEST" not in os.environ
or "GITEA_FORCE_MCP_RUNTIME_CHECK" in os.environ
) and os.environ.get("GITEA_ALLOW_STALE_RUNTIME") != "1":
runtime_reasons = _check_mcp_runtimes_diagnostics(task, matching_profiles) runtime_reasons = _check_mcp_runtimes_diagnostics(task, matching_profiles)
if runtime_reasons: if runtime_reasons:
restart_required = True restart_required = True
-2
View File
@@ -73,8 +73,6 @@ AUTHOR_TASKS = frozenset({
"claim_issue", "claim_issue",
"create_branch", "create_branch",
"push_branch", "push_branch",
"bootstrap_author_issue_worktree",
"gitea_bootstrap_author_issue_worktree",
"create_pr", "create_pr",
"comment_pr", "comment_pr",
"address_pr_change_requests", "address_pr_change_requests",
-30
View File
@@ -78,14 +78,6 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
"permission": "gitea.branch.create", "permission": "gitea.branch.create",
"role": "author", "role": "author",
}, },
"bootstrap_author_issue_worktree": {
"permission": "gitea.branch.create",
"role": "author",
},
"gitea_bootstrap_author_issue_worktree": {
"permission": "gitea.branch.create",
"role": "author",
},
"push_branch": { "push_branch": {
"permission": "gitea.branch.push", "permission": "gitea.branch.push",
"role": "author", "role": "author",
@@ -363,21 +355,6 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
"role": "controller", "role": "controller",
}, },
# #642: sanctioned host-daemon lifecycle controls. Deliberately *not* a
# ``gitea.*`` operation — restarting an MCP namespace is a host action, not
# a Gitea API call, and no configured Gitea profile should be able to
# satisfy it by accident. Authority comes from the console RBAC model plus
# out-of-band operator authorization (#630); these entries exist so the
# console cannot invent an authority the capability layer never declared.
"restart_namespace": {
"permission": "runtime.restart_namespace",
"role": "controller",
},
"reload_namespace": {
"permission": "runtime.reload_namespace",
"role": "controller",
},
# #601 first-class lease lifecycle — inspect/list need read; mutations gate on # #601 first-class lease lifecycle — inspect/list need read; mutations gate on
# ownership in the control-plane DB (not a separate Gitea write permission). # ownership in the control-plane DB (not a separate Gitea write permission).
"list_workflow_leases": { "list_workflow_leases": {
@@ -520,10 +497,6 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
# merger lease (#763). # merger lease (#763).
_PREFLIGHT_TASK_TRANSITIONS = frozenset({ _PREFLIGHT_TASK_TRANSITIONS = frozenset({
("review_pr", "acquire_reviewer_pr_lease"), ("review_pr", "acquire_reviewer_pr_lease"),
# #850: native author issue worktree bootstrap
("work_issue", "bootstrap_author_issue_worktree"),
("bootstrap_author_issue_worktree", "lock_issue"),
# #860: dirty-orphan recovery and related work_issue transitions (master)
("work_issue", "lock_issue"), ("work_issue", "lock_issue"),
("work_issue", "recover_dirty_orphaned_issue_worktree"), ("work_issue", "recover_dirty_orphaned_issue_worktree"),
("work_issue", "gitea_recover_dirty_orphaned_issue_worktree"), ("work_issue", "gitea_recover_dirty_orphaned_issue_worktree"),
@@ -575,8 +548,6 @@ ROLE_EXCLUSIVE_TASKS: frozenset[str] = frozenset(
"gitea_release_merger_pr_lease", "gitea_release_merger_pr_lease",
"create_branch", "create_branch",
"push_branch", "push_branch",
"bootstrap_author_issue_worktree",
"gitea_bootstrap_author_issue_worktree",
"publish_unpublished_branch", "publish_unpublished_branch",
"create_pr", "create_pr",
"commit_files", "commit_files",
@@ -602,7 +573,6 @@ ISSUE_MUTATION_TOOL_TASKS: dict[str, str] = {
"gitea_set_issue_labels": "set_issue_labels", "gitea_set_issue_labels": "set_issue_labels",
"gitea_cleanup_terminal_pr_labels": "cleanup_terminal_pr_labels", "gitea_cleanup_terminal_pr_labels": "cleanup_terminal_pr_labels",
"gitea_create_label": "create_label", "gitea_create_label": "create_label",
"gitea_bootstrap_author_issue_worktree": "bootstrap_author_issue_worktree",
"gitea_commit_files": "commit_files", "gitea_commit_files": "commit_files",
} }
-601
View File
@@ -1,601 +0,0 @@
"""Regression test suite for native author issue worktree bootstrap (#850)."""
from __future__ import annotations
import json
import os
import shutil
import subprocess
import tempfile
import unittest
from unittest import mock
import author_issue_bootstrap
import task_capability_map
def _concurrent_bootstrap_worker(args: tuple[str, int, str, str, str, str]) -> dict:
repo_dir, issue_num, key, lock_dir, journal_dir, master_sha = args
os.environ["GITEA_BOOTSTRAP_JOURNAL_DIR"] = journal_dir
return author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=issue_num,
canonical_repo_root=repo_dir,
expected_base_sha=master_sha,
idempotency_key=key,
lock_dir=lock_dir,
owner_session="session-concurrent-test",
active_identity="jcwalker3",
active_profile="prgs-author",
)
class TestAuthorIssueBootstrap(unittest.TestCase):
"""Test suite covering AC1-AC10 and comment #14959 specification."""
def setUp(self):
self.tmp_dir = tempfile.mkdtemp(prefix="test_bootstrap_")
self.repo_dir = os.path.join(self.tmp_dir, "repo")
os.makedirs(self.repo_dir)
# Initialize synthetic git repo
subprocess.run(["git", "init", "-b", "master"], cwd=self.repo_dir, check=True, capture_output=True)
subprocess.run(["git", "config", "user.name", "Test User"], cwd=self.repo_dir, check=True)
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=self.repo_dir, check=True)
readme = os.path.join(self.repo_dir, "README.md")
with open(readme, "w", encoding="utf-8") as f:
f.write("# Test Repo\n")
subprocess.run(["git", "add", "README.md"], cwd=self.repo_dir, check=True, capture_output=True)
subprocess.run(["git", "commit", "-m", "initial commit"], cwd=self.repo_dir, check=True, capture_output=True)
rev_res = subprocess.run(["git", "rev-parse", "HEAD"], cwd=self.repo_dir, capture_output=True, text=True, check=True)
self.master_sha = rev_res.stdout.strip()
self.branches_dir = os.path.join(self.repo_dir, "branches")
os.makedirs(self.branches_dir, exist_ok=True)
self.lock_dir = os.path.join(self.tmp_dir, "locks")
os.makedirs(self.lock_dir, exist_ok=True)
self.journal_dir = os.path.join(self.tmp_dir, "journals")
os.makedirs(self.journal_dir, exist_ok=True)
os.environ["GITEA_BOOTSTRAP_JOURNAL_DIR"] = self.journal_dir
def tearDown(self):
os.environ.pop("GITEA_BOOTSTRAP_JOURNAL_DIR", None)
shutil.rmtree(self.tmp_dir, ignore_errors=True)
def test_bootstrap_success_path(self):
"""AC1/AC3/AC8: Successful bootstrap creates branch, worktree, registration, and lock proof."""
key = "test_key_success_1"
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
# assignment_id/lease_id omitted: optional unless verified live.
expected_base_sha=self.master_sha,
idempotency_key=key,
remote="prgs",
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertTrue(res.get("success"), f"Bootstrap failed: {res}")
self.assertFalse(res.get("replayed"))
self.assertEqual(res.get("issue_number"), 850)
self.assertEqual(res.get("base_sha"), self.master_sha)
self.assertIn("branches/fix-issue-850-native-mcp-bootstrap", res.get("worktree_path"))
# Verify worktree directory exists and is registered
worktree_path = res["worktree_path"]
self.assertTrue(os.path.isdir(worktree_path))
wt_list = subprocess.run(["git", "-C", self.repo_dir, "worktree", "list"], capture_output=True, text=True, check=True)
self.assertIn(worktree_path, wt_list.stdout)
# Verify phase journal written
journal = author_issue_bootstrap.load_phase_journal(key, journal_dir=self.lock_dir)
self.assertIsNotNone(journal)
self.assertTrue(journal.get("completed"))
self.assertEqual(journal.get("current_phase"), author_issue_bootstrap.PHASE_7_TRANSITION_COMPLETED)
def test_idempotent_replay(self):
"""Item 2: Replaying with identical key returns cached transition without duplicate creation."""
key = "test_key_idempotent_1"
res1 = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
idempotency_key=key,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertTrue(res1["success"], f"res1 failed: {res1}")
self.assertFalse(res1.get("replayed"))
# Second call
res2 = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
idempotency_key=key,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertTrue(res2["success"], f"res2 failed: {res2}")
self.assertTrue(res2.get("replayed"))
self.assertEqual(res1["worktree_path"], res2["worktree_path"])
def test_stale_concurrency_pin_refusal(self):
"""Item 3: Mismatched expected base SHA fails closed without silent rebasing."""
stale_sha = "0000000000000000000000000000000000000000"
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
expected_base_sha=stale_sha,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertFalse(res["success"])
self.assertEqual(res.get("reason_code"), "stale_concurrency_pin")
self.assertIn("exact_next_action", res)
def test_path_outside_branches_root_refusal(self):
"""Item 6: Worktree path outside branches/ root is refused."""
outside_path = os.path.join(self.tmp_dir, "outside_worktree")
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
worktree_path=outside_path,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertFalse(res["success"])
self.assertEqual(res.get("reason_code"), "path_outside_canonical_branches_root")
def test_preexisting_dirty_worktree_preservation(self):
"""Item 6: Preexisting dirty worktree fails closed and is NOT modified or cleaned."""
branch = "fix/issue-850-dirty-test"
wt_path = os.path.join(self.branches_dir, "fix-issue-850-dirty-test")
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", "-b", branch, wt_path], check=True, capture_output=True)
# Create dirty untracked file
dirty_file = os.path.join(wt_path, "dirty.txt")
with open(dirty_file, "w") as f:
f.write("dirty edits\n")
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
branch_name=branch,
worktree_path=wt_path,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertFalse(res["success"])
self.assertEqual(res.get("reason_code"), "preexisting_dirty_worktree")
# Prove dirty file is preserved byte-for-byte
self.assertTrue(os.path.exists(dirty_file))
with open(dirty_file, "r") as f:
self.assertEqual(f.read(), "dirty edits\n")
def test_compensating_recovery_on_failed_phase(self):
"""AC4/Item 4: Failure during transition rolls back ONLY newly created artifacts."""
key = "test_key_recovery_1"
# Simulate partial progress in journal
journal = {
"idempotency_key": key,
"issue_number": 850,
"branch_name": "fix/issue-850-recovery-test",
"worktree_path": os.path.join(self.branches_dir, "fix-issue-850-recovery-test"),
"artifacts_created": {
"branch_created": True,
"worktree_dir_created": True,
"worktree_registered": True,
"lock_created": False,
},
"failure_reason": "simulated lock failure",
"current_phase": author_issue_bootstrap.PHASE_5_REGISTRATION_VERIFIED,
"completed": False,
}
# Create the branch and worktree manually to simulate partial state
subprocess.run(["git", "-C", self.repo_dir, "branch", journal["branch_name"]], check=True, capture_output=True)
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", journal["worktree_path"], journal["branch_name"]], check=True, capture_output=True)
# Run compensating recovery
rec = author_issue_bootstrap.run_compensating_recovery(journal, self.repo_dir)
self.assertTrue(rec["executed"])
self.assertIn(f"worktree_path:{journal['worktree_path']}", rec["rolled_back"])
self.assertIn(f"branch:{journal['branch_name']}", rec["rolled_back"])
# Prove worktree directory and branch were rolled back
self.assertFalse(os.path.exists(journal["worktree_path"]))
branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", journal["branch_name"]], capture_output=True, text=True, check=False)
self.assertNotEqual(branch_check.returncode, 0)
def test_cross_process_concurrency(self):
"""Review #525 Finding 1: Genuine cross-process concurrency locking prevents corruption."""
import concurrent.futures
key = "test_concurrent_key_850"
args = (self.repo_dir, 850, key, self.lock_dir, self.journal_dir, self.master_sha)
with concurrent.futures.ProcessPoolExecutor(max_workers=2) as executor:
fut1 = executor.submit(_concurrent_bootstrap_worker, args)
fut2 = executor.submit(_concurrent_bootstrap_worker, args)
res1 = fut1.result(timeout=10)
res2 = fut2.result(timeout=10)
self.assertTrue(res1["success"], f"res1 failed: {res1}")
self.assertTrue(res2["success"], f"res2 failed: {res2}")
# One process performs creation, the other process receives idempotent replay
replayed_count = sum(1 for r in (res1, res2) if r.get("replayed"))
created_count = sum(1 for r in (res1, res2) if not r.get("replayed"))
self.assertEqual(replayed_count, 1)
self.assertEqual(created_count, 1)
self.assertEqual(res1["worktree_path"], res2["worktree_path"])
def test_interrupted_replay_preserves_artifacts_created_provenance(self):
"""Review #525 Finding 2: Replaying incomplete journal preserves creation provenance monotonically."""
key = "test_key_interrupted_replay_1"
branch = "fix/issue-850-interrupted-replay"
wt_path = os.path.join(self.branches_dir, "fix-issue-850-interrupted-replay")
# Simulate Phase 2/3 completion where branch and worktree directory were created by this transition
journal = {
"idempotency_key": key,
"issue_number": 850,
"branch_name": branch,
"worktree_path": wt_path,
"active_identity": "jcwalker3",
"active_profile": "prgs-author",
"remote": "prgs",
"org": "Scaled-Tech-Consulting",
"repo": "Gitea-Tools",
"phases": {
author_issue_bootstrap.PHASE_1_REQUEST_ACCEPTED: {"status": "completed"},
author_issue_bootstrap.PHASE_2_BRANCH_CONFIRMED: {"status": "completed", "created": True},
},
"artifacts_created": {
"branch_created": True,
"worktree_dir_created": True,
"worktree_registered": True,
"lock_created": False,
},
"current_phase": author_issue_bootstrap.PHASE_3_PATH_RESERVED,
"completed": False,
}
# Pre-create the branch and worktree on disk to simulate partial state after crash
subprocess.run(["git", "-C", self.repo_dir, "branch", branch, self.master_sha], check=True, capture_output=True)
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", wt_path, branch], check=True, capture_output=True)
author_issue_bootstrap.save_phase_journal(journal, journal_dir=self.lock_dir)
# Now resume/replay the transition but simulate lock binding failure during Phase 6
with mock.patch("issue_lock_store.bind_session_lock", side_effect=RuntimeError("Lock failure test")):
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
branch_name=branch,
worktree_path=wt_path,
idempotency_key=key,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertFalse(res["success"])
self.assertEqual(res.get("reason_code"), "issue_lock_acquisition_failed")
# Verify that compensating recovery correctly deleted transition-created branch & worktree
# because creation provenance was preserved across replay (NOT downgraded to False!)
self.assertFalse(os.path.exists(wt_path))
branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", branch], capture_output=True, text=True, check=False)
self.assertNotEqual(branch_check.returncode, 0)
def test_transition_created_only_compensation(self):
"""Review #525 Finding 4: Preexisting branch is NOT deleted by compensation when only worktree was transition-created."""
key = "test_key_preexisting_branch_compensation"
preexisting_branch = "fix/issue-850-preexisting"
wt_path = os.path.join(self.branches_dir, "fix-issue-850-preexisting")
# Create branch BEFORE bootstrap (preexisting branch)
subprocess.run(["git", "-C", self.repo_dir, "branch", preexisting_branch, self.master_sha], check=True, capture_output=True)
# Call bootstrap with simulated failure during Phase 6 (lock binding)
with mock.patch("issue_lock_store.bind_session_lock", side_effect=RuntimeError("Simulated lock failure")):
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
branch_name=preexisting_branch,
worktree_path=wt_path,
idempotency_key=key,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertFalse(res["success"])
# Worktree dir was created by transition -> removed by compensation
self.assertFalse(os.path.exists(wt_path))
# Preexisting branch was NOT created by transition -> MUST BE PRESERVED!
branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", preexisting_branch], capture_output=True, text=True, check=False)
self.assertEqual(branch_check.returncode, 0, "Preexisting branch was deleted by mistake!")
def test_incompatible_idempotency_replay_refusal(self):
"""Review #525 Finding 4: Replaying key with incompatible parameters returns refusal."""
key = "test_key_incompatible_replay"
res1 = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
branch_name="fix/issue-850-param-a",
idempotency_key=key,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertTrue(res1["success"])
# Second call with different branch_name
res2 = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
branch_name="fix/issue-850-param-b",
idempotency_key=key,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertFalse(res2["success"])
self.assertEqual(res2.get("reason_code"), "incompatible_idempotency_replay")
def test_exact_next_action_satisfiable_via_mcp(self):
"""Review #525 Finding 4: exact_next_action provides satisfiable MCP actions, not shell commands."""
key = "test_key_next_action_mcp"
stale_sha = "0000000000000000000000000000000000000000"
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
expected_base_sha=stale_sha,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
next_action = res.get("exact_next_action", "")
self.assertNotIn("scripts/worktree-start", next_action)
self.assertNotIn("git worktree add", next_action)
self.assertNotIn("bash", next_action.lower())
def test_missing_owner_session_refusal(self):
"""Finding D: Missing owner_session context fails closed with typed refusal and zero mutation."""
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
owner_session=None,
lock_dir=self.lock_dir,
)
self.assertFalse(res["success"])
self.assertEqual(res.get("reason_code"), "missing_owner_session")
self.assertIn("exact_next_action", res)
def test_symlink_lock_file_refusal(self):
"""Finding C: BootstrapTransitionLock refuses to follow symlinks."""
key = "test_symlink_lock_key"
safe_key = "".join(c if c.isalnum() or c in ("-", "_", ".") else "_" for c in key)
lock_path = os.path.join(self.lock_dir, f"{safe_key}.lock")
target_file = os.path.join(self.tmp_dir, "fake_target")
with open(target_file, "w") as f:
f.write("target")
os.symlink(target_file, lock_path)
with self.assertRaises(RuntimeError) as ctx:
with author_issue_bootstrap.BootstrapTransitionLock(key, journal_dir=self.lock_dir):
pass
self.assertIn("symlink", str(ctx.exception).lower())
def test_lock_directory_escape_refusal(self):
"""Finding C: BootstrapTransitionLock refuses keys that escape lock directory."""
with mock.patch("os.path.abspath", return_value="/tmp/outside/evil_key.lock"):
with self.assertRaises(RuntimeError) as ctx:
author_issue_bootstrap.BootstrapTransitionLock("key", journal_dir=self.lock_dir)
self.assertIn("escapes", str(ctx.exception).lower())
def test_missing_active_identity_refusal(self):
"""F-5: Missing active_identity parameter fails closed."""
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
owner_session="session-test-1234",
active_identity=None,
active_profile="prgs-author",
lock_dir=self.lock_dir,
)
self.assertFalse(res["success"])
self.assertEqual(res.get("reason_code"), "missing_active_identity")
def test_missing_active_profile_refusal(self):
"""F-5: Missing active_profile parameter fails closed."""
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
owner_session="session-test-1234",
active_identity="jcwalker3",
active_profile=None,
lock_dir=self.lock_dir,
)
self.assertFalse(res["success"])
self.assertEqual(res.get("reason_code"), "missing_active_profile")
def test_dirty_worktree_preserved_during_recovery(self):
"""F-4: Compensating recovery does not delete dirty worktree."""
branch = "fix/issue-850-rec-dirty"
wt_path = os.path.join(self.branches_dir, "fix-issue-850-rec-dirty")
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", "-b", branch, wt_path], check=True, capture_output=True)
dirty_file = os.path.join(wt_path, "dirty.txt")
with open(dirty_file, "w") as f:
f.write("uncommitted work")
journal = {
"idempotency_key": "test_dirty_rec",
"issue_number": 850,
"branch_name": branch,
"worktree_path": wt_path,
"artifacts_created": {
"worktree_dir_created": True,
"worktree_registered": True,
},
"failure_reason": "test dirty recovery",
}
rec = author_issue_bootstrap.run_compensating_recovery(journal, self.repo_dir, journal_dir=self.lock_dir)
self.assertTrue(os.path.exists(wt_path))
self.assertIn(f"worktree_path_preserved_dirty:{wt_path}", rec["rolled_back"])
def test_branch_with_commits_preserved_during_recovery(self):
"""F-4: Compensating recovery does not delete branch with author commits."""
branch = "fix/issue-850-rec-commits"
subprocess.run(["git", "-C", self.repo_dir, "branch", branch, self.master_sha], check=True, capture_output=True)
# Add a commit on the branch
wt_path = os.path.join(self.branches_dir, "fix-issue-850-rec-commits")
subprocess.run(["git", "-C", self.repo_dir, "worktree", "add", wt_path, branch], check=True, capture_output=True)
cfile = os.path.join(wt_path, "commit.txt")
with open(cfile, "w") as f:
f.write("author commit")
subprocess.run(["git", "-C", wt_path, "add", "commit.txt"], check=True, capture_output=True)
subprocess.run(["git", "-C", wt_path, "commit", "-m", "author commit"], check=True, capture_output=True)
subprocess.run(["git", "-C", self.repo_dir, "worktree", "remove", "--force", wt_path], check=True, capture_output=True)
journal = {
"idempotency_key": "test_commits_rec",
"issue_number": 850,
"branch_name": branch,
"resolved_base_sha": self.master_sha,
"artifacts_created": {
"branch_created": True,
},
"failure_reason": "test commit branch recovery",
}
rec = author_issue_bootstrap.run_compensating_recovery(journal, self.repo_dir, journal_dir=self.lock_dir)
branch_check = subprocess.run(["git", "-C", self.repo_dir, "rev-parse", "--verify", branch], capture_output=True, text=True, check=False)
self.assertEqual(branch_check.returncode, 0, "Branch with commits was deleted!")
self.assertIn(f"branch_preserved_commits:{branch}", rec["rolled_back"])
def test_task_capability_map_integration(self):
"""Verify task_capability_map has bootstrap_author_issue_worktree configured correctly."""
self.assertEqual(task_capability_map.required_role("bootstrap_author_issue_worktree"), "author")
self.assertEqual(task_capability_map.required_permission("bootstrap_author_issue_worktree"), "gitea.branch.create")
self.assertTrue(task_capability_map.preflight_task_matches("work_issue", "bootstrap_author_issue_worktree"))
self.assertTrue(task_capability_map.preflight_task_matches("bootstrap_author_issue_worktree", "lock_issue"))
def test_unverified_assignment_lease_ids_fail_closed(self):
"""Review #531 Finding 4: fabricated assignment/lease IDs are refused."""
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
assignment_id="asn-fabricated",
lease_id="lease-fabricated",
expected_base_sha=self.master_sha,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertFalse(res["success"])
self.assertIn(
res.get("reason_code"),
{
"unknown_lease_id",
"assignment_lease_lookup_failed",
"incomplete_assignment_lease_ids",
},
)
def test_partial_assignment_lease_ids_fail_closed(self):
"""Review #531 Finding 4: one of assignment_id/lease_id alone is incomplete."""
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
assignment_id="asn-only",
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertFalse(res["success"])
self.assertEqual(res.get("reason_code"), "incomplete_assignment_lease_ids")
def test_stale_diverged_branch_is_not_accepted_via_merge_base(self):
"""Review #531 Finding 3: any common ancestor is not enough; require master ⊆ branch."""
branch = "fix/issue-850-stale-divergent"
# Create branch from current master, then advance master so branch lacks tip.
subprocess.run(
["git", "-C", self.repo_dir, "branch", branch, self.master_sha],
check=True,
capture_output=True,
)
# Make a new commit on master (orphan path so branch does not contain it).
marker = os.path.join(self.repo_dir, "master-advance.txt")
with open(marker, "w") as f:
f.write("advance master\n")
subprocess.run(["git", "-C", self.repo_dir, "add", "master-advance.txt"], check=True, capture_output=True)
subprocess.run(
["git", "-C", self.repo_dir, "commit", "-m", "advance master past branch"],
check=True,
capture_output=True,
)
new_master = subprocess.run(
["git", "-C", self.repo_dir, "rev-parse", "HEAD"],
capture_output=True,
text=True,
check=True,
).stdout.strip()
res = author_issue_bootstrap.bootstrap_author_issue_worktree(
issue_number=850,
canonical_repo_root=self.repo_dir,
branch_name=branch,
expected_base_sha=new_master,
lock_dir=self.lock_dir,
owner_session="session-test-1234",
)
self.assertFalse(res["success"])
self.assertEqual(res.get("reason_code"), "incompatible_existing_branch")
def test_compensating_recovery_attempts_lease_release(self):
"""Review #531 Finding 5: recovery invokes lease release when lease_id is present."""
from unittest import mock
journal = {
"idempotency_key": "test_lease_rec",
"issue_number": 850,
"owner_session": "session-test-1234",
"lease_id": "lease-abc",
"branch_name": "fix/issue-850-lease-rec",
"artifacts_created": {"lock_created": True},
"failure_reason": "simulated",
"completed": False,
}
with mock.patch.object(
author_issue_bootstrap.lease_lifecycle,
"release_lease",
return_value={"success": True},
) as rel, mock.patch.object(
author_issue_bootstrap.control_plane_db,
"ControlPlaneDB",
return_value=mock.Mock(),
):
rec = author_issue_bootstrap.run_compensating_recovery(
journal, self.repo_dir, journal_dir=self.lock_dir
)
self.assertTrue(rec["executed"])
rel.assert_called_once()
self.assertIn("lease:lease-abc", rec["rolled_back"])
class TestCanonicalRootNoStringSplit(unittest.TestCase):
def test_fallback_uses_commonpath_not_substring_split(self):
"""Review #531 Finding 2: no norm.split('/branches/') fallback."""
import inspect
import author_mutation_worktree as amw
src = inspect.getsource(amw.resolve_canonical_repo_root)
self.assertNotIn('split("/branches/")', src)
self.assertNotIn("split('/branches/')", src)
# Fallback recovers repo root from a nested branches worktree path.
with tempfile.TemporaryDirectory() as tmp:
repo = os.path.join(tmp, "repo")
wt = os.path.join(repo, "branches", "fix-issue-850-x")
os.makedirs(wt)
# git unavailable path: pass missing workspace so fallback is used.
root = amw.resolve_canonical_repo_root("/missing/path", wt)
self.assertEqual(root, os.path.realpath(repo))
if __name__ == "__main__":
unittest.main()
-22
View File
@@ -28,21 +28,6 @@ class TestPathUnderBranches(unittest.TestCase):
amw.is_path_under_branches("/repo/other-checkout", self.ROOT) amw.is_path_under_branches("/repo/other-checkout", self.ROOT)
) )
def test_unrelated_branches_dir_fails(self):
self.assertFalse(
amw.is_path_under_branches("/tmp/branches/evil", self.ROOT)
)
def test_prefix_confusion_fails(self):
self.assertFalse(
amw.is_path_under_branches(f"{self.ROOT}/branches-other/foo", self.ROOT)
)
def test_traversal_fails(self):
self.assertFalse(
amw.is_path_under_branches(f"{self.ROOT}/branches/../evil", self.ROOT)
)
class TestAssessAuthorMutationWorktree(unittest.TestCase): class TestAssessAuthorMutationWorktree(unittest.TestCase):
ROOT = "/repo/Gitea-Tools" ROOT = "/repo/Gitea-Tools"
@@ -86,13 +71,6 @@ class TestAssessAuthorMutationWorktree(unittest.TestCase):
self.assertTrue(result["proven"]) self.assertTrue(result["proven"])
self.assertFalse(result["block"]) self.assertFalse(result["block"])
def test_path_shaped_branches_ancestry_without_isdir(self):
"""Review #551: commonpath recovery must not require on-disk isdir."""
fake_wt = "/repo/Gitea-Tools/branches/issue-274"
root = amw.resolve_canonical_repo_root(fake_wt, fake_wt)
self.assertEqual(root, "/repo/Gitea-Tools")
self.assertNotIn('split("/branches/")', open(amw.__file__).read())
class TestPreflightIntegration(unittest.TestCase): class TestPreflightIntegration(unittest.TestCase):
def test_verify_preflight_blocks_control_checkout_with_test_porcelain(self): def test_verify_preflight_blocks_control_checkout_with_test_porcelain(self):
+1 -1
View File
@@ -36,7 +36,7 @@ class ControlPlaneDBTest(unittest.TestCase):
rows = dict(conn.execute("SELECT key, value FROM schema_meta").fetchall()) rows = dict(conn.execute("SELECT key, value FROM schema_meta").fetchall())
finally: finally:
conn.close() conn.close()
self.assertEqual(rows["schema_version"], "4") self.assertEqual(rows["schema_version"], "5")
self.assertIn("DB coordinates", rows["architecture"]) self.assertIn("DB coordinates", rows["architecture"])
self.assertIn("bridge", rows["architecture"].lower()) self.assertIn("bridge", rows["architecture"].lower())
@@ -139,8 +139,6 @@ EXPECTED_ROLE_EXCLUSIVE_TASKS = frozenset(
"gitea_release_merger_pr_lease", "gitea_release_merger_pr_lease",
"create_branch", "create_branch",
"push_branch", "push_branch",
"bootstrap_author_issue_worktree",
"gitea_bootstrap_author_issue_worktree",
# #812 AC20: publishing an unpublished local head is author-only for the # #812 AC20: publishing an unpublished local head is author-only for the
# same reason every other push is — it writes a branch to the remote. # same reason every other push is — it writes a branch to the remote.
"publish_unpublished_branch", "publish_unpublished_branch",
+196
View File
@@ -0,0 +1,196 @@
"""Unit and integration tests for Model Usage & Performance Analytics (#651)."""
from __future__ import annotations
import os
import tempfile
import unittest
from starlette.testclient import TestClient
import control_plane_db
from webui.analytics_loader import (
ANALYTICS_SCHEMA_VERSION,
compute_percentile,
load_analytics,
record_usage,
)
from webui.app import create_app
from webui import console_redaction
class AnalyticsLoaderTest(unittest.TestCase):
def setUp(self) -> None:
self.temp_dir = tempfile.TemporaryDirectory()
self.db_path = os.path.join(self.temp_dir.name, "test_control_plane.sqlite3")
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
self.db = control_plane_db.ControlPlaneDB(db_path=self.db_path)
def tearDown(self) -> None:
self.temp_dir.cleanup()
def test_compute_percentile(self) -> None:
self.assertIsNone(compute_percentile([], 50.0))
self.assertEqual(compute_percentile([100], 50.0), 100.0)
# 2 elements: [100, 200]
self.assertEqual(compute_percentile([100, 200], 50.0), 150.0)
# 100 elements: 1..100
vals = list(range(1, 101))
self.assertEqual(compute_percentile(vals, 50.0), 50.5)
self.assertAlmostEqual(compute_percentile(vals, 90.0), 90.1)
def test_record_and_aggregate_usage(self) -> None:
# Record event 1 (complete data)
u1 = record_usage(
db_path=self.db_path,
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
role="author",
model="gemini-3.6-flash",
issue_number=651,
stage="implementation",
input_tokens=1000,
output_tokens=500,
estimated_cost_usd=0.0015,
latency_ms=200,
duration_ms=3000,
metadata={"secret_key": "secret123", "note": "token=secret123"},
)
self.assertGreater(u1, 0)
# Record event 2 (missing tokens and cost -> unknown)
u2 = record_usage(
db_path=self.db_path,
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
role="reviewer",
model="claude-3-5-sonnet",
pr_number=846,
stage="review",
latency_ms=500,
duration_ms=6000,
)
self.assertGreater(u2, u1)
snapshot = load_analytics(
db_path=self.db_path,
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
)
self.assertTrue(snapshot.ok)
self.assertEqual(snapshot.schema_version, ANALYTICS_SCHEMA_VERSION)
self.assertEqual(snapshot.total_events, 2)
# Verify overall summary
summary = snapshot.overall_summary
self.assertEqual(summary.total_events, 2)
self.assertEqual(summary.events_with_tokens, 1)
self.assertEqual(summary.total_tokens, 1500)
self.assertEqual(summary.events_with_cost, 1)
self.assertEqual(summary.estimated_cost_usd, 0.0015)
self.assertEqual(summary.events_with_latency, 2)
self.assertEqual(summary.latency_p50_ms, 350.0)
# Verify missing data handling (AC 3: not zero-fabricated)
reviewer_model = snapshot.by_model.get("claude-3-5-sonnet")
self.assertIsNotNone(reviewer_model)
self.assertEqual(reviewer_model.total_events, 1)
self.assertEqual(reviewer_model.events_with_tokens, 0)
self.assertIsNone(reviewer_model.total_tokens)
self.assertEqual(reviewer_model.display_tokens, "Unknown")
self.assertEqual(reviewer_model.events_with_cost, 0)
self.assertIsNone(reviewer_model.estimated_cost_usd)
self.assertEqual(reviewer_model.display_cost, "Unknown")
# Verify redaction (AC 4)
e1 = [e for e in snapshot.events if e.usage_id == u1][0]
self.assertIsNotNone(e1.metadata)
self.assertNotIn("secret123", e1.metadata)
self.assertIn("[REDACTED]", e1.metadata)
def test_missing_db_fail_soft(self) -> None:
invalid_path = "/nonexistent_path_dir/db.sqlite3"
snapshot = load_analytics(db_path=invalid_path)
self.assertFalse(snapshot.ok)
self.assertIn("control_plane_db_unavailable", snapshot.reason)
self.assertEqual(snapshot.overall_summary.display_tokens, "Unknown")
class AnalyticsWebUITest(unittest.TestCase):
def setUp(self) -> None:
self.temp_dir = tempfile.TemporaryDirectory()
self.db_path = os.path.join(self.temp_dir.name, "test_webui.sqlite3")
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
self.app = create_app()
self.client = TestClient(self.app)
record_usage(
db_path=self.db_path,
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
role="author",
model="gemini-3.6-flash",
issue_number=651,
stage="implementation",
input_tokens=2000,
output_tokens=1000,
estimated_cost_usd=0.003,
latency_ms=150,
duration_ms=2500,
)
def tearDown(self) -> None:
self.temp_dir.cleanup()
def test_analytics_html_route(self) -> None:
response = self.client.get("/analytics")
self.assertEqual(response.status_code, 200)
self.assertIn("Model Usage & Performance Analytics", response.text)
self.assertIn("gemini-3.6-flash", response.text)
self.assertIn("3,000", response.text)
def test_analytics_api_route(self) -> None:
response = self.client.get("/api/v1/analytics")
self.assertEqual(response.status_code, 200)
data = response.json()
self.assertTrue(data["ok"])
self.assertEqual(data["total_events"], 1)
self.assertIn("gemini-3.6-flash", data["by_model"])
def test_analytics_ingest_endpoint(self) -> None:
payload = {
"remote": "dadeschools",
"org": "Scaled-Tech-Consulting",
"repo": "Gitea-Tools",
"role": "reviewer",
"model": "claude-3-5-sonnet",
"pr_number": 846,
"stage": "review",
"input_tokens": 500,
"output_tokens": 100,
"latency_ms": 400,
"metadata": "Review note token=secret456",
}
response = self.client.post("/api/v1/analytics/usage", json=payload)
self.assertEqual(response.status_code, 201)
res_json = response.json()
self.assertTrue(res_json["ok"])
self.assertGreater(res_json["usage_id"], 0)
# Check that it appears in GET /api/v1/analytics
res2 = self.client.get("/api/v1/analytics")
self.assertEqual(res2.status_code, 200)
data2 = res2.json()
self.assertEqual(data2["total_events"], 2)
if __name__ == "__main__":
unittest.main()
-468
View File
@@ -1,468 +0,0 @@
"""Tests for the unified web-console inventory API (#636).
Covers the four cases the issue names — empty, populated, partial failure, and
the no-false-unowned invariant — plus redaction, collision detection, the
resource-split routes, and read-only guarantees against a real control-plane
database and real durable lock files.
"""
import json
import os
import sys
import tempfile
import unittest
from datetime import datetime, timedelta, timezone
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from starlette.testclient import TestClient
import control_plane_db
from webui.app import create_app
from webui import inventory
def _iso(dt: datetime) -> str:
return dt.astimezone(timezone.utc).isoformat()
def _write_lock(lock_dir: str, name: str, payload: dict) -> str:
path = os.path.join(lock_dir, name)
with open(path, "w", encoding="utf-8") as handle:
json.dump(payload, handle)
return path
def _live_lock_payload(
*,
issue_number: int,
branch: str,
worktree_path: str,
pid: int,
username: str = "jcwalker3",
profile: str = "prgs-author",
) -> dict:
now = datetime.now(timezone.utc)
future = now + timedelta(hours=2)
return {
"branch_name": branch,
"issue_number": issue_number,
"org": "Scaled-Tech-Consulting",
"repo": "Gitea-Tools",
"remote": "prgs",
"pid": pid,
"session_pid": pid,
"lock_generation": 1,
"worktree_path": worktree_path,
"claimant": {"username": username, "profile": profile},
"work_lease": {
"branch": branch,
"issue_number": issue_number,
"operation_type": "author_issue_work",
"created_at": _iso(now),
"expires_at": _iso(future),
"last_heartbeat_at": _iso(now),
"claimant": {"username": username, "profile": profile},
},
}
class _FixtureMixin(unittest.TestCase):
def setUp(self) -> None:
self._tmp = tempfile.TemporaryDirectory()
self.tmp = self._tmp.name
self.lock_dir = os.path.join(self.tmp, "locks")
os.makedirs(self.lock_dir, mode=0o700)
self.db_path = os.path.join(self.tmp, "control_plane.db")
self.addCleanup(self._tmp.cleanup)
def _seed_db(self) -> control_plane_db.ControlPlaneDB:
db = control_plane_db.ControlPlaneDB(self.db_path)
db.upsert_session(
session_id="prgs-author-1",
role="author",
profile="prgs-author",
namespace="gitea-author",
pid=os.getpid(),
)
db.upsert_work_item(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
kind="issue",
number=636,
)
db.assign_and_lease(
session_id="prgs-author-1",
role="author",
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
kind="issue",
number=636,
)
return db
class TestRedaction(unittest.TestCase):
def test_redact_path_collapses_home(self):
home = os.path.expanduser("~")
self.assertEqual(
inventory.redact_path(f"{home}/Development/Gitea-Tools"),
"~/Development/Gitea-Tools",
)
def test_redact_url_strips_userinfo_and_query(self):
self.assertEqual(
inventory.redact_url("https://user:[email protected]/api?token=abc"),
"https://gitea.prgs.cc/api",
)
def test_scrub_drops_credential_keys(self):
scrubbed = inventory.scrub(
{"token": "abc123", "api_key": "k", "profile": "prgs-author"}
)
self.assertEqual(scrubbed["token"], "[redacted]")
self.assertEqual(scrubbed["api_key"], "[redacted]")
self.assertEqual(scrubbed["profile"], "prgs-author")
def test_scrub_is_recursive_and_never_raises(self):
class Weird:
def __repr__(self) -> str:
return "weird-obj"
out = inventory.scrub({"nested": [{"password": "p", "obj": Weird()}]})
self.assertEqual(out["nested"][0]["password"], "[redacted]")
self.assertEqual(out["nested"][0]["obj"], "weird-obj")
class TestEmptyInventory(_FixtureMixin):
def test_empty_db_and_locks_degrade_without_raising(self):
# No DB file, no locks: sessions/leases unavailable, locks ok+empty.
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
sessions = snap.section("sessions")
leases = snap.section("leases")
locks = snap.section("locks")
self.assertEqual(sessions.status, inventory.STATUS_UNAVAILABLE)
self.assertEqual(leases.status, inventory.STATUS_UNAVAILABLE)
self.assertEqual(locks.status, inventory.STATUS_OK)
self.assertEqual(len(locks.items), 0)
# Ownership authority is incomplete because the DB is missing.
self.assertFalse(snap.ownership_authority_complete)
self.assertEqual(snap.collisions, ())
def test_empty_db_present_but_unpopulated(self):
control_plane_db.ControlPlaneDB(self.db_path) # creates schema, no rows
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
self.assertEqual(snap.section("sessions").status, inventory.STATUS_OK)
self.assertEqual(len(snap.section("sessions").items), 0)
self.assertEqual(snap.section("leases").status, inventory.STATUS_OK)
self.assertTrue(snap.ownership_authority_complete)
class TestPopulatedInventory(_FixtureMixin):
def test_sections_populated_and_correlated(self):
self._seed_db()
wt = f"{self.tmp}/branches/issue-636-inventory-api"
_write_lock(
self.lock_dir,
"prgs-Scaled-Tech-Consulting-Gitea-Tools-636.json",
_live_lock_payload(
issue_number=636,
branch="feat/issue-636-inventory-api",
worktree_path=wt,
pid=os.getpid(),
),
)
hygiene = _StubHygiene(
entries=(
_StubEntry(
rel_path="branches/issue-636-inventory-api",
branch="feat/issue-636-inventory-api",
classification="active-issue",
),
)
)
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: hygiene,
)
self.assertTrue(snap.ownership_authority_complete)
self.assertEqual(len(snap.section("sessions").items), 1)
self.assertEqual(len(snap.section("leases").items), 1)
self.assertEqual(len(snap.section("locks").items), 1)
self.assertEqual(len(snap.section("worktrees").items), 1)
# The lease, lock, and worktree for #636 correlate onto one row.
row = next(r for r in snap.correlations if r["issue_number"] == 636)
self.assertEqual(row["branch"], "feat/issue-636-inventory-api")
self.assertTrue(row["lock_live"])
self.assertEqual(row["worktree_classification"], "active-issue")
self.assertEqual(len(row["lease_ids"]), 1)
# No collision: live lock, live pid, matching worktree.
self.assertEqual(snap.collisions, ())
def test_serialized_payload_declares_field_authority(self):
self._seed_db()
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
payload = inventory.snapshot_to_dict(snap)
self.assertEqual(payload["api_version"], "v1")
self.assertEqual(payload["schema_version"], 1)
self.assertEqual(payload["field_authority"]["sessions"], "control_plane_db")
self.assertEqual(payload["field_authority"]["locks"], "filesystem")
self.assertIn("sessions", payload["sections"])
class TestPartialFailure(_FixtureMixin):
def test_worktree_scan_failure_degrades_only_that_section(self):
self._seed_db()
def _boom():
raise RuntimeError("git worktree list exploded")
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=_boom,
)
self.assertEqual(
snap.section("worktrees").status, inventory.STATUS_UNAVAILABLE
)
self.assertIn("exploded", snap.section("worktrees").reason)
# DB-backed sections still healthy.
self.assertEqual(snap.section("sessions").status, inventory.STATUS_OK)
self.assertIn("worktrees", snap.degraded_sections)
def test_degraded_ownership_suppresses_unowned_claim(self):
# DB absent → sessions/leases unavailable → ownership incomplete even
# though a lock exists and could look "unclaimed" by the DB alone.
_write_lock(
self.lock_dir,
"prgs-Scaled-Tech-Consulting-Gitea-Tools-636.json",
_live_lock_payload(
issue_number=636,
branch="feat/issue-636-inventory-api",
worktree_path=f"{self.tmp}/wt",
pid=os.getpid(),
),
)
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
self.assertFalse(snap.ownership_authority_complete)
payload = inventory.snapshot_to_dict(snap)
self.assertIn("may be treated as unowned", payload["ownership_note"])
class TestCollisionDetection(_FixtureMixin):
def test_live_lock_dead_owner_flagged(self):
_write_lock(
self.lock_dir,
"prgs-Scaled-Tech-Consulting-Gitea-Tools-700.json",
_live_lock_payload(
issue_number=700,
branch="feat/issue-700-x",
worktree_path=f"{self.tmp}/wt700",
pid=999_999_999, # not a running pid
),
)
control_plane_db.ControlPlaneDB(self.db_path) # empty but present
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
kinds = {c.kind for c in snap.collisions}
self.assertIn("live-lock-dead-owner", kinds)
# Also lock-without-worktree, since no worktree carries the branch.
self.assertIn("lock-without-worktree", kinds)
def test_duplicate_live_lock_on_same_branch(self):
for issue in (800, 801):
_write_lock(
self.lock_dir,
f"prgs-Scaled-Tech-Consulting-Gitea-Tools-{issue}.json",
_live_lock_payload(
issue_number=issue,
branch="feat/issue-800-shared",
worktree_path=f"{self.tmp}/wt{issue}",
pid=os.getpid(),
),
)
control_plane_db.ControlPlaneDB(self.db_path)
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
self.assertIn(
"duplicate-live-lock", {c.kind for c in snap.collisions}
)
def test_no_collision_when_sections_degraded(self):
# locks ok but worktrees unavailable → lock-without-worktree must NOT
# be asserted (a missing scan is not a missing worktree).
_write_lock(
self.lock_dir,
"prgs-Scaled-Tech-Consulting-Gitea-Tools-636.json",
_live_lock_payload(
issue_number=636,
branch="feat/issue-636-inventory-api",
worktree_path=f"{self.tmp}/wt",
pid=os.getpid(),
),
)
control_plane_db.ControlPlaneDB(self.db_path)
def _boom():
raise RuntimeError("scan down")
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=_boom,
)
self.assertNotIn(
"lock-without-worktree", {c.kind for c in snap.collisions}
)
class TestSectionInclude(_FixtureMixin):
def test_include_restricts_scanned_sections(self):
self._seed_db()
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
include=("locks",),
)
self.assertIsNotNone(snap.section("locks"))
self.assertIsNone(snap.section("sessions"))
self.assertIsNone(snap.section("worktrees"))
class TestRoutes(_FixtureMixin):
def setUp(self) -> None:
super().setUp()
# Point the loaders at the fixture DB and lock dir via env, and stub
# the worktree scan so the route does not shell out to git.
self._prev_env = {
"GITEA_CONTROL_PLANE_DB": os.environ.get("GITEA_CONTROL_PLANE_DB"),
"GITEA_ISSUE_LOCK_DIR": os.environ.get("GITEA_ISSUE_LOCK_DIR"),
"WEBUI_TEST_OFFLINE": os.environ.get("WEBUI_TEST_OFFLINE"),
}
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir
os.environ["WEBUI_TEST_OFFLINE"] = "1"
self._seed_db()
self.client = TestClient(create_app())
def tearDown(self) -> None:
for key, value in self._prev_env.items():
if value is None:
os.environ.pop(key, None)
else:
os.environ[key] = value
def test_inventory_route_returns_versioned_payload(self):
resp = self.client.get("/api/v1/inventory")
self.assertEqual(resp.status_code, 200)
body = resp.json()
self.assertEqual(body["api_version"], "v1")
self.assertIn("sessions", body["sections"])
self.assertIn("field_authority", body)
def test_section_route_restricts_and_labels(self):
resp = self.client.get("/api/v1/inventory/locks")
self.assertEqual(resp.status_code, 200)
body = resp.json()
self.assertEqual(body["requested_section"], "locks")
self.assertIn("locks", body["sections"])
self.assertNotIn("sessions", body["sections"])
def test_unknown_section_is_404(self):
resp = self.client.get("/api/v1/inventory/bogus")
self.assertEqual(resp.status_code, 404)
self.assertEqual(resp.json()["error"], "unknown_section")
def test_inventory_route_rejects_post(self):
resp = self.client.post("/api/v1/inventory")
self.assertEqual(resp.status_code, 405)
class TestReadOnly(_FixtureMixin):
def test_snapshot_does_not_create_db_file(self):
missing = os.path.join(self.tmp, "does-not-exist.db")
inventory.load_inventory_snapshot(
db_path=missing,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
self.assertFalse(os.path.exists(missing))
def test_readonly_connection_refuses_write(self):
self._seed_db()
conn = inventory._open_readonly(self.db_path)
try:
with self.assertRaises(Exception):
conn.execute(
"INSERT INTO sessions(session_id, role, started_at, "
"last_heartbeat_at, status) VALUES ('x','author',"
"'t','t','active')"
)
conn.commit()
finally:
conn.close()
# ── lightweight stand-ins for the #432 hygiene snapshot ──────────────────────
class _StubEntry:
def __init__(
self,
*,
rel_path: str,
branch: str | None = None,
classification: str = "stale-clean",
head_sha: str | None = "abc123",
dirty_tracked: int = 0,
dirty_untracked: bool = False,
detached: bool = False,
registered_worktree: bool = True,
notes: str = "",
) -> None:
self.rel_path = rel_path
self.folder_name = rel_path.split("/", 1)[-1]
self.branch = branch
self.classification = classification
self.head_sha = head_sha
self.dirty_tracked = dirty_tracked
self.dirty_untracked = dirty_untracked
self.detached = detached
self.registered_worktree = registered_worktree
self.notes = notes
class _StubHygiene:
def __init__(self, *, entries=(), scan_error=None) -> None:
self.entries = tuple(entries)
self.scan_error = scan_error
if __name__ == "__main__":
unittest.main()
-502
View File
@@ -1,502 +0,0 @@
"""Sanctioned restart / graceful reload control tests (#642).
Acceptance criteria under test:
1. The sanctioned restart path is implemented behind gates (capability,
confirmation, operator authorization, host hook).
2. Manual ``pkill`` stays forbidden and is classified as contamination.
3. Post-restart mutations require clean health/session proof.
4. Authorized restart preview, unauthorized deny, contamination classification.
5. No entry point exposes a raw kill.
"""
import json
import os
import tempfile
import unittest
import mcp_namespace_health
import runtime_recovery_guard
from task_capability_map import TASK_CAPABILITY_MAP
from webui import console_audit, console_authz, gated_actions, sanctioned_restart
NAMESPACE = "gitea-author"
# An operator-authorized, hook-configured host. Passed explicitly so no test
# depends on (or mutates) the real process environment.
READY_ENV = {
sanctioned_restart.RESTART_HOOK_ENV: "launchd:cc.prgs.gitea-author",
runtime_recovery_guard.OPERATOR_AUTHORIZATION_ENV: "ops-ticket-4821",
}
def admin(subject: str = "[email protected]") -> console_authz.Principal:
return console_authz.Principal(
subject=subject,
role=console_authz.ADMIN,
identity_source=console_authz.IDENTITY_ACCESS_PROXY,
authenticated=True,
)
def viewer() -> console_authz.Principal:
return console_authz.Principal(
subject="[email protected]",
role=console_authz.VIEWER,
identity_source=console_authz.IDENTITY_ACCESS_PROXY,
authenticated=True,
)
class TestCapabilityWiring(unittest.TestCase):
"""AC1: authority is declared, not invented by the console."""
def test_actions_resolve_through_the_capability_map(self):
for action_id in (
sanctioned_restart.ACTION_RESTART_NAMESPACE,
sanctioned_restart.ACTION_RELOAD_NAMESPACE,
):
with self.subTest(action=action_id):
action = console_authz.get_action(action_id)
self.assertIsNotNone(action)
self.assertIn(action.task_key, TASK_CAPABILITY_MAP)
self.assertEqual(
action.mcp_permission,
TASK_CAPABILITY_MAP[action.task_key]["permission"],
)
def test_restart_permission_is_not_a_gitea_operation(self):
"""No configured Gitea profile should satisfy a host restart."""
permission = TASK_CAPABILITY_MAP["restart_namespace"]["permission"]
self.assertFalse(permission.startswith("gitea."))
def test_restart_is_destructive_dual_control_break_glass(self):
action = console_authz.get_action(
sanctioned_restart.ACTION_RESTART_NAMESPACE
)
self.assertEqual(action.action_class, console_authz.CLASS_DESTRUCTIVE)
self.assertEqual(action.minimum_role, console_authz.ADMIN)
self.assertTrue(action.dual_control)
self.assertTrue(action.break_glass)
self.assertTrue(action.requires_confirmation)
def test_reload_is_privileged_but_not_destructive(self):
action = console_authz.get_action(
sanctioned_restart.ACTION_RELOAD_NAMESPACE
)
self.assertEqual(action.action_class, console_authz.CLASS_PRIVILEGED)
self.assertTrue(action.requires_confirmation)
class TestPreview(unittest.TestCase):
"""AC4: an authorized preview renders the plan without executing it."""
def test_preview_lists_the_mutation_ledger(self):
preview = sanctioned_restart.build_restart_preview(
NAMESPACE, principal=admin(), env=READY_ENV
)
steps = [entry["step"] for entry in preview["mutation_ledger"]]
self.assertEqual(
steps, ["quiesce", "host_restart_hook", "health_recheck", "audit"]
)
self.assertTrue(preview["scope_valid"])
self.assertTrue(preview["post_restart_verification_required"])
def test_reload_preview_drains_instead_of_restarting(self):
preview = sanctioned_restart.build_restart_preview(
NAMESPACE, sanctioned_restart.MODE_RELOAD,
principal=admin(), env=READY_ENV,
)
steps = [entry["step"] for entry in preview["mutation_ledger"]]
self.assertIn("host_graceful_reload", steps)
self.assertNotIn("host_restart_hook", steps)
def test_preview_never_enables_execution(self):
preview = sanctioned_restart.build_restart_preview(
NAMESPACE, principal=admin(), env=READY_ENV
)
self.assertFalse(preview["execution_enabled"])
self.assertFalse(preview["authorization"]["execution_enabled"])
def test_confirmation_phrase_binds_the_namespace(self):
self.assertTrue(
sanctioned_restart.confirmation_matches(
NAMESPACE, sanctioned_restart.MODE_RESTART,
"restart gitea-author",
)
)
# A phrase typed for one namespace must not authorize another.
self.assertFalse(
sanctioned_restart.confirmation_matches(
"gitea-merger", sanctioned_restart.MODE_RESTART,
"restart gitea-author",
)
)
class TestGates(unittest.TestCase):
"""AC1/AC4: every gate denies with a stable reason code."""
def _assess(self, **kwargs):
params = {
"principal": admin(),
"confirmation": f"restart {NAMESPACE}",
"env": READY_ENV,
}
params.update(kwargs)
namespace = params.pop("namespace", NAMESPACE)
mode = params.pop("mode", sanctioned_restart.MODE_RESTART)
return sanctioned_restart.assess_restart_request(
namespace, mode, **params
)
def test_authorized_confirmed_request_passes_every_gate(self):
result = self._assess()
self.assertTrue(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.ALLOW_HOST_ACTION_REQUIRED
)
def test_passing_every_gate_is_not_an_execution_grant(self):
"""An allowed request still never lets the console touch the process."""
result = self._assess()
self.assertTrue(result["allowed"])
self.assertFalse(result["execution_enabled"])
self.assertFalse(result["console_executes"])
def test_unauthorized_principal_is_denied(self):
result = self._assess(principal=viewer())
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.DENY_UNAUTHORIZED
)
def test_anonymous_principal_is_denied(self):
result = self._assess(principal=None)
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.DENY_UNAUTHORIZED
)
def test_missing_confirmation_is_denied(self):
result = self._assess(confirmation=None)
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.DENY_CONFIRMATION_MISSING
)
def test_confirmation_for_another_namespace_is_denied(self):
result = self._assess(confirmation="restart gitea-merger")
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.DENY_CONFIRMATION_MISMATCH
)
def test_missing_operator_authorization_is_denied(self):
env = {sanctioned_restart.RESTART_HOOK_ENV: "launchd:cc.prgs.author"}
result = self._assess(env=env)
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"],
sanctioned_restart.DENY_OPERATOR_AUTHORIZATION,
)
def test_missing_host_hook_is_denied_without_kill_fallback(self):
env = {
runtime_recovery_guard.OPERATOR_AUTHORIZATION_ENV: "ops-ticket-1",
}
result = self._assess(env=env)
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.DENY_HOOK_NOT_CONFIGURED
)
def test_fleet_scope_is_refused(self):
for scope in ("all", "*", "fleet"):
with self.subTest(scope=scope):
result = self._assess(
namespace=scope, confirmation=f"restart {scope}"
)
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.DENY_FLEET_SCOPE
)
def test_unknown_namespace_is_refused(self):
result = self._assess(
namespace="gitea-nope", confirmation="restart gitea-nope"
)
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.DENY_UNKNOWN_NAMESPACE
)
def test_unknown_mode_is_refused(self):
result = self._assess(mode="obliterate")
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.DENY_UNKNOWN_MODE
)
def test_live_contamination_marker_blocks_restart(self):
marker = runtime_recovery_guard.build_contamination_record(
reason_class=runtime_recovery_guard.REASON_MANUAL_DAEMON_KILL,
command_redacted="pkill -f mcp_server.py",
)
result = self._assess(contamination_marker=marker)
self.assertFalse(result["allowed"])
self.assertEqual(
result["reason_code"], sanctioned_restart.DENY_CONTAMINATED_RUNTIME
)
def test_reconciler_cleared_marker_no_longer_blocks(self):
marker = runtime_recovery_guard.build_contamination_record(
reason_class=runtime_recovery_guard.REASON_MANUAL_DAEMON_KILL,
command_redacted="pkill -f mcp_server.py",
)
marker = dict(marker, cleared_by_reconciler=True)
result = self._assess(contamination_marker=marker)
self.assertTrue(result["allowed"])
class TestExecutionNeverKills(unittest.TestCase):
"""AC5: no path exposes or runs a raw process kill."""
def test_authorized_execution_defers_to_the_host_supervisor(self):
result = sanctioned_restart.execute_restart(
NAMESPACE,
principal=admin(),
confirmation=f"restart {NAMESPACE}",
env=READY_ENV,
)
self.assertTrue(result["allowed"])
self.assertFalse(result["success"])
self.assertFalse(result["process_kill_executed"])
self.assertEqual(
result["outcome"], sanctioned_restart.ALLOW_HOST_ACTION_REQUIRED
)
def test_denied_execution_reports_the_refusing_gate(self):
result = sanctioned_restart.execute_restart(
NAMESPACE, principal=viewer(), confirmation=f"restart {NAMESPACE}",
env=READY_ENV,
)
self.assertFalse(result["allowed"])
self.assertEqual(
result["outcome"], sanctioned_restart.DENY_UNAUTHORIZED
)
self.assertFalse(result["process_kill_executed"])
def test_module_never_spawns_a_process(self):
path = os.path.join(
os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
"webui", "sanctioned_restart.py",
)
with open(path, encoding="utf-8") as handle:
source = handle.read()
for forbidden in (
"import subprocess", "import signal", "os.kill", "os.system",
"popen",
):
with self.subTest(forbidden=forbidden):
self.assertNotIn(forbidden, source.lower())
def test_no_surface_returns_a_kill_command(self):
payloads = [
sanctioned_restart.build_restart_preview(
NAMESPACE, principal=admin(), env=READY_ENV
),
sanctioned_restart.restart_policy(),
sanctioned_restart.execute_restart(
NAMESPACE, principal=admin(),
confirmation=f"restart {NAMESPACE}", env=READY_ENV,
),
]
for payload in payloads:
rendered = json.dumps(payload, default=str).lower()
self.assertNotIn("kill -9", rendered)
self.assertNotIn("pkill -f", rendered)
def test_policy_declares_no_raw_kill_and_no_silent_restart(self):
policy = sanctioned_restart.restart_policy()
self.assertFalse(policy["raw_kill_exposed"])
self.assertFalse(policy["console_executes_process_kill"])
self.assertFalse(policy["fleet_scope_permitted"])
self.assertFalse(policy["silent_auto_restart_permitted"])
self.assertTrue(policy["audit_required"])
class TestContaminationClassification(unittest.TestCase):
"""AC2: manual pkill is contamination, and it blocks clean claims."""
def test_manual_daemon_pkill_is_contamination(self):
result = sanctioned_restart.classify_restart_command(
"pkill -f mcp_server.py"
)
self.assertTrue(result["contamination"])
self.assertFalse(result["clean_claim_allowed"])
self.assertIsNotNone(result["contamination_marker"])
self.assertEqual(
result["sanctioned_alternative"],
sanctioned_restart.ACTION_RESTART_NAMESPACE,
)
def test_broad_process_kill_is_contamination(self):
result = sanctioned_restart.classify_restart_command("killall -9 Python")
self.assertTrue(result["contamination"])
self.assertFalse(result["clean_claim_allowed"])
def test_marker_names_the_sanctioned_alternative(self):
result = sanctioned_restart.classify_restart_command(
"pkill -f mcp_server.py"
)
marker = result["contamination_marker"]
self.assertIn(
sanctioned_restart.ACTION_RESTART_NAMESPACE, marker["detail"]
)
def test_benign_command_is_not_contamination(self):
result = sanctioned_restart.classify_restart_command("git status")
self.assertFalse(result["contamination"])
self.assertTrue(result["clean_claim_allowed"])
def test_no_command_is_not_contamination(self):
result = sanctioned_restart.classify_restart_command(None)
self.assertFalse(result["contamination"])
self.assertTrue(result["clean_claim_allowed"])
class TestPostRestartHealth(unittest.TestCase):
"""AC3: a clean post-restart claim needs live client-namespace proof."""
def test_live_client_probe_clears_the_session(self):
result = sanctioned_restart.verify_post_restart_health(
NAMESPACE,
probe_result={"success": True},
probe_source=mcp_namespace_health.PROBE_SOURCE_CLIENT,
registered_tools=["gitea_whoami"],
required_tool="gitea_whoami",
)
self.assertEqual(result["status"], sanctioned_restart.HEALTH_CLEAN)
self.assertTrue(result["clean_claim_allowed"])
self.assertTrue(result["mutations_allowed"])
def test_offline_probe_does_not_clear_the_session(self):
result = sanctioned_restart.verify_post_restart_health(
NAMESPACE,
probe_result={"success": True},
probe_source=mcp_namespace_health.PROBE_SOURCE_OFFLINE,
registered_tools=["gitea_whoami"],
required_tool="gitea_whoami",
)
self.assertFalse(result["clean_claim_allowed"])
self.assertFalse(result["mutations_allowed"])
def test_failed_probe_is_unhealthy(self):
result = sanctioned_restart.verify_post_restart_health(
NAMESPACE,
probe_result={"success": False, "error": "client is closing: EOF"},
probe_source=mcp_namespace_health.PROBE_SOURCE_CLIENT,
registered_tools=["gitea_whoami"],
required_tool="gitea_whoami",
)
self.assertEqual(result["status"], sanctioned_restart.HEALTH_UNHEALTHY)
self.assertFalse(result["clean_claim_allowed"])
def test_static_registration_alone_never_clears_the_session(self):
result = sanctioned_restart.verify_post_restart_health(
NAMESPACE,
registered_tools=["gitea_whoami"],
required_tool="gitea_whoami",
)
self.assertFalse(result["clean_claim_allowed"])
class TestAuditEmission(unittest.TestCase):
"""Every restart attempt is audited with actor, target, and result."""
def _run(self, principal, sink):
prior = os.environ.get(console_audit.AUDIT_LOG_ENV)
os.environ[console_audit.AUDIT_LOG_ENV] = sink
try:
return sanctioned_restart.execute_restart(
NAMESPACE,
principal=principal,
confirmation=f"restart {NAMESPACE}",
env=READY_ENV,
request_id="req-642",
)
finally:
if prior is None:
os.environ.pop(console_audit.AUDIT_LOG_ENV, None)
else:
os.environ[console_audit.AUDIT_LOG_ENV] = prior
def test_allowed_attempt_is_written_with_actor_and_target(self):
with tempfile.TemporaryDirectory() as tmp:
sink = os.path.join(tmp, "audit.jsonl")
result = self._run(admin(), sink)
self.assertTrue(result["audit"]["written"])
with open(sink, encoding="utf-8") as handle:
record = json.loads(handle.read().strip())
self.assertEqual(
record["action"], sanctioned_restart.ACTION_RESTART_NAMESPACE
)
self.assertEqual(record["target"]["namespace"], NAMESPACE)
self.assertEqual(record["target"]["mode"], "restart")
self.assertEqual(record["result"], console_audit.RESULT_ALLOWED)
self.assertEqual(record["actor"]["subject"], "[email protected]")
self.assertFalse(record["metadata"]["process_kill_executed"])
def test_denied_attempt_is_audited_too(self):
with tempfile.TemporaryDirectory() as tmp:
sink = os.path.join(tmp, "audit.jsonl")
self._run(viewer(), sink)
with open(sink, encoding="utf-8") as handle:
record = json.loads(handle.read().strip())
self.assertEqual(record["result"], console_audit.RESULT_DENIED)
self.assertEqual(
record["reason_code"], sanctioned_restart.DENY_UNAUTHORIZED
)
def test_restart_audit_uses_break_glass_retention(self):
action = console_authz.get_action(
sanctioned_restart.ACTION_RESTART_NAMESPACE
)
self.assertEqual(
console_audit.retention_class_for(action),
console_audit.RETENTION_BREAK_GLASS,
)
class TestRegistrySurface(unittest.TestCase):
"""AC5: the console surfaces the control, still disabled, with no kill."""
def test_registry_exposes_both_actions_disabled(self):
registry = gated_actions.load_action_registry()
for action_id in (
sanctioned_restart.ACTION_RESTART_NAMESPACE,
sanctioned_restart.ACTION_RELOAD_NAMESPACE,
):
with self.subTest(action=action_id):
action = registry.get(action_id)
self.assertIsNotNone(action)
self.assertFalse(action.enabled)
def test_registry_preview_names_the_namespace_target(self):
preview = gated_actions.preview_action(
sanctioned_restart.ACTION_RESTART_NAMESPACE, namespace=NAMESPACE
)
target = preview["mutation_ledger"][0]["target"]
self.assertIn(NAMESPACE, target)
self.assertFalse(preview["enabled"])
def test_registry_attempt_fails_closed(self):
result = gated_actions.attempt_action(
sanctioned_restart.ACTION_RESTART_NAMESPACE, namespace=NAMESPACE
)
self.assertFalse(result["success"])
if __name__ == "__main__":
unittest.main()
-345
View File
@@ -1,345 +0,0 @@
"""Tests for the system-health dashboard view (#639).
Covers the acceptance criteria directly: the page renders the health DTO
fields (AC1), degraded dependencies are visible (AC2), stale runtime is warned
prominently and never rendered as mutation-safe (AC3), healthy and degraded
fixtures both render (AC4), and the shell carries a nav entry (AC5).
"""
import sys
import unittest
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from starlette.testclient import TestClient
from webui.app import create_app
from webui.deployment_boundary import scan_text_for_client_secrets
from webui.layout import render_page
from webui.nav import iter_nav_items
from webui.system_health import (
STATUS_DEGRADED,
STATUS_DOWN,
STATUS_OK,
STATUS_SKIPPED,
STATUS_UNPROVEN,
DependencyProbe,
StaleRuntime,
SystemHealthSnapshot,
VersionInfo,
)
from webui.system_health_views import render_system_health_page
DASHBOARD_PATH = "/system-health"
def _version(*, known: bool = True) -> VersionInfo:
return VersionInfo(
git_sha="1c455b6ec0f9cb761fe6248de68c17e061fb5ecd" if known else None,
git_describe="v0.4.1-12-g1c455b6" if known else None,
control_plane_schema_version=4 if known else None,
python_version="3.13.1",
known=known,
)
def _parity(*, stale: bool = False, determinable: bool = True) -> StaleRuntime:
if stale:
return StaleRuntime(
daemon_head="aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
checkout_head="bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
remote_head="cccccccccccccccccccccccccccccccccccccccc",
stale=True,
determinable=True,
mutation_safe=False,
reasons=("runtime, checkout, and remote commits disagree",),
)
if not determinable:
return StaleRuntime(
daemon_head=None,
checkout_head=None,
remote_head=None,
stale=False,
determinable=False,
mutation_safe=False,
reasons=("local checkout HEAD could not be read",),
)
return StaleRuntime(
daemon_head="1c455b6ec0f9cb761fe6248de68c17e061fb5ecd",
checkout_head="1c455b6ec0f9cb761fe6248de68c17e061fb5ecd",
remote_head="1c455b6ec0f9cb761fe6248de68c17e061fb5ecd",
stale=False,
determinable=True,
mutation_safe=True,
reasons=(),
)
def _snapshot(
*,
status: str = STATUS_OK,
ready: bool = True,
readiness_complete: bool = True,
readiness_reasons: tuple[str, ...] = (),
dependencies: tuple[DependencyProbe, ...] | None = None,
parity: StaleRuntime | None = None,
namespaces: tuple[dict, ...] = (),
probe_errors: tuple[str, ...] = (),
version_known: bool = True,
) -> SystemHealthSnapshot:
if dependencies is None:
dependencies = (
DependencyProbe(
name="control_plane_db",
kind="sqlite",
status=STATUS_OK,
detail="schema version 4",
required=True,
latency_ms=1.25,
metadata={"schema_version": 4},
),
)
return SystemHealthSnapshot(
status=status,
ready=ready,
readiness_complete=readiness_complete,
readiness_reasons=readiness_reasons,
service="mcp-control-plane-webui",
mode="read-only",
version=_version(known=version_known),
started_at="2026-07-23T19:50:47+00:00",
uptime_seconds=3661.5,
timestamp="2026-07-23T20:51:48+00:00",
deep_probes_requested=False,
dependencies=dependencies,
mcp_namespaces=namespaces,
stale_runtime=parity if parity is not None else _parity(),
probe_errors=probe_errors,
)
class TestHealthyRender(unittest.TestCase):
"""AC1 / AC4 — every health DTO field reaches the page."""
def setUp(self):
self.html = render_system_health_page(_snapshot())
def test_readiness_fields_render(self):
self.assertIn("System health", self.html)
self.assertIn("Ready", self.html)
self.assertIn("mcp-control-plane-webui", self.html)
self.assertIn("read-only", self.html)
self.assertIn("2026-07-23T20:51:48+00:00", self.html)
def test_version_and_uptime_render(self):
self.assertIn("1c455b6ec0f9cb761fe6248de68c17e061fb5ecd", self.html)
self.assertIn("v0.4.1-12-g1c455b6", self.html)
self.assertIn("3.13.1", self.html)
self.assertIn("3661.500s", self.html)
self.assertIn("1.02h", self.html)
def test_dependency_row_renders_with_latency(self):
self.assertIn("control_plane_db", self.html)
self.assertIn("sqlite", self.html)
self.assertIn("schema version 4", self.html)
self.assertIn("1.2 ms", self.html)
def test_healthy_page_shows_no_stale_warning(self):
self.assertNotIn("Stale runtime:", self.html)
self.assertNotIn("Staleness", self.html)
def test_unknown_version_is_labelled_not_faked(self):
html = render_system_health_page(_snapshot(version_known=False))
self.assertIn("unknown", html)
self.assertIn("unresolved", html)
class TestDegradedRender(unittest.TestCase):
"""AC2 — a degraded or unrun dependency is visible, not swallowed."""
def setUp(self):
self.deps = (
DependencyProbe(
name="control_plane_db",
kind="sqlite",
status=STATUS_OK,
detail="schema version 4",
required=True,
latency_ms=0.9,
),
DependencyProbe(
name="repository",
kind="git",
status=STATUS_DOWN,
detail="repository root is not a git checkout",
required=True,
latency_ms=4.0,
),
DependencyProbe(
name="gitea",
kind="http",
status=STATUS_SKIPPED,
detail="deep probe not requested",
required=False,
),
)
self.html = render_system_health_page(
_snapshot(
status=STATUS_DEGRADED,
ready=False,
readiness_complete=False,
readiness_reasons=("required dependency 'repository' is down",),
dependencies=self.deps,
)
)
def test_degraded_banner_names_the_dependency(self):
self.assertIn("Degraded dependencies:", self.html)
self.assertIn("repository", self.html)
def test_not_run_probe_is_reported_separately(self):
self.assertIn("Not probed:", self.html)
self.assertIn("gitea", self.html)
self.assertIn("not counted", self.html)
def test_not_ready_headline_and_reason(self):
self.assertIn("Not ready", self.html)
self.assertIn("required dependency &#x27;repository&#x27; is down", self.html)
def test_degraded_status_badge_present(self):
self.assertIn("badge-health-degraded", self.html)
self.assertIn("badge-health-down", self.html)
def test_ready_but_incomplete_is_not_shown_as_plain_ready(self):
html = render_system_health_page(
_snapshot(ready=True, readiness_complete=False)
)
self.assertIn("Ready (incomplete evidence)", html)
class TestStaleRuntimeWarning(unittest.TestCase):
"""AC3 — staleness is prominent and never claims mutation safety."""
def test_stale_runtime_warns_and_denies_mutation_safety(self):
html = render_system_health_page(_snapshot(parity=_parity(stale=True)))
self.assertIn("Stale runtime:", html)
self.assertIn("do not treat this runtime as mutation-safe", html)
self.assertIn("<tr><th>Mutation safe</th><td>False</td></tr>", html)
def test_indeterminate_parity_is_not_reported_safe(self):
html = render_system_health_page(
_snapshot(parity=_parity(determinable=False))
)
self.assertIn("Staleness", html)
self.assertIn("<tr><th>Mutation safe</th><td>False</td></tr>", html)
self.assertIn("<tr><th>Determinable</th><td>False</td></tr>", html)
def test_healthy_parity_reports_mutation_safe_true(self):
html = render_system_health_page(_snapshot())
self.assertIn("<tr><th>Mutation safe</th><td>True</td></tr>", html)
class TestNamespacesAndErrors(unittest.TestCase):
def test_unproven_namespace_rows_render(self):
html = render_system_health_page(
_snapshot(
namespaces=(
{
"namespace": "gitea-author",
"required_tool": "gitea_lock_issue",
"status": STATUS_UNPROVEN,
"ide_namespace_proven": False,
"reason": "the web console cannot invoke the IDE-managed MCP client",
},
)
)
)
self.assertIn("gitea-author", html)
self.assertIn("gitea_lock_issue", html)
self.assertIn("badge-health-unproven", html)
def test_no_namespaces_degrades_gracefully(self):
html = render_system_health_page(_snapshot(namespaces=()))
self.assertIn("No MCP namespaces are declared.", html)
def test_probe_errors_render_when_present(self):
html = render_system_health_page(
_snapshot(probe_errors=("probe raised: disk offline",))
)
self.assertIn("Probe errors", html)
self.assertIn("disk offline", html)
def test_probe_error_card_absent_when_clean(self):
self.assertNotIn("Probe errors", render_system_health_page(_snapshot()))
class TestReadOnlyAndRedaction(unittest.TestCase):
def test_no_restart_or_kill_controls(self):
html = render_system_health_page(_snapshot())
self.assertNotIn("<button", html)
self.assertNotIn("<form", html)
self.assertNotIn("pkill", html)
self.assertIn("read-only", html)
def test_recovery_points_at_sanctioned_path(self):
html = render_system_health_page(_snapshot())
self.assertIn("Reconnect the MCP client", html)
self.assertIn("Never kill the daemon process manually", html)
def test_secret_shaped_detail_is_redacted(self):
leaky = DependencyProbe(
name="gitea",
kind="http",
status=STATUS_DOWN,
detail="auth failed for token=ghp_ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789",
required=False,
latency_ms=12.0,
)
html = render_system_health_page(_snapshot(dependencies=(leaky,)))
self.assertNotIn("ghp_ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789", html)
def test_html_in_detail_is_escaped(self):
hostile = DependencyProbe(
name="repository",
kind="git",
status=STATUS_DOWN,
detail="<script>alert(1)</script>",
required=True,
)
html = render_system_health_page(_snapshot(dependencies=(hostile,)))
self.assertNotIn("<script>", html)
self.assertIn("&lt;script&gt;", html)
class TestNavAndRoute(unittest.TestCase):
"""AC5 — the shell links the dashboard, and the route serves it."""
def setUp(self):
self.client = TestClient(create_app())
def test_nav_contains_system_health(self):
self.assertIn(
(DASHBOARD_PATH, "System health"),
[(item.href, item.label) for item in iter_nav_items()],
)
def test_rendered_shell_links_dashboard(self):
page = render_page(title="Home", body_html="<p>x</p>")
self.assertIn(f'href="{DASHBOARD_PATH}"', page)
def test_route_renders_dashboard(self):
response = self.client.get(DASHBOARD_PATH)
self.assertEqual(response.status_code, 200)
self.assertIn("System health", response.text)
self.assertIn("Stale-runtime parity", response.text)
def test_route_is_read_only(self):
self.assertEqual(self.client.post(DASHBOARD_PATH).status_code, 405)
def test_live_page_leaks_no_client_secret(self):
findings = scan_text_for_client_secrets(self.client.get(DASHBOARD_PATH).text)
self.assertEqual(findings, [])
if __name__ == "__main__": # pragma: no cover
unittest.main()
+1 -3
View File
@@ -35,10 +35,8 @@ class TestCanonicalRepoRoot(unittest.TestCase):
self.assertEqual(root, CONTROL_ROOT) self.assertEqual(root, CONTROL_ROOT)
def test_falls_back_when_git_unavailable(self): def test_falls_back_when_git_unavailable(self):
# When fallback is a path under <repo>/branches/<worktree>, recover
# <repo> via commonpath ancestry (never string-split on "/branches/").
root = amw.resolve_canonical_repo_root("/missing/path", MCP_PROCESS_ROOT) root = amw.resolve_canonical_repo_root("/missing/path", MCP_PROCESS_ROOT)
self.assertEqual(root, os.path.realpath(CONTROL_ROOT)) self.assertEqual(root, os.path.realpath(MCP_PROCESS_ROOT))
class TestWorkspaceRepoMembership(unittest.TestCase): class TestWorkspaceRepoMembership(unittest.TestCase):
+428
View File
@@ -0,0 +1,428 @@
"""Model usage, token cost, latency, and workflow-performance analytics (#651, Phase 4).
Ingests session instrumentation metrics, aggregates usage/cost/latency percentiles
by project, role, model, issue/PR, and stage, enforcing secret redaction and
explicitly rendering missing metrics as "Unknown" without zero-fabrication.
"""
from __future__ import annotations
import math
from dataclasses import asdict, dataclass
from typing import Any, Sequence
import control_plane_db
from webui import console_redaction
ANALYTICS_SCHEMA_VERSION = 1
@dataclass(frozen=True)
class UsageEvent:
usage_id: int
session_id: str | None
remote: str
org: str
repo: str
project_id: str | None
role: str
model: str
issue_number: int | None
pr_number: int | None
stage: str
input_tokens: int | None
output_tokens: int | None
total_tokens: int | None
estimated_cost_usd: float | None
latency_ms: int | None
duration_ms: int | None
status: str
metadata: str | None
created_at: str
def to_dict(self) -> dict[str, Any]:
d = asdict(self)
if d["metadata"]:
d["metadata"] = console_redaction.redact_text(d["metadata"])
return d
@dataclass(frozen=True)
class GroupMetrics:
name: str
total_events: int
events_with_tokens: int
input_tokens: int | None
output_tokens: int | None
total_tokens: int | None
events_with_cost: int
estimated_cost_usd: float | None
events_with_latency: int
latency_p50_ms: float | None
latency_p90_ms: float | None
latency_p95_ms: float | None
latency_p99_ms: float | None
latency_avg_ms: float | None
events_with_duration: int
duration_avg_ms: float | None
display_tokens: str
display_cost: str
display_latency_p50: str
display_latency_p90: str
display_duration_avg: str
def to_dict(self) -> dict[str, Any]:
return asdict(self)
@dataclass(frozen=True)
class AnalyticsSnapshot:
ok: bool
reason: str
schema_version: int
remote: str
org: str
repo: str
total_events: int
overall_summary: GroupMetrics
by_project: dict[str, GroupMetrics]
by_role: dict[str, GroupMetrics]
by_model: dict[str, GroupMetrics]
by_work_item: dict[str, GroupMetrics]
by_stage: dict[str, GroupMetrics]
events: tuple[UsageEvent, ...]
def to_dict(self) -> dict[str, Any]:
return {
"ok": self.ok,
"reason": self.reason,
"schema_version": self.schema_version,
"remote": self.remote,
"org": self.org,
"repo": self.repo,
"total_events": self.total_events,
"overall_summary": self.overall_summary.to_dict(),
"by_project": {k: v.to_dict() for k, v in self.by_project.items()},
"by_role": {k: v.to_dict() for k, v in self.by_role.items()},
"by_model": {k: v.to_dict() for k, v in self.by_model.items()},
"by_work_item": {k: v.to_dict() for k, v in self.by_work_item.items()},
"by_stage": {k: v.to_dict() for k, v in self.by_stage.items()},
"events": [e.to_dict() for e in self.events],
}
def compute_percentile(values: Sequence[float | int], percentile: float) -> float | None:
if not values:
return None
sorted_vals = sorted(values)
n = len(sorted_vals)
if n == 1:
return float(sorted_vals[0])
k = (n - 1) * (percentile / 100.0)
f = math.floor(k)
c = math.ceil(k)
if f == c:
return float(sorted_vals[int(f)])
d0 = sorted_vals[int(f)] * (c - k)
d1 = sorted_vals[int(c)] * (k - f)
return float(d0 + d1)
def aggregate_events(group_name: str, events: Sequence[UsageEvent]) -> GroupMetrics:
total_events = len(events)
if total_events == 0:
return GroupMetrics(
name=group_name,
total_events=0,
events_with_tokens=0,
input_tokens=None,
output_tokens=None,
total_tokens=None,
events_with_cost=0,
estimated_cost_usd=None,
events_with_latency=0,
latency_p50_ms=None,
latency_p90_ms=None,
latency_p95_ms=None,
latency_p99_ms=None,
latency_avg_ms=None,
events_with_duration=0,
duration_avg_ms=None,
display_tokens="Unknown",
display_cost="Unknown",
display_latency_p50="Unknown",
display_latency_p90="Unknown",
display_duration_avg="Unknown",
)
token_events = [
e for e in events
if e.total_tokens is not None or e.input_tokens is not None or e.output_tokens is not None
]
events_with_tokens = len(token_events)
if events_with_tokens > 0:
input_tokens = sum(e.input_tokens or 0 for e in token_events)
output_tokens = sum(e.output_tokens or 0 for e in token_events)
total_tokens = sum(
e.total_tokens if e.total_tokens is not None else ((e.input_tokens or 0) + (e.output_tokens or 0))
for e in token_events
)
display_tokens = f"{total_tokens:,}"
else:
input_tokens = None
output_tokens = None
total_tokens = None
display_tokens = "Unknown"
cost_events = [e for e in events if e.estimated_cost_usd is not None]
events_with_cost = len(cost_events)
if events_with_cost > 0:
estimated_cost_usd = round(sum(e.estimated_cost_usd for e in cost_events), 6)
display_cost = f"${estimated_cost_usd:.4f}"
else:
estimated_cost_usd = None
display_cost = "Unknown"
latency_vals = [e.latency_ms for e in events if e.latency_ms is not None]
events_with_latency = len(latency_vals)
if events_with_latency > 0:
latency_p50_ms = compute_percentile(latency_vals, 50.0)
latency_p90_ms = compute_percentile(latency_vals, 90.0)
latency_p95_ms = compute_percentile(latency_vals, 95.0)
latency_p99_ms = compute_percentile(latency_vals, 99.0)
latency_avg_ms = round(sum(latency_vals) / events_with_latency, 2)
display_latency_p50 = f"{round(latency_p50_ms, 1)} ms" if latency_p50_ms is not None else "Unknown"
display_latency_p90 = f"{round(latency_p90_ms, 1)} ms" if latency_p90_ms is not None else "Unknown"
else:
latency_p50_ms = None
latency_p90_ms = None
latency_p95_ms = None
latency_p99_ms = None
latency_avg_ms = None
display_latency_p50 = "Unknown"
display_latency_p90 = "Unknown"
duration_vals = [e.duration_ms for e in events if e.duration_ms is not None]
events_with_duration = len(duration_vals)
if events_with_duration > 0:
duration_avg_ms = round(sum(duration_vals) / events_with_duration, 2)
display_duration_avg = f"{round(duration_avg_ms / 1000.0, 2)} s" if duration_avg_ms >= 1000 else f"{round(duration_avg_ms, 1)} ms"
else:
duration_avg_ms = None
display_duration_avg = "Unknown"
return GroupMetrics(
name=group_name,
total_events=total_events,
events_with_tokens=events_with_tokens,
input_tokens=input_tokens,
output_tokens=output_tokens,
total_tokens=total_tokens,
events_with_cost=events_with_cost,
estimated_cost_usd=estimated_cost_usd,
events_with_latency=events_with_latency,
latency_p50_ms=latency_p50_ms,
latency_p90_ms=latency_p90_ms,
latency_p95_ms=latency_p95_ms,
latency_p99_ms=latency_p99_ms,
latency_avg_ms=latency_avg_ms,
events_with_duration=events_with_duration,
duration_avg_ms=duration_avg_ms,
display_tokens=display_tokens,
display_cost=display_cost,
display_latency_p50=display_latency_p50,
display_latency_p90=display_latency_p90,
display_duration_avg=display_duration_avg,
)
def record_usage(
*,
db_path: str | None = None,
session_id: str | None = None,
remote: str = "dadeschools",
org: str = "",
repo: str = "",
project_id: str | None = None,
role: str = "unknown",
model: str = "unknown",
issue_number: int | None = None,
pr_number: int | None = None,
stage: str = "unknown",
input_tokens: int | None = None,
output_tokens: int | None = None,
total_tokens: int | None = None,
estimated_cost_usd: float | None = None,
latency_ms: int | None = None,
duration_ms: int | None = None,
status: str = "success",
metadata: str | dict[str, Any] | None = None,
created_at: str | None = None,
) -> int:
"""Ingest/record a single usage event with optional metrics."""
db = control_plane_db.ControlPlaneDB(db_path=db_path)
return db.record_usage_event(
session_id=session_id,
remote=remote,
org=org,
repo=repo,
project_id=project_id,
role=role,
model=model,
issue_number=issue_number,
pr_number=pr_number,
stage=stage,
input_tokens=input_tokens,
output_tokens=output_tokens,
total_tokens=total_tokens,
estimated_cost_usd=estimated_cost_usd,
latency_ms=latency_ms,
duration_ms=duration_ms,
status=status,
metadata=metadata,
created_at=created_at,
)
def load_analytics(
*,
db_path: str | None = None,
remote: str | None = None,
org: str | None = None,
repo: str | None = None,
project_id: str | None = None,
role: str | None = None,
model: str | None = None,
stage: str | None = None,
issue_number: int | None = None,
pr_number: int | None = None,
limit: int = 500,
) -> AnalyticsSnapshot:
"""Load analytics snapshot aggregated by project, role, model, issue/PR, and stage."""
remote_filter = (remote or "").strip() or None
org_filter = (org or "").strip() or None
repo_filter = (repo or "").strip() or None
role_filter = (role or "").strip() or None
model_filter = (model or "").strip() or None
stage_filter = (stage or "").strip() or None
try:
db = control_plane_db.ControlPlaneDB(db_path=db_path)
rows = db.query_usage_events(
remote=remote_filter,
org=org_filter,
repo=repo_filter,
project_id=project_id,
role=role_filter,
model=model_filter,
stage=stage_filter,
issue_number=issue_number,
pr_number=pr_number,
limit=limit,
)
except Exception as exc:
empty_summary = aggregate_events("Overall", [])
return AnalyticsSnapshot(
ok=False,
reason=f"control_plane_db_unavailable: {exc}",
schema_version=ANALYTICS_SCHEMA_VERSION,
remote=remote,
org=org,
repo=repo,
total_events=0,
overall_summary=empty_summary,
by_project={},
by_role={},
by_model={},
by_work_item={},
by_stage={},
events=(),
)
parsed_events: list[UsageEvent] = []
for r in rows:
meta = console_redaction.redact_text(r.get("metadata")) if r.get("metadata") else None
parsed_events.append(
UsageEvent(
usage_id=r["usage_id"],
session_id=r.get("session_id"),
remote=r.get("remote", remote),
org=r.get("org", org),
repo=r.get("repo", repo),
project_id=r.get("project_id"),
role=r.get("role", "unknown"),
model=r.get("model", "unknown"),
issue_number=r.get("issue_number"),
pr_number=r.get("pr_number"),
stage=r.get("stage", "unknown"),
input_tokens=r.get("input_tokens"),
output_tokens=r.get("output_tokens"),
total_tokens=r.get("total_tokens"),
estimated_cost_usd=r.get("estimated_cost_usd"),
latency_ms=r.get("latency_ms"),
duration_ms=r.get("duration_ms"),
status=r.get("status", "success"),
metadata=meta,
created_at=r.get("created_at", ""),
)
)
overall_summary = aggregate_events("Overall", parsed_events)
# Group by project
groups_by_project: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
key = e.project_id or (f"{e.org}/{e.repo}" if e.org and e.repo else "default")
groups_by_project.setdefault(key, []).append(e)
by_project = {k: aggregate_events(k, v) for k, v in groups_by_project.items()}
# Group by role
groups_by_role: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
groups_by_role.setdefault(e.role, []).append(e)
by_role = {k: aggregate_events(k, v) for k, v in groups_by_role.items()}
# Group by model
groups_by_model: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
groups_by_model.setdefault(e.model, []).append(e)
by_model = {k: aggregate_events(k, v) for k, v in groups_by_model.items()}
# Group by work item
groups_by_work_item: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
if e.issue_number:
key = f"issue #{e.issue_number}"
elif e.pr_number:
key = f"pr #{e.pr_number}"
else:
key = "unlinked"
groups_by_work_item.setdefault(key, []).append(e)
by_work_item = {k: aggregate_events(k, v) for k, v in groups_by_work_item.items()}
# Group by stage
groups_by_stage: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
groups_by_stage.setdefault(e.stage, []).append(e)
by_stage = {k: aggregate_events(k, v) for k, v in groups_by_stage.items()}
return AnalyticsSnapshot(
ok=True,
reason="ok",
schema_version=ANALYTICS_SCHEMA_VERSION,
remote=remote,
org=org,
repo=repo,
total_events=len(parsed_events),
overall_summary=overall_summary,
by_project=by_project,
by_role=by_role,
by_model=by_model,
by_work_item=by_work_item,
by_stage=by_stage,
events=tuple(parsed_events),
)
def snapshot_to_dict(snapshot: AnalyticsSnapshot) -> dict[str, Any]:
return snapshot.to_dict()
+201
View File
@@ -0,0 +1,201 @@
"""HTML views for the Model Usage & Performance Analytics console (#651)."""
from __future__ import annotations
from webui.analytics_loader import AnalyticsSnapshot, GroupMetrics, UsageEvent
from webui.layout import render_page
def _render_badge(text: str, badge_type: str = "muted") -> str:
return f'<span class="badge badge-{badge_type}">{text}</span>'
def _render_group_table(title: str, groups: dict[str, GroupMetrics], key_header: str = "Group") -> str:
if not groups:
return (
f"<h3>{title}</h3>"
'<div class="card"><p class="muted">No telemetry events recorded for this dimension.</p></div>'
)
rows = []
for key, g in sorted(groups.items(), key=lambda x: x[1].total_events, reverse=True):
cost_cell = (
f'<span class="accent">{g.display_cost}</span>'
if g.events_with_cost > 0
else _render_badge("Unknown")
)
tokens_cell = (
g.display_tokens
if g.events_with_tokens > 0
else _render_badge("Unknown")
)
lat_p50 = (
g.display_latency_p50
if g.events_with_latency > 0
else _render_badge("Unknown")
)
lat_p90 = (
g.display_latency_p90
if g.events_with_latency > 0
else _render_badge("Unknown")
)
dur_avg = (
g.display_duration_avg
if g.events_with_duration > 0
else _render_badge("Unknown")
)
rows.append(
"<tr>"
f"<td><strong>{key}</strong></td>"
f"<td>{g.total_events}</td>"
f"<td>{tokens_cell}</td>"
f"<td>{cost_cell}</td>"
f"<td>{lat_p50}</td>"
f"<td>{lat_p90}</td>"
f"<td>{dur_avg}</td>"
"</tr>"
)
rows_html = "".join(rows)
return f"""
<h3>{title}</h3>
<div class="card" style="overflow-x: auto;">
<table class="data-table">
<thead>
<tr>
<th>{key_header}</th>
<th>Events</th>
<th>Total Tokens</th>
<th>Est. Cost</th>
<th>Latency (p50)</th>
<th>Latency (p90)</th>
<th>Avg Stage Duration</th>
</tr>
</thead>
<tbody>
{rows_html}
</tbody>
</table>
</div>
"""
def _render_events_table(events: tuple[UsageEvent, ...]) -> str:
if not events:
return (
"<h3>Recent Usage & Instrumentation Events</h3>"
'<div class="card"><p class="muted">No individual telemetry events recorded yet. Opt-in instrumentation via session logging or POST /api/v1/analytics/usage.</p></div>'
)
rows = []
for e in list(events)[-50:]: # Display latest 50
work_item = f"issue #{e.issue_number}" if e.issue_number else (f"pr #{e.pr_number}" if e.pr_number else "unlinked")
tokens = f"{e.total_tokens:,}" if e.total_tokens is not None else _render_badge("Unknown")
cost = f"${e.estimated_cost_usd:.4f}" if e.estimated_cost_usd is not None else _render_badge("Unknown")
latency = f"{e.latency_ms} ms" if e.latency_ms is not None else _render_badge("Unknown")
duration = f"{e.duration_ms} ms" if e.duration_ms is not None else _render_badge("Unknown")
status_badge = _render_badge(e.status, "success" if e.status == "success" else "danger")
rows.append(
"<tr>"
f"<td>#{e.usage_id}</td>"
f"<td><small>{e.created_at}</small></td>"
f"<td><span class=\"badge\">{e.role}</span></td>"
f"<td><strong>{e.model}</strong></td>"
f"<td>{e.stage}</td>"
f"<td>{work_item}</td>"
f"<td>{tokens}</td>"
f"<td>{cost}</td>"
f"<td>{latency}</td>"
f"<td>{duration}</td>"
f"<td>{status_badge}</td>"
"</tr>"
)
rows_html = "".join(rows)
return f"""
<h3>Recent Telemetry Events</h3>
<div class="card" style="overflow-x: auto;">
<table class="data-table">
<thead>
<tr>
<th>ID</th>
<th>Timestamp</th>
<th>Role</th>
<th>Model</th>
<th>Stage</th>
<th>Work Item</th>
<th>Tokens</th>
<th>Cost</th>
<th>Latency</th>
<th>Duration</th>
<th>Status</th>
</tr>
</thead>
<tbody>
{rows_html}
</tbody>
</table>
</div>
"""
def render_analytics_page(snapshot: AnalyticsSnapshot) -> str:
"""Render the main Model Usage & Performance Analytics console page."""
summary = snapshot.overall_summary
kpi_tokens = summary.display_tokens if summary.events_with_tokens > 0 else _render_badge("Unknown")
kpi_cost = summary.display_cost if summary.events_with_cost > 0 else _render_badge("Unknown")
kpi_lat_p50 = summary.display_latency_p50 if summary.events_with_latency > 0 else _render_badge("Unknown")
kpi_dur_avg = summary.display_duration_avg if summary.events_with_duration > 0 else _render_badge("Unknown")
status_notice = ""
if not snapshot.ok:
status_notice = (
f'<div class="card warning-card"><strong>Degraded Data Source:</strong> {snapshot.reason}</div>'
)
body_html = f"""
<h2>Model Usage & Performance Analytics (Phase 4)</h2>
<p class="muted">
Durable console analytics for model usage, token cost, latency percentiles, and workflow-stage performance correlated to issues, PRs, and worker roles.
</p>
{status_notice}
<div class="notice-card" style="background: rgba(91, 159, 212, 0.1); border: 1px solid var(--border); padding: 0.75rem 1rem; border-radius: 6px; margin-bottom: 1.5rem;">
<small><strong>Note on telemetry fidelity:</strong> Missing data or untracked metrics are explicitly labeled as <em>Unknown</em>. No token costs or latency metrics are zero-fabricated.</small>
</div>
<div class="card-grid" style="display: grid; grid-template-columns: repeat(auto-fit, minmax(180px, 1fr)); gap: 1rem; margin-bottom: 1.5rem;">
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Total Events</span>
<h3 style="margin: 0.25rem 0 0 0;">{summary.total_events}</h3>
</div>
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Total Tokens</span>
<h3 style="margin: 0.25rem 0 0 0;">{kpi_tokens}</h3>
</div>
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Est. Token Cost</span>
<h3 style="margin: 0.25rem 0 0 0;">{kpi_cost}</h3>
</div>
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Latency (p50)</span>
<h3 style="margin: 0.25rem 0 0 0;">{kpi_lat_p50}</h3>
</div>
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Avg Stage Duration</span>
<h3 style="margin: 0.25rem 0 0 0;">{kpi_dur_avg}</h3>
</div>
</div>
{_render_group_table("Usage & Cost by Model", snapshot.by_model, "Model")}
{_render_group_table("Performance by Workflow Stage", snapshot.by_stage, "Stage")}
{_render_group_table("Usage & Cost by Role", snapshot.by_role, "Role")}
{_render_group_table("Work Item Analytics", snapshot.by_work_item, "Work Item")}
{_render_events_table(snapshot.events)}
"""
return render_page(title="Model Usage & Performance Analytics", body_html=body_html)
+76 -55
View File
@@ -46,19 +46,19 @@ from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as wo
from webui.worktree_views import render_worktrees_page from webui.worktree_views import render_worktrees_page
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
from webui.runtime_views import render_runtime_page from webui.runtime_views import render_runtime_page
from webui.inventory import (
SECTION_NAMES as _INVENTORY_SECTIONS,
load_inventory_snapshot,
snapshot_to_dict as inventory_snapshot_to_dict,
)
from webui.timeline import load_timeline, snapshot_to_dict as timeline_snapshot_to_dict from webui.timeline import load_timeline, snapshot_to_dict as timeline_snapshot_to_dict
from webui.analytics_loader import (
load_analytics,
record_usage,
snapshot_to_dict as analytics_snapshot_to_dict,
)
from webui.analytics_views import render_analytics_page
from webui.system_health import ( from webui.system_health import (
API_PATH as SYSTEM_HEALTH_API_PATH, API_PATH as SYSTEM_HEALTH_API_PATH,
load_system_health, load_system_health,
process_uptime, process_uptime,
snapshot_to_dict as system_health_to_dict, snapshot_to_dict as system_health_to_dict,
) )
from webui.system_health_views import render_system_health_page
_READ_ONLY_METHODS = frozenset({"GET", "HEAD", "OPTIONS"}) _READ_ONLY_METHODS = frozenset({"GET", "HEAD", "OPTIONS"})
_AUDIT_MUTATION_PATHS = frozenset({"/audit", "/api/audit"}) _AUDIT_MUTATION_PATHS = frozenset({"/audit", "/api/audit"})
@@ -167,24 +167,6 @@ async def api_system_health(request: Request) -> JSONResponse:
return JSONResponse(payload, status_code=200 if snapshot.ready else 503) return JSONResponse(payload, status_code=200 if snapshot.ready else 503)
async def system_health(request: Request) -> HTMLResponse:
"""Read-only system-health dashboard (#639).
Shares the #634 snapshot loader with the JSON API so the page can never
disagree with it. `?deep=1` opts into the network probe exactly as the API
does; the default page load stays cheap. The response is always 200: this
is an operator view that must render the degraded state, not withhold it.
"""
deep = _truthy_flag(request.query_params.get("deep"))
snapshot = load_system_health(deep=deep)
return HTMLResponse(
render_page(
title="System health",
body_html=render_system_health_page(snapshot),
)
)
async def queue(_request: Request) -> HTMLResponse: async def queue(_request: Request) -> HTMLResponse:
snapshot = load_queue_snapshot() snapshot = load_queue_snapshot()
return HTMLResponse(render_page(title="Queue", body_html=render_queue_page(snapshot))) return HTMLResponse(render_page(title="Queue", body_html=render_queue_page(snapshot)))
@@ -465,30 +447,6 @@ async def api_action_attempt(request: Request) -> JSONResponse:
return JSONResponse(result, status_code=status) return JSONResponse(result, status_code=status)
async def api_inventory(_request: Request) -> JSONResponse:
"""Unified read-only session/lease/lock/worktree inventory (#636)."""
snapshot = load_inventory_snapshot()
return JSONResponse(inventory_snapshot_to_dict(snapshot))
async def api_inventory_section(request: Request) -> JSONResponse:
"""Resource-split view: one inventory section under the shared schema."""
section = request.path_params["section"]
if section not in _INVENTORY_SECTIONS:
return JSONResponse(
{
"error": "unknown_section",
"detail": f"no inventory section named {section!r}",
"available": sorted(_INVENTORY_SECTIONS),
},
status_code=404,
)
snapshot = load_inventory_snapshot(include=(section,))
payload = inventory_snapshot_to_dict(snapshot)
payload["requested_section"] = section
return JSONResponse(payload)
async def api_console_security_model(_request: Request) -> JSONResponse: async def api_console_security_model(_request: Request) -> JSONResponse:
"""Read-only publication of the #633 authorization/redaction/audit model.""" """Read-only publication of the #633 authorization/redaction/audit model."""
return JSONResponse({ return JSONResponse({
@@ -596,6 +554,72 @@ async def api_v1_timeline(request: Request) -> JSONResponse:
return JSONResponse(timeline_snapshot_to_dict(snapshot), status_code=status_code) return JSONResponse(timeline_snapshot_to_dict(snapshot), status_code=status_code)
async def analytics(request: Request) -> HTMLResponse:
"""Read-only model usage, token cost, latency, and performance analytics HTML view (#651)."""
snapshot = load_analytics(
remote=request.query_params.get("remote"),
org=request.query_params.get("org"),
repo=request.query_params.get("repo"),
role=request.query_params.get("role"),
model=request.query_params.get("model"),
stage=request.query_params.get("stage"),
issue_number=_query_int(request, "issue"),
pr_number=_query_int(request, "pr"),
limit=_query_int(request, "limit") or 200,
)
return HTMLResponse(render_analytics_page(snapshot))
async def api_v1_analytics(request: Request) -> JSONResponse:
"""Read-only model usage, token cost, latency, and performance analytics API (#651)."""
snapshot = load_analytics(
remote=request.query_params.get("remote"),
org=request.query_params.get("org"),
repo=request.query_params.get("repo"),
role=request.query_params.get("role"),
model=request.query_params.get("model"),
stage=request.query_params.get("stage"),
issue_number=_query_int(request, "issue"),
pr_number=_query_int(request, "pr"),
limit=_query_int(request, "limit") or 500,
)
status_code = 200 if snapshot.ok else 500
return JSONResponse(analytics_snapshot_to_dict(snapshot), status_code=status_code)
async def api_v1_analytics_ingest(request: Request) -> JSONResponse:
"""Optional session instrumentation ingestion endpoint (#651)."""
try:
body = await request.json()
except Exception:
return JSONResponse({"error": "invalid_json", "detail": "body must be valid JSON"}, status_code=400)
if not isinstance(body, dict):
return JSONResponse({"error": "invalid_payload", "detail": "payload must be a JSON object"}, status_code=400)
usage_id = record_usage(
session_id=body.get("session_id"),
remote=body.get("remote", "dadeschools"),
org=body.get("org", ""),
repo=body.get("repo", ""),
project_id=body.get("project_id"),
role=body.get("role", "unknown"),
model=body.get("model", "unknown"),
issue_number=body.get("issue_number") or body.get("issue"),
pr_number=body.get("pr_number") or body.get("pr"),
stage=body.get("stage", "unknown"),
input_tokens=body.get("input_tokens"),
output_tokens=body.get("output_tokens"),
total_tokens=body.get("total_tokens"),
estimated_cost_usd=body.get("estimated_cost_usd"),
latency_ms=body.get("latency_ms"),
duration_ms=body.get("duration_ms"),
status=body.get("status", "success"),
metadata=body.get("metadata"),
)
return JSONResponse({"ok": True, "usage_id": usage_id}, status_code=201)
async def method_not_allowed(request: Request, _exc: Exception) -> Response: async def method_not_allowed(request: Request, _exc: Exception) -> Response:
path = request.url.path path = request.url.path
if path in _AUDIT_MUTATION_PATHS and request.method == "POST": if path in _AUDIT_MUTATION_PATHS and request.method == "POST":
@@ -619,7 +643,6 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
Route("/", home, methods=["GET"]), Route("/", home, methods=["GET"]),
Route("/health", health, methods=["GET"]), Route("/health", health, methods=["GET"]),
Route(SYSTEM_HEALTH_API_PATH, api_system_health, methods=["GET"]), Route(SYSTEM_HEALTH_API_PATH, api_system_health, methods=["GET"]),
Route("/system-health", system_health, methods=["GET"]),
Route("/queue", queue, methods=["GET"]), Route("/queue", queue, methods=["GET"]),
Route("/api/queue", api_queue, methods=["GET"]), Route("/api/queue", api_queue, methods=["GET"]),
Route("/projects", projects, methods=["GET"]), Route("/projects", projects, methods=["GET"]),
@@ -637,6 +660,10 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
Route("/runtime", runtime, methods=["GET"]), Route("/runtime", runtime, methods=["GET"]),
Route("/api/runtime", api_runtime, methods=["GET"]), Route("/api/runtime", api_runtime, methods=["GET"]),
Route("/api/v1/timeline", api_v1_timeline, methods=["GET"]), Route("/api/v1/timeline", api_v1_timeline, methods=["GET"]),
Route("/analytics", analytics, methods=["GET"]),
Route("/api/analytics", api_v1_analytics, methods=["GET"]),
Route("/api/v1/analytics", api_v1_analytics, methods=["GET"]),
Route("/api/v1/analytics/usage", api_v1_analytics_ingest, methods=["POST"]),
Route("/audit", audit, methods=["GET", "POST"]), Route("/audit", audit, methods=["GET", "POST"]),
Route("/api/audit", api_audit, methods=["GET", "POST"]), Route("/api/audit", api_audit, methods=["GET", "POST"]),
Route("/worktrees", worktrees, methods=["GET"]), Route("/worktrees", worktrees, methods=["GET"]),
@@ -655,12 +682,6 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
methods=["POST"], methods=["POST"],
), ),
Route("/api/leases", api_leases, methods=["GET"]), Route("/api/leases", api_leases, methods=["GET"]),
Route("/api/v1/inventory", api_inventory, methods=["GET"]),
Route(
"/api/v1/inventory/{section}",
api_inventory_section,
methods=["GET"],
),
Route( Route(
"/api/console/security-model", "/api/console/security-model",
api_console_security_model, api_console_security_model,
-28
View File
@@ -236,34 +236,6 @@ _ACTION_SPECS: tuple[ConsoleAction, ...] = (
phase=3, phase=3,
summary="Remove a remote feature branch.", summary="Remove a remote feature branch.",
), ),
# #642: sanctioned daemon lifecycle. These exist so operators have an
# audited path off `pkill -f mcp_server.py` (#630). Restart drops every
# in-flight request on a namespace, so it carries the same dual-control and
# break-glass weight as a merge; reload drains first and is privileged but
# not destructive. Neither ever exposes a raw kill: execution is handed to
# a host supervisor by ``webui.sanctioned_restart``.
ConsoleAction(
action_id="system.reload_namespace",
task_key="reload_namespace",
action_class=CLASS_PRIVILEGED,
minimum_role=CONTROLLER,
requires_confirmation=True,
dual_control=False,
break_glass=False,
phase=2,
summary="Gracefully reload one MCP namespace via the host supervisor.",
),
ConsoleAction(
action_id="system.restart_namespace",
task_key="restart_namespace",
action_class=CLASS_DESTRUCTIVE,
minimum_role=ADMIN,
requires_confirmation=True,
dual_control=True,
break_glass=True,
phase=2,
summary="Restart one MCP namespace via the host supervisor.",
),
) )
ACTIONS: dict[str, ConsoleAction] = {a.action_id: a for a in _ACTION_SPECS} ACTIONS: dict[str, ConsoleAction] = {a.action_id: a for a in _ACTION_SPECS}
-13
View File
@@ -110,8 +110,6 @@ def _format_target(action_id: str, params: dict[str, Any]) -> str:
) )
if action_id == "create_issue": if action_id == "create_issue":
return f"issue {params.get('title', '?')!r}" return f"issue {params.get('title', '?')!r}"
if action_id in {"system.restart_namespace", "system.reload_namespace"}:
return f"MCP namespace {params.get('namespace', '?')!r}"
return "unspecified" return "unspecified"
@@ -167,17 +165,6 @@ def build_action_registry() -> ActionRegistry:
"gitea_create_issue_comment", "Post a PR review thread comment."), "gitea_create_issue_comment", "Post a PR review thread comment."),
("close_pr", "Close PR", "close_pr", "gitea_edit_pr", ("close_pr", "Close PR", "close_pr", "gitea_edit_pr",
"Close a pull request without merge."), "Close a pull request without merge."),
# #642: the sanctioned replacement for the forbidden manual daemon-kill
# recovery path (#630). The "tool" is a host supervisor hook, not an MCP
# call — the console never signals a process. Preview and gating live in
# ``webui.sanctioned_restart``; these stay disabled like every other
# registry entry.
("system.reload_namespace", "Reload MCP namespace", "reload_namespace",
"host.supervisor_reload",
"Gracefully reload one MCP namespace via the host supervisor."),
("system.restart_namespace", "Restart MCP namespace",
"restart_namespace", "host.supervisor_restart",
"Restart one MCP namespace via the host supervisor."),
) )
actions = tuple( actions = tuple(
GatedAction( GatedAction(
-952
View File
@@ -1,952 +0,0 @@
"""Unified session/lease/lock/worktree inventory for the web console (#636).
Leases (#433), worktrees (#432), and runtime (#430) each ship their own MVP
view, each with its own shape and its own idea of what "owned" means. A
traffic-control or recovery operator has to read all three and correlate them
by hand, which is exactly the step that goes wrong under collision pressure.
This module aggregates them into one versioned, read-only snapshot so the
console, and any worker asking "what is safe to do next", read the same
inventory from the same authority.
Field authority is explicit and never blended. Every section declares where its
rows came from:
* ``control_plane_db`` the #613 substrate: sessions, leases, assignments.
Authoritative for *exclusive ownership* (#600/#601).
* ``filesystem`` durable per-issue lock files (:mod:`issue_lock_store`) and
registered git worktrees. Authoritative for *what exists on this machine*.
* ``gitea`` remote issue/PR state, reached only through existing loaders.
Safety invariants:
* **Read-only.** The control-plane database is opened through a ``mode=ro``
URI. :class:`control_plane_db.ControlPlaneDB` creates directories and runs
migrations in its constructor, which an inventory read must never do, so this
module talks to sqlite directly rather than through that class.
* **Fail-soft, never fail-silent.** A subsystem that cannot be read degrades to
a section carrying ``status`` and ``reason``. It never raises, and it never
produces an empty list that reads like "nothing is there".
* **Never invent active ownership.** This is the invariant that matters most.
A degraded ownership source sets ``ownership_authority_complete`` false, and
while that flag is false no work item is reported unowned and no collision is
asserted. Absence of evidence is reported as absence of evidence.
* **Redaction at the boundary.** Absolute paths are collapsed against the home
directory, URLs lose userinfo and query strings, and no credential-shaped
value is emitted. No session token exists in these sources and none is read.
Phase 1 is read-only. Lease steal/release and worktree deletion are Phase 2+
and deliberately have no representation here, not even a disabled one.
"""
from __future__ import annotations
import os
import re
import sqlite3
import time
from dataclasses import dataclass, field
from datetime import datetime, timezone
from typing import Any, Callable
from urllib.parse import urlparse
import control_plane_db
import issue_lock_store
SCHEMA_VERSION = 1
API_VERSION = "v1"
#: Sections whose absence would make an ownership claim unprovable. If any of
#: these is not ``ok``, the snapshot refuses to describe anything as unowned.
OWNERSHIP_SECTIONS = ("sessions", "leases", "locks")
SECTION_NAMES = ("sessions", "leases", "locks", "worktrees", "namespaces")
STATUS_OK = "ok"
STATUS_DEGRADED = "degraded"
STATUS_UNAVAILABLE = "unavailable"
AUTHORITY_CONTROL_PLANE_DB = "control_plane_db"
AUTHORITY_FILESYSTEM = "filesystem"
AUTHORITY_GITEA = "gitea"
_CREDENTIAL_KEY_RE = re.compile(
r"(token|secret|password|passwd|api[_-]?key|authorization|bearer|credential)",
re.IGNORECASE,
)
_REDACTED = "[redacted]"
@dataclass(frozen=True)
class InventorySection:
"""One subsystem's contribution, with its authority and health."""
name: str
authority: str
status: str
items: tuple[dict[str, Any], ...] = ()
reason: str | None = None
scan_ms: float | None = None
@property
def ok(self) -> bool:
return self.status == STATUS_OK
def to_dict(self) -> dict[str, Any]:
return {
"name": self.name,
"authority": self.authority,
"status": self.status,
"count": len(self.items),
"reason": self.reason,
"scan_ms": self.scan_ms,
"items": [dict(item) for item in self.items],
}
@dataclass(frozen=True)
class CollisionSignal:
"""A detected conflict between two ownership records."""
kind: str
message: str
severity: str = "warning"
issue_number: int | None = None
branch: str | None = None
worktree_path: str | None = None
session_ids: tuple[str, ...] = ()
def to_dict(self) -> dict[str, Any]:
return {
"kind": self.kind,
"severity": self.severity,
"message": self.message,
"issue_number": self.issue_number,
"branch": self.branch,
"worktree_path": self.worktree_path,
"session_ids": list(self.session_ids),
}
@dataclass(frozen=True)
class InventorySnapshot:
"""Versioned aggregate of every inventory section."""
generated_at: str
sections: tuple[InventorySection, ...]
collisions: tuple[CollisionSignal, ...] = ()
correlations: tuple[dict[str, Any], ...] = ()
schema_version: int = SCHEMA_VERSION
api_version: str = API_VERSION
scan_ms: float | None = None
_section_index: dict[str, InventorySection] = field(
default_factory=dict, repr=False, compare=False
)
def section(self, name: str) -> InventorySection | None:
return self._section_index.get(name)
@property
def degraded_sections(self) -> tuple[str, ...]:
return tuple(s.name for s in self.sections if not s.ok)
@property
def ownership_authority_complete(self) -> bool:
"""True only when every ownership-bearing section read cleanly.
While this is false the snapshot must not describe any work item as
unowned: a lease the reader could not load is not an absent lease.
"""
for name in OWNERSHIP_SECTIONS:
section = self._section_index.get(name)
if section is None or not section.ok:
return False
return True
@property
def status(self) -> str:
if all(s.ok for s in self.sections):
return STATUS_OK
return STATUS_DEGRADED
# ── redaction ────────────────────────────────────────────────────────────────
def redact_path(path: str | None) -> str | None:
"""Collapse an absolute path against ``$HOME`` for browser display."""
if not path:
return path
text = str(path)
home = os.path.expanduser("~")
if home and home != "/" and text.startswith(home):
return "~" + text[len(home) :]
return text
def redact_url(value: str | None) -> str | None:
"""Strip userinfo and query string from a URL."""
if not value:
return value
text = str(value)
try:
parsed = urlparse(text)
except ValueError:
return _REDACTED
if not parsed.scheme or not parsed.netloc:
return text
netloc = parsed.hostname or ""
if parsed.port:
netloc = f"{netloc}:{parsed.port}"
rebuilt = f"{parsed.scheme}://{netloc}{parsed.path}"
return rebuilt.rstrip("/") or rebuilt
def scrub(value: Any, *, key: str | None = None) -> Any:
"""Recursively drop credential-shaped values and redact paths/URLs.
Never raises: an unexpected object degrades to its ``repr`` rather than
propagating out of a read-only view.
"""
if key and _CREDENTIAL_KEY_RE.search(key):
return _REDACTED
if isinstance(value, dict):
return {str(k): scrub(v, key=str(k)) for k, v in value.items()}
if isinstance(value, (list, tuple)):
return [scrub(v, key=key) for v in value]
if isinstance(value, str):
if value.startswith(("http://", "https://")):
return redact_url(value)
if value.startswith("/") or value.startswith("~"):
return redact_path(value)
return value
if isinstance(value, (int, float, bool)) or value is None:
return value
return repr(value)
# ── control-plane database (read-only) ───────────────────────────────────────
def _open_readonly(db_path: str) -> sqlite3.Connection:
"""Open the control-plane DB without creating or migrating anything."""
conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=5)
conn.row_factory = sqlite3.Row
return conn
def _table_names(conn: sqlite3.Connection) -> set[str]:
rows = conn.execute(
"SELECT name FROM sqlite_master WHERE type = 'table'"
).fetchall()
return {str(row[0]) for row in rows}
def _load_cp_db_sections(
*,
db_path: str | None = None,
limit: int = 200,
) -> tuple[InventorySection, InventorySection]:
"""Return the ``sessions`` and ``leases`` sections from the #613 DB."""
path = (db_path or control_plane_db.default_db_path()).strip()
def _both_unavailable(reason: str) -> tuple[InventorySection, InventorySection]:
return (
InventorySection(
name="sessions",
authority=AUTHORITY_CONTROL_PLANE_DB,
status=STATUS_UNAVAILABLE,
reason=reason,
),
InventorySection(
name="leases",
authority=AUTHORITY_CONTROL_PLANE_DB,
status=STATUS_UNAVAILABLE,
reason=reason,
),
)
if not path:
return _both_unavailable("control-plane database path is not configured")
if not os.path.exists(path):
return _both_unavailable(
f"control-plane database not present at {redact_path(path)}; "
"no session or lease authority available"
)
started = time.perf_counter()
try:
conn = _open_readonly(path)
except sqlite3.Error as exc:
return _both_unavailable(f"control-plane database could not be opened: {exc}")
try:
tables = _table_names(conn)
if "sessions" not in tables or "leases" not in tables:
missing = sorted({"sessions", "leases"} - tables)
return _both_unavailable(
"control-plane database is missing required tables: "
+ ", ".join(missing)
)
session_rows = [
dict(row)
for row in conn.execute(
"SELECT session_id, role, profile, namespace, pid, started_at,"
" last_heartbeat_at, status FROM sessions"
" ORDER BY last_heartbeat_at DESC LIMIT ?",
(max(1, int(limit)),),
).fetchall()
]
has_work_items = "work_items" in tables
if has_work_items:
lease_sql = (
"SELECT l.lease_id, l.session_id, l.role, l.phase, l.status,"
" l.expires_at, w.remote, w.org, w.repo, w.kind AS work_kind,"
" w.number AS work_number, w.state AS work_state,"
" s.pid AS session_pid, s.profile AS session_profile,"
" s.namespace AS session_namespace, s.status AS session_status"
" FROM leases l"
" JOIN work_items w ON w.work_item_id = l.work_item_id"
" LEFT JOIN sessions s ON s.session_id = l.session_id"
" ORDER BY l.expires_at DESC LIMIT ?"
)
else:
lease_sql = (
"SELECT l.lease_id, l.session_id, l.role, l.phase, l.status,"
" l.expires_at FROM leases l"
" ORDER BY l.expires_at DESC LIMIT ?"
)
lease_rows = [
dict(row)
for row in conn.execute(lease_sql, (max(1, int(limit)),)).fetchall()
]
except sqlite3.Error as exc:
return _both_unavailable(f"control-plane database read failed: {exc}")
finally:
conn.close()
elapsed = (time.perf_counter() - started) * 1000.0
now = datetime.now(timezone.utc)
sessions = tuple(
scrub(
{
"session_id": row.get("session_id"),
"role": row.get("role"),
"profile": row.get("profile"),
"namespace": row.get("namespace"),
"pid": row.get("pid"),
"pid_alive": issue_lock_store.is_process_alive(row.get("pid")),
"started_at": row.get("started_at"),
"last_heartbeat_at": row.get("last_heartbeat_at"),
"status": row.get("status"),
}
)
for row in session_rows
)
leases = tuple(
scrub(
{
"lease_id": row.get("lease_id"),
"session_id": row.get("session_id"),
"role": row.get("role"),
"phase": row.get("phase"),
"status": row.get("status"),
"expires_at": row.get("expires_at"),
"expired": _is_expired(row.get("expires_at"), now=now),
"remote": row.get("remote"),
"org": row.get("org"),
"repo": row.get("repo"),
"work_kind": row.get("work_kind"),
"work_number": row.get("work_number"),
"work_state": row.get("work_state"),
"session_pid": row.get("session_pid"),
"session_profile": row.get("session_profile"),
"session_namespace": row.get("session_namespace"),
"session_status": row.get("session_status"),
}
)
for row in lease_rows
)
degraded_reason = (
None
if has_work_items
else "work_items table absent; lease rows carry no work linkage"
)
lease_status = STATUS_OK if has_work_items else STATUS_DEGRADED
return (
InventorySection(
name="sessions",
authority=AUTHORITY_CONTROL_PLANE_DB,
status=STATUS_OK,
items=sessions,
scan_ms=round(elapsed, 3),
),
InventorySection(
name="leases",
authority=AUTHORITY_CONTROL_PLANE_DB,
status=lease_status,
items=leases,
reason=degraded_reason,
scan_ms=round(elapsed, 3),
),
)
def _is_expired(expires_at: str | None, *, now: datetime) -> bool | None:
if not expires_at:
return None
text = str(expires_at).strip().replace("Z", "+00:00")
try:
parsed = datetime.fromisoformat(text)
except ValueError:
return None
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=timezone.utc)
return parsed <= now
# ── durable issue locks (filesystem) ─────────────────────────────────────────
def _load_locks_section(*, lock_dir: str | None = None) -> InventorySection:
started = time.perf_counter()
try:
paths = issue_lock_store.iter_lock_files(lock_dir)
except OSError as exc:
return InventorySection(
name="locks",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_UNAVAILABLE,
reason=f"issue lock directory could not be listed: {exc}",
)
items: list[dict[str, Any]] = []
unreadable = 0
for path in paths:
try:
record = issue_lock_store.read_lock_file(path)
except (OSError, ValueError):
unreadable += 1
continue
if not record:
unreadable += 1
continue
try:
freshness = issue_lock_store.assess_lock_freshness(record)
except Exception: # noqa: BLE001 — a read-only view never raises
freshness = {"status": "unknown", "live": False, "stale": False}
claimant = record.get("claimant") or (
(record.get("work_lease") or {}).get("claimant") or {}
)
items.append(
scrub(
{
"issue_number": record.get("issue_number"),
"branch_name": record.get("branch_name"),
"remote": record.get("remote"),
"org": record.get("org"),
"repo": record.get("repo"),
"worktree_path": record.get("worktree_path"),
"pid": record.get("session_pid") or record.get("pid"),
"pid_alive": issue_lock_store.is_process_alive(
record.get("session_pid") or record.get("pid")
),
"claimant_username": (claimant or {}).get("username"),
"claimant_profile": (claimant or {}).get("profile"),
"lock_generation": record.get("lock_generation"),
"freshness_status": freshness.get("status"),
"live": bool(freshness.get("live")),
"stale": bool(freshness.get("stale")),
"freshness_reason": freshness.get("reason"),
"lock_path": record.get("lock_file_path") or path,
}
)
)
elapsed = (time.perf_counter() - started) * 1000.0
reason = (
f"{unreadable} lock file(s) were unreadable and are not represented"
if unreadable
else None
)
return InventorySection(
name="locks",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_DEGRADED if unreadable else STATUS_OK,
items=tuple(items),
reason=reason,
scan_ms=round(elapsed, 3),
)
# ── worktrees (filesystem, via the #432 scanner) ─────────────────────────────
def _load_worktrees_section(
*, load_hygiene: Callable[[], Any] | None = None
) -> InventorySection:
started = time.perf_counter()
try:
loader = load_hygiene
if loader is None:
from webui.worktree_scanner import load_hygiene_snapshot
loader = load_hygiene_snapshot
snapshot = loader()
except Exception as exc: # noqa: BLE001 — fail soft, never fail the request
return InventorySection(
name="worktrees",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_UNAVAILABLE,
reason=f"worktree scan failed: {exc}",
)
items = tuple(
scrub(
{
"rel_path": entry.rel_path,
"folder_name": entry.folder_name,
"classification": entry.classification,
"branch": entry.branch,
"head_sha": entry.head_sha,
"dirty_tracked": entry.dirty_tracked,
"dirty_untracked": entry.dirty_untracked,
"detached": entry.detached,
"registered_worktree": entry.registered_worktree,
"notes": entry.notes,
}
)
for entry in snapshot.entries
)
scan_error = getattr(snapshot, "scan_error", None)
elapsed = (time.perf_counter() - started) * 1000.0
return InventorySection(
name="worktrees",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_DEGRADED if scan_error else STATUS_OK,
items=items,
reason=scan_error,
scan_ms=round(elapsed, 3),
)
# ── namespaces / capability summary ──────────────────────────────────────────
def _load_namespaces_section() -> InventorySection:
started = time.perf_counter()
try:
from gitea_auth import get_profile
profile = get_profile() or {}
except Exception as exc: # noqa: BLE001
return InventorySection(
name="namespaces",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_UNAVAILABLE,
reason=f"active profile could not be resolved: {exc}",
)
allowed = list(profile.get("allowed_operations") or [])
forbidden = list(profile.get("forbidden_operations") or [])
profile_name = str(profile.get("profile_name") or "")
namespace = None
try:
import role_namespace_gate
namespace = role_namespace_gate.infer_mcp_namespace(profile_name)
except Exception: # noqa: BLE001 — namespace inference is advisory
namespace = None
item = scrub(
{
"profile_name": profile_name,
"role": profile.get("role"),
"mcp_namespace": namespace,
"allowed_operations": sorted(allowed),
"forbidden_operations": sorted(forbidden),
"capability_summary": {
"can_author": "gitea.pr.create" in allowed,
"can_review": "gitea.pr.approve" in allowed,
"can_merge": "gitea.pr.merge" in allowed,
"can_close_pr": "gitea.pr.close" in allowed,
},
"active": True,
}
)
elapsed = (time.perf_counter() - started) * 1000.0
return InventorySection(
name="namespaces",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_OK,
items=(item,),
reason=(
"only the profile serving this web process is observable; other "
"namespaces are not enumerable from here"
),
scan_ms=round(elapsed, 3),
)
# ── correlation and collision detection ──────────────────────────────────────
def _issue_from_branch(branch: str | None) -> int | None:
match = re.search(r"issue-(\d+)", str(branch or ""), re.IGNORECASE)
return int(match.group(1)) if match else None
def correlate(
*,
leases: InventorySection,
locks: InventorySection,
worktrees: InventorySection,
sessions: InventorySection,
) -> tuple[tuple[dict[str, Any], ...], tuple[CollisionSignal, ...]]:
"""Join lease owner ↔ lock ↔ worktree ↔ namespace where evidence allows.
Correlation rows are emitted from whatever sections did load. Collision
signals are only emitted from sections that are ``ok``: a conflict inferred
from a partially-read source would be a false accusation.
"""
correlations: list[dict[str, Any]] = []
collisions: list[CollisionSignal] = []
worktree_by_branch: dict[str, dict[str, Any]] = {}
for entry in worktrees.items:
branch = (entry.get("branch") or "").strip()
if branch:
worktree_by_branch.setdefault(branch, entry)
session_by_id = {
str(s.get("session_id")): s for s in sessions.items if s.get("session_id")
}
# Lock-centred rows: a durable lock names an issue, a branch, and a worktree.
for lock in locks.items:
branch = (lock.get("branch_name") or "").strip()
worktree = worktree_by_branch.get(branch)
matching_leases = [
lease
for lease in leases.items
if lease.get("work_kind") == "issue"
and lease.get("work_number") == lock.get("issue_number")
]
correlations.append(
{
"issue_number": lock.get("issue_number"),
"branch": branch or None,
"lock_live": bool(lock.get("live")),
"lock_claimant": lock.get("claimant_profile"),
"lock_pid": lock.get("pid"),
"lock_pid_alive": lock.get("pid_alive"),
"worktree_rel_path": (worktree or {}).get("rel_path"),
"worktree_classification": (worktree or {}).get("classification"),
"worktree_registered": (worktree or {}).get("registered_worktree"),
"lease_ids": [
lease.get("lease_id")
for lease in matching_leases
if lease.get("lease_id")
],
"lease_sessions": [
lease.get("session_id")
for lease in matching_leases
if lease.get("session_id")
],
}
)
if locks.ok and worktrees.ok:
# A claim whose lease window is still open but has no registered
# worktree is an anomaly regardless of whether its pid is alive; a
# fully time-expired lease is on its way out and is not flagged.
if (
lock.get("freshness_status") != "expired"
and branch
and worktree is None
):
collisions.append(
CollisionSignal(
kind="lock-without-worktree",
severity="warning",
issue_number=lock.get("issue_number"),
branch=branch,
worktree_path=lock.get("worktree_path"),
message=(
f"Live lock on issue #{lock.get('issue_number')} names "
f"branch {branch!r} but no registered worktree carries "
"that branch (#404)"
),
)
)
if locks.ok:
# A lock whose recorded pid is gone is held by nobody: a clean #753
# dead-session recovery candidate. Subclassify by the lease window,
# because the two cases need different operator urgency. When the
# window is still open the lock would read as live to a naive
# timestamp check even though the owner is dead — the more dangerous
# case — so it is flagged distinctly from a fully time-expired lease.
if lock.get("pid_alive") is False and lock.get("stale"):
if lock.get("freshness_status") == "expired":
collisions.append(
CollisionSignal(
kind="stale-lock-dead-owner",
severity="warning",
issue_number=lock.get("issue_number"),
branch=branch or None,
message=(
f"Lock on issue #{lock.get('issue_number')} is stale "
f"and its recorded pid {lock.get('pid')} is not running "
"(#753 dead-session recovery candidate)"
),
)
)
else:
collisions.append(
CollisionSignal(
kind="live-lock-dead-owner",
severity="warning",
issue_number=lock.get("issue_number"),
branch=branch or None,
message=(
f"Lock on issue #{lock.get('issue_number')} has an "
"unexpired lease but its recorded pid "
f"{lock.get('pid')} is not running; it would read as "
"live to a timestamp check (#753 dead-session recovery "
"candidate)"
),
)
)
# The #635 trap: the lease has expired but the recorded pid is a
# still-running daemon, so neither dead-pid reclaim nor exact-owner
# renewal applies. This is the collision an operator must see.
elif (
lock.get("freshness_status") == "expired"
and lock.get("pid_alive") is True
):
collisions.append(
CollisionSignal(
kind="expired-lock-live-owner",
severity="error",
issue_number=lock.get("issue_number"),
branch=branch or None,
message=(
f"Lock on issue #{lock.get('issue_number')} has an expired "
f"lease but its recorded pid {lock.get('pid')} is still "
"running (daemon-pid deadlock; needs an operator decision, "
"#635/#760)"
),
)
)
# Two live locks on one branch, or two active leases on one work item.
if locks.ok:
by_branch: dict[str, list[dict[str, Any]]] = {}
for lock in locks.items:
if not lock.get("live"):
continue
branch = (lock.get("branch_name") or "").strip()
if branch:
by_branch.setdefault(branch, []).append(lock)
for branch, entries in sorted(by_branch.items()):
if len(entries) > 1:
collisions.append(
CollisionSignal(
kind="duplicate-live-lock",
severity="error",
branch=branch,
message=(
f"{len(entries)} live locks name branch {branch!r}: "
"issues "
+ ", ".join(
f"#{e.get('issue_number')}" for e in entries
)
),
)
)
if leases.ok:
by_work: dict[tuple[str, int], list[dict[str, Any]]] = {}
for lease in leases.items:
if str(lease.get("status") or "").lower() != "active":
continue
kind = str(lease.get("work_kind") or "").strip().lower()
number = lease.get("work_number")
if not kind or number is None:
continue
by_work.setdefault((kind, int(number)), []).append(lease)
for (kind, number), entries in sorted(by_work.items()):
sessions_held = {
str(e.get("session_id")) for e in entries if e.get("session_id")
}
if len(sessions_held) > 1:
collisions.append(
CollisionSignal(
kind="concurrent-active-lease",
severity="error",
issue_number=number if kind == "issue" else None,
session_ids=tuple(sorted(sessions_held)),
message=(
f"{len(sessions_held)} sessions hold an active lease on "
f"{kind} #{number}"
),
)
)
for entry in entries:
if entry.get("expired") is True:
collisions.append(
CollisionSignal(
kind="active-lease-past-expiry",
severity="warning",
issue_number=number if kind == "issue" else None,
session_ids=(
(str(entry.get("session_id")),)
if entry.get("session_id")
else ()
),
message=(
f"Lease {entry.get('lease_id')} on {kind} #{number} "
"is still marked active past its expiry"
),
)
)
# A lease whose owning session is gone is an orphan, not free work.
if leases.ok and sessions.ok:
for lease in leases.items:
if str(lease.get("status") or "").lower() != "active":
continue
session_id = str(lease.get("session_id") or "")
if session_id and session_id not in session_by_id:
collisions.append(
CollisionSignal(
kind="orphan-lease",
severity="error",
session_ids=(session_id,),
message=(
f"Active lease {lease.get('lease_id')} names session "
f"{session_id}, which has no session record"
),
)
)
return tuple(correlations), tuple(collisions)
# ── snapshot assembly ────────────────────────────────────────────────────────
def load_inventory_snapshot(
*,
db_path: str | None = None,
lock_dir: str | None = None,
load_hygiene: Callable[[], Any] | None = None,
include: tuple[str, ...] | None = None,
) -> InventorySnapshot:
"""Build the unified inventory snapshot.
Every section is loaded independently and fails soft. *include* restricts
the sections that are scanned; omitted sections are simply absent rather
than reported as empty, so a resource-split request cannot be mistaken for
a whole-inventory answer.
"""
started = time.perf_counter()
wanted = tuple(include) if include else SECTION_NAMES
sections: list[InventorySection] = []
sessions_section: InventorySection | None = None
leases_section: InventorySection | None = None
if "sessions" in wanted or "leases" in wanted:
sessions_section, leases_section = _load_cp_db_sections(db_path=db_path)
if "sessions" in wanted:
sections.append(sessions_section)
if "leases" in wanted:
sections.append(leases_section)
locks_section = (
_load_locks_section(lock_dir=lock_dir)
if "locks" in wanted
else _empty_section("locks", AUTHORITY_FILESYSTEM)
)
if "locks" in wanted:
sections.append(locks_section)
worktrees_section = (
_load_worktrees_section(load_hygiene=load_hygiene)
if "worktrees" in wanted
else _empty_section("worktrees", AUTHORITY_FILESYSTEM)
)
if "worktrees" in wanted:
sections.append(worktrees_section)
if "namespaces" in wanted:
sections.append(_load_namespaces_section())
correlations, collisions = correlate(
leases=leases_section or _empty_section("leases", AUTHORITY_CONTROL_PLANE_DB),
locks=locks_section,
worktrees=worktrees_section,
sessions=sessions_section
or _empty_section("sessions", AUTHORITY_CONTROL_PLANE_DB),
)
elapsed = (time.perf_counter() - started) * 1000.0
index = {section.name: section for section in sections}
return InventorySnapshot(
generated_at=datetime.now(timezone.utc).isoformat(),
sections=tuple(sections),
collisions=collisions,
correlations=correlations,
scan_ms=round(elapsed, 3),
_section_index=index,
)
def _empty_section(name: str, authority: str) -> InventorySection:
"""A section that was not requested — never a claim that it is empty."""
return InventorySection(
name=name,
authority=authority,
status=STATUS_UNAVAILABLE,
reason="section not requested in this scan",
)
def snapshot_to_dict(snapshot: InventorySnapshot) -> dict[str, Any]:
"""Serialize the snapshot for the versioned API."""
return {
"api_version": snapshot.api_version,
"schema_version": snapshot.schema_version,
"generated_at": snapshot.generated_at,
"status": snapshot.status,
"scan_ms": snapshot.scan_ms,
"ownership_authority_complete": snapshot.ownership_authority_complete,
"ownership_note": (
"Every ownership source read cleanly; an item absent from leases "
"and locks is genuinely unclaimed."
if snapshot.ownership_authority_complete
else "One or more ownership sources are degraded; nothing in this "
"snapshot may be treated as unowned. Collisions are reported only "
"from sections that read cleanly."
),
"degraded_sections": list(snapshot.degraded_sections),
"field_authority": {
"sessions": AUTHORITY_CONTROL_PLANE_DB,
"leases": AUTHORITY_CONTROL_PLANE_DB,
"locks": AUTHORITY_FILESYSTEM,
"worktrees": AUTHORITY_FILESYSTEM,
"namespaces": AUTHORITY_FILESYSTEM,
},
"sections": {section.name: section.to_dict() for section in snapshot.sections},
"correlations": [dict(row) for row in snapshot.correlations],
"collisions": [signal.to_dict() for signal in snapshot.collisions],
}
-19
View File
@@ -236,25 +236,6 @@ def render_page(*, title: str, body_html: str, extra_head: str = "") -> str:
.badge-in-review {{ color: #9ec8f0; border-color: #3d5f7a; }} .badge-in-review {{ color: #9ec8f0; border-color: #3d5f7a; }}
.badge-duplicate {{ color: #e0c27a; border-color: #6b5730; }} .badge-duplicate {{ color: #e0c27a; border-color: #6b5730; }}
.badge-stale {{ color: #c9b8e8; border-color: #5a4a78; }} .badge-stale {{ color: #c9b8e8; border-color: #5a4a78; }}
.badge-health-ok {{ color: #8fd19e; border-color: #3d6b4a; }}
.badge-health-degraded {{ color: #e0c27a; border-color: #6b5730; }}
.badge-health-down {{ color: #f0a8a8; border-color: #7a3b3b; }}
.badge-health-skipped {{ color: var(--muted); }}
.badge-health-unproven {{ color: #c9b8e8; border-color: #5a4a78; }}
.health-card {{
margin: 1.25rem 0;
padding: 0.85rem 1rem 1rem;
border: 1px solid var(--border);
border-radius: 8px;
background: var(--surface);
}}
.health-card h3 {{ margin: 0 0 0.5rem; font-size: 1.05rem; }}
.health-card h4 {{ margin: 1rem 0 0.35rem; font-size: 0.92rem; color: var(--muted); }}
.health-headline {{ color: var(--text); font-size: 1rem; margin: 0 0 0.5rem; }}
.health-degraded {{ border-left-color: #e0c27a; }}
.health-stale {{ border-left-color: #f0a8a8; }}
ul.reasons {{ margin: 0.35rem 0; padding-left: 1.15rem; color: var(--muted); font-size: 0.9rem; }}
ul.reasons li {{ margin-bottom: 0.3rem; }}
</style> </style>
{extra_head} {extra_head}
</head> </head>
+1 -1
View File
@@ -38,7 +38,6 @@ class NavGroup:
NAV_GROUPS: tuple[NavGroup, ...] = ( NAV_GROUPS: tuple[NavGroup, ...] = (
NavGroup("Health", ( NavGroup("Health", (
NavItem("/health", "Liveness"), NavItem("/health", "Liveness"),
NavItem("/system-health", "System health"),
)), )),
NavGroup("Traffic", ( NavGroup("Traffic", (
NavItem("/queue", "Queue"), NavItem("/queue", "Queue"),
@@ -65,6 +64,7 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
)), )),
NavGroup("Insights", ( NavGroup("Insights", (
NavItem("/insights", "Insights", "stub"), NavItem("/insights", "Insights", "stub"),
NavItem("/analytics", "Analytics"),
NavItem("/audit", "Audit"), NavItem("/audit", "Audit"),
)), )),
) )
-613
View File
@@ -1,613 +0,0 @@
"""Sanctioned MCP restart and graceful reload controls (#642, Phase 2).
Sessions have historically recovered MCP connectivity by killing the host
daemon (``pkill -f mcp_server.py``, #630). That path stays forbidden: it kills
every namespace on the host, contaminates the surviving session, and leaves no
audit trail. This module is the sanctioned replacement.
A restart is modelled as a *gated action*, never as a command:
1. **Capability** the console action resolves through ``console_authz``
against ``task_capability_map``, so the console cannot invent an authority
the MCP layer does not already define.
2. **Preview** :func:`build_restart_preview` renders a mutation ledger and
the exact confirmation phrase. It never returns a shell command.
3. **Confirmation** the operator echoes a phrase naming the exact namespace
and mode. A phrase for one namespace never authorizes another.
4. **Operator authorization** host daemon maintenance is authorized out of
band through the environment (#630, and #710 finding F1: a worker session
cannot set an env var for an already-running daemon, so this cannot be
self-asserted the way a tool argument could).
5. **Execution** :func:`execute_restart` never spawns a process. Once every
gate passes it hands the request to the configured host-managed restart
hook; with no hook configured it fails closed.
6. **Health recheck** :func:`verify_post_restart_health` requires live
client-namespace probe evidence before any post-restart clean claim.
Manual ``pkill`` remains forbidden and is classified as contamination by
:func:`classify_restart_command`, which blocks clean claims (#630 AC3).
This module performs no I/O beyond reading its own environment configuration,
imports no MCP client, and holds no credential.
"""
from __future__ import annotations
import os
from dataclasses import asdict, dataclass
from typing import Any
import mcp_namespace_health
import runtime_recovery_guard
from webui import console_audit, console_authz
# --- Operations -------------------------------------------------------------
MODE_RESTART = "restart"
MODE_RELOAD = "reload"
MODES: tuple[str, ...] = (MODE_RESTART, MODE_RELOAD)
ACTION_RESTART_NAMESPACE = "system.restart_namespace"
ACTION_RELOAD_NAMESPACE = "system.reload_namespace"
ACTION_FOR_MODE: dict[str, str] = {
MODE_RESTART: ACTION_RESTART_NAMESPACE,
MODE_RELOAD: ACTION_RELOAD_NAMESPACE,
}
# Namespaces the console may target. An unlisted name fails closed rather than
# being passed through to a host hook.
KNOWN_NAMESPACES: tuple[str, ...] = tuple(
sorted(
set(mcp_namespace_health.DEFAULT_NAMESPACES)
| {"gitea-author", "gitea-reviewer", "gitea-merger",
"gitea-reconciler", "gitea-controller"}
)
)
# Scope tokens that would mean "everything at once". Explicit non-goal: the
# console never offers a fleet-wide restart, because that is the blast radius
# `pkill -f mcp_server.py` already had.
_FLEET_TOKENS = frozenset({"*", "all", "fleet", "any", ""})
# --- Environment configuration ----------------------------------------------
# Read server-side only; the value is an opaque host hook reference (e.g. a
# launchd label), never a command line, and is never rendered to a client.
RESTART_HOOK_ENV = "GITEA_SANCTIONED_RESTART_HOOK"
# --- Reason codes -----------------------------------------------------------
DENY_UNKNOWN_MODE = "unknown_mode"
DENY_UNKNOWN_NAMESPACE = "unknown_namespace"
DENY_FLEET_SCOPE = "fleet_scope_not_permitted"
DENY_UNAUTHORIZED = "unauthorized"
DENY_CONFIRMATION_MISSING = "confirmation_required"
DENY_CONFIRMATION_MISMATCH = "confirmation_mismatch"
DENY_OPERATOR_AUTHORIZATION = "operator_authorization_missing"
DENY_HOOK_NOT_CONFIGURED = "restart_hook_not_configured"
DENY_CONTAMINATED_RUNTIME = "contaminated_runtime"
ALLOW_HOST_ACTION_REQUIRED = "host_action_required"
# Post-restart verification outcomes.
HEALTH_CLEAN = "clean"
HEALTH_UNPROVEN = "unproven"
HEALTH_UNHEALTHY = "unhealthy"
def _clean(value: Any) -> str:
return str(value or "").strip()
# --- Mutation ledger --------------------------------------------------------
@dataclass(frozen=True)
class RestartLedgerEntry:
"""One planned step, shown before anything is asked of the host."""
sequence: int
step: str
summary: str
executes_process_kill: bool = False
def _mutation_ledger(namespace: str, mode: str) -> tuple[RestartLedgerEntry, ...]:
if mode == MODE_RELOAD:
middle = RestartLedgerEntry(
sequence=2,
step="host_graceful_reload",
summary=(
f"Ask the configured host supervisor to reload {namespace} "
"in place, draining in-flight requests. The console does not "
"signal the process itself."
),
)
else:
middle = RestartLedgerEntry(
sequence=2,
step="host_restart_hook",
summary=(
f"Ask the configured host supervisor to restart {namespace}. "
"The console never sends a signal and never runs a kill."
),
)
return (
RestartLedgerEntry(
sequence=1,
step="quiesce",
summary=(
f"Stop admitting new gated mutations for {namespace} and "
"record the intent before anything restarts."
),
),
middle,
RestartLedgerEntry(
sequence=3,
step="health_recheck",
summary=(
f"Re-probe {namespace} through the live client namespace and "
"prove the required tool is callable again."
),
),
RestartLedgerEntry(
sequence=4,
step="audit",
summary=(
"Append actor, target namespace, mode, and result to the "
"console audit log."
),
),
)
# --- Confirmation -----------------------------------------------------------
def confirmation_phrase(namespace: str, mode: str) -> str:
"""Exact phrase an operator must echo, naming the namespace and mode.
Binding the namespace into the phrase is the point: a confirmation typed
for ``gitea-author`` cannot be replayed against ``gitea-merger``.
"""
return f"{_clean(mode)} {_clean(namespace)}"
def confirmation_matches(
namespace: str, mode: str, confirmation: str | None
) -> bool:
"""Compare *confirmation* to the required phrase (exact, whitespace-trimmed)."""
return _clean(confirmation) == confirmation_phrase(namespace, mode)
# --- Scope validation -------------------------------------------------------
def _validate_scope(namespace: str, mode: str) -> tuple[str, str] | None:
"""Return ``(reason_code, detail)`` when the scope is refused."""
ns = _clean(namespace)
md = _clean(mode)
if md not in MODES:
return (
DENY_UNKNOWN_MODE,
f"Mode {md!r} is not one of {', '.join(MODES)}.",
)
if ns.lower() in _FLEET_TOKENS:
return (
DENY_FLEET_SCOPE,
(
"Fleet-wide restart is an explicit non-goal: it reproduces the "
"blast radius of `pkill -f mcp_server.py` (#630). Restart one "
"namespace at a time."
),
)
if ns not in KNOWN_NAMESPACES:
return (
DENY_UNKNOWN_NAMESPACE,
f"Namespace {ns!r} is not a known MCP namespace.",
)
return None
# --- Host hook --------------------------------------------------------------
def restart_hook(env: dict[str, str] | None = None) -> dict[str, Any]:
"""Report the configured host-managed restart hook.
The hook is a reference the *host* resolves (a supervisor label), not a
command this process runs. ``configured=False`` fails restart closed.
"""
source = env if env is not None else os.environ
reference = _clean(source.get(RESTART_HOOK_ENV))
return {
"configured": bool(reference),
"reference": reference or None,
"source": RESTART_HOOK_ENV if reference else None,
"self_assertable": False,
"console_executes_process": False,
}
# --- Preview ----------------------------------------------------------------
def build_restart_preview(
namespace: str,
mode: str = MODE_RESTART,
*,
principal: console_authz.Principal | None = None,
env: dict[str, str] | None = None,
) -> dict[str, Any]:
"""Render the dry-run preview for a restart/reload request.
Read-only: no authorization is granted, no host is contacted, and the
result never contains a shell command.
"""
ns = _clean(namespace)
md = _clean(mode)
action_id = ACTION_FOR_MODE.get(md, ACTION_RESTART_NAMESPACE)
action = console_authz.get_action(action_id)
decision = console_authz.authorize(action_id, principal)
scope_error = _validate_scope(ns, md)
hook = restart_hook(env)
operator = runtime_recovery_guard.operator_authorization(env)
return {
"action_id": action_id,
"namespace": ns,
"mode": md,
"scope_valid": scope_error is None,
"scope_reason_code": scope_error[0] if scope_error else None,
"scope_detail": scope_error[1] if scope_error else None,
"required_role": action.minimum_role if action else None,
"required_permission": action.mcp_permission if action else None,
"action_class": action.action_class if action else None,
"dual_control": action.dual_control if action else True,
"break_glass": action.break_glass if action else True,
"requires_confirmation": True,
"confirmation_phrase": confirmation_phrase(ns, md),
"mutation_ledger": [asdict(entry) for entry in _mutation_ledger(ns, md)],
"authorization": decision.to_dict(),
"operator_authorization": operator,
"restart_hook": hook,
"execution_enabled": False,
"raw_process_kill_exposed": False,
"known_namespaces": list(KNOWN_NAMESPACES),
"post_restart_verification_required": True,
}
# --- Gate -------------------------------------------------------------------
def assess_restart_request(
namespace: str,
mode: str = MODE_RESTART,
*,
principal: console_authz.Principal | None = None,
confirmation: str | None = None,
contamination_marker: dict[str, Any] | None = None,
env: dict[str, str] | None = None,
) -> dict[str, Any]:
"""Decide whether a restart request may proceed to the host hook.
Every gate must pass. The first failure wins and is reported with a stable
reason code; a pass never means "restarted", only "may be handed to the
configured host hook".
"""
ns = _clean(namespace)
md = _clean(mode)
action_id = ACTION_FOR_MODE.get(md, ACTION_RESTART_NAMESPACE)
preview = build_restart_preview(ns, md, principal=principal, env=env)
def refuse(reason_code: str, detail: str) -> dict[str, Any]:
return {
"allowed": False,
"gates_passed": False,
"reason_code": reason_code,
"detail": detail,
"action_id": action_id,
"namespace": ns,
"mode": md,
"preview": preview,
"execution_enabled": False,
}
scope_error = _validate_scope(ns, md)
if scope_error is not None:
return refuse(*scope_error)
# Authority is checked as an authorization decision, not an execution
# grant. ``for_execution=True`` asks "may the console perform this write?",
# and the answer here is permanently no: step 2 of the ledger is a request
# to the host supervisor, so the console's Phase 2 execution gate is not
# the relevant gate. Every branch below keeps ``execution_enabled`` False
# and :func:`execute_restart` never touches a process.
decision = console_authz.authorize(action_id, principal)
if not decision.allowed:
return refuse(DENY_UNAUTHORIZED, decision.detail)
if not _clean(confirmation):
return refuse(
DENY_CONFIRMATION_MISSING,
(
"Type the confirmation phrase "
f"{preview['confirmation_phrase']!r} to proceed."
),
)
if not confirmation_matches(ns, md, confirmation):
return refuse(
DENY_CONFIRMATION_MISMATCH,
(
"Confirmation does not name this namespace and mode; expected "
f"{preview['confirmation_phrase']!r}."
),
)
operator = preview["operator_authorization"]
if not operator["authorized"]:
return refuse(
DENY_OPERATOR_AUTHORIZATION,
(
"Host daemon maintenance requires out-of-band operator "
"authorization via "
f"{runtime_recovery_guard.OPERATOR_AUTHORIZATION_ENV}."
),
)
# #630's task-scoped gate deliberately lets a contaminated worker keep
# commenting and handing off. Restart is stricter and unconditional: a
# runtime already contaminated by a manual kill must be reconciled before
# it is restarted again, or the restart just launders the contamination.
if contamination_marker and not contamination_marker.get(
"cleared_by_reconciler"
):
return refuse(
DENY_CONTAMINATED_RUNTIME,
(
"A live contamination marker is present; clear it through the "
"reconciler path before restarting."
),
)
hook = preview["restart_hook"]
if not hook["configured"]:
return refuse(
DENY_HOOK_NOT_CONFIGURED,
(
"No host-managed restart hook is configured "
f"({RESTART_HOOK_ENV}). The console will not fall back to a "
"process kill."
),
)
return {
"allowed": True,
"gates_passed": True,
"reason_code": ALLOW_HOST_ACTION_REQUIRED,
"detail": (
"Every gate passed. The restart must be performed by the "
"configured host supervisor; the console does not signal the "
"process."
),
"action_id": action_id,
"namespace": ns,
"mode": md,
"preview": preview,
"execution_enabled": False,
"console_executes": False,
"console_active_phase": console_authz.ACTIVE_PHASE,
}
# --- Execution --------------------------------------------------------------
def execute_restart(
namespace: str,
mode: str = MODE_RESTART,
*,
principal: console_authz.Principal | None = None,
confirmation: str | None = None,
contamination_marker: dict[str, Any] | None = None,
env: dict[str, str] | None = None,
request_id: str | None = None,
session_id: str | None = None,
) -> dict[str, Any]:
"""Run every gate, audit the outcome, and hand off to the host.
This function never spawns a process, never sends a signal, and never
builds a command line. ``success`` is False in both directions: a refused
request is refused, and an authorized request still requires the host
supervisor to act.
"""
assessment = assess_restart_request(
namespace,
mode,
principal=principal,
confirmation=confirmation,
contamination_marker=contamination_marker,
env=env,
)
action_id = assessment["action_id"]
allowed = assessment["allowed"]
audit = console_audit.record_event(
action_id=action_id,
result=(
console_audit.RESULT_ALLOWED if allowed
else console_audit.RESULT_DENIED
),
principal=principal,
target={"namespace": assessment["namespace"], "mode": assessment["mode"]},
reason_code=assessment["reason_code"],
detail=assessment["detail"],
request_id=request_id,
session_id=session_id,
metadata={
"gates_passed": assessment["gates_passed"],
"process_kill_executed": False,
"post_restart_verification_required": True,
},
)
return {
"success": False,
"outcome": (
ALLOW_HOST_ACTION_REQUIRED if allowed else assessment["reason_code"]
),
"allowed": allowed,
"detail": assessment["detail"],
"namespace": assessment["namespace"],
"mode": assessment["mode"],
"action_id": action_id,
"process_kill_executed": False,
"host_hook": assessment["preview"]["restart_hook"],
"next_action": (
"Have the host supervisor perform the restart, then call "
"verify_post_restart_health with live client-namespace evidence "
"before claiming a clean session."
if allowed
else assessment["detail"]
),
"assessment": assessment,
"audit": audit,
}
# --- Contamination classification -------------------------------------------
def classify_restart_command(
command: str | None,
*,
mcp_pids: list[Any] | tuple[Any, ...] | None = None,
session_id: str | None = None,
remote: str | None = None,
role: str | None = None,
) -> dict[str, Any]:
"""Classify an operator-proposed recovery command (#630 AC2).
A manual ``pkill``/``kill`` of the MCP daemon is contamination, not a
restart. When contaminating, a durable marker is returned so downstream
gated mutations and clean claims fail closed.
"""
classification = runtime_recovery_guard.classify_recovery_command(
command, mcp_pids=mcp_pids
)
contaminating = bool(classification.get("contamination"))
marker = None
if contaminating:
marker = runtime_recovery_guard.build_contamination_record(
reason_class=(
classification.get("reason_class")
or runtime_recovery_guard.REASON_MANUAL_DAEMON_KILL
),
command_redacted=classification.get("redacted_command"),
session_id=session_id,
remote=remote,
role=role,
detail=(
"Manual daemon kill is forbidden; use the sanctioned "
f"{ACTION_RESTART_NAMESPACE} gated action instead."
),
)
return {
"contamination": contaminating,
"sanctioned": not contaminating and not classification.get("process_kill"),
"clean_claim_allowed": not contaminating,
"reason_class": classification.get("reason_class"),
"redacted_command": classification.get("redacted_command"),
"classification": classification,
"contamination_marker": marker,
"sanctioned_alternative": ACTION_RESTART_NAMESPACE,
}
# --- Post-restart health verification ---------------------------------------
def verify_post_restart_health(
namespace: str,
*,
probe_result: dict[str, Any] | None = None,
probe_source: str | None = None,
registered_tools: list[str] | tuple[str, ...] | None = None,
required_tool: str | None = None,
profile: str | None = None,
) -> dict[str, Any]:
"""Require live proof a namespace is callable before any clean claim (AC3).
Static registration is not proof and neither is an offline subprocess
probe: only ``probe_source=client_namespace`` evidence can clear a
post-restart session for mutations.
"""
ns = _clean(namespace)
health = mcp_namespace_health.classify_namespace_probe(
ns,
required_tool=required_tool,
registered_tools=registered_tools,
probe_result=probe_result,
profile=profile,
probe_source=probe_source,
)
healthy = bool(health.get("healthy"))
proven = bool(health.get("ide_namespace_proven"))
if healthy and proven:
status = HEALTH_CLEAN
elif healthy:
status = HEALTH_UNPROVEN
else:
status = HEALTH_UNHEALTHY
reasons = list(health.get("reasons") or [])
if status == HEALTH_UNPROVEN:
reasons.append(
"Namespace reported healthy without live client-namespace "
"evidence; a post-restart clean claim requires "
f"probe_source={mcp_namespace_health.PROBE_SOURCE_CLIENT!r}."
)
return {
"namespace": ns,
"status": status,
"healthy": healthy,
"ide_namespace_proven": proven,
"clean_claim_allowed": status == HEALTH_CLEAN,
"mutations_allowed": status == HEALTH_CLEAN,
"reasons": reasons,
"health": health,
}
# --- Policy surface ---------------------------------------------------------
def restart_policy() -> dict[str, Any]:
"""Machine-readable description of the sanctioned restart contract."""
return {
"policy_version": 1,
"modes": list(MODES),
"actions": [ACTION_RESTART_NAMESPACE, ACTION_RELOAD_NAMESPACE],
"known_namespaces": list(KNOWN_NAMESPACES),
"fleet_scope_permitted": False,
"console_executes_process_kill": False,
"raw_kill_exposed": False,
"requires_confirmation": True,
"confirmation_binds_namespace": True,
"operator_authorization_env": (
runtime_recovery_guard.OPERATOR_AUTHORIZATION_ENV
),
"restart_hook_env": RESTART_HOOK_ENV,
"manual_kill_classified_as": runtime_recovery_guard.CONTAMINATION_KIND,
"post_restart_clean_claim_requires": (
mcp_namespace_health.PROBE_SOURCE_CLIENT
),
"audit_required": True,
"silent_auto_restart_permitted": False,
}
-307
View File
@@ -1,307 +0,0 @@
"""HTML views for the system-health dashboard (#639).
Renders the read-only :class:`~webui.system_health.SystemHealthSnapshot`
produced by the Phase 1 system-health API (#634). The page offers no restart,
reload, or process-kill control: those are Phase 2 work, and manual process
kills are the contamination path #630 exists to prevent.
Every free-text field passes through :func:`webui.system_health.redact` before
it reaches HTML, so a probe detail that captured a token or a credentialed URL
cannot leak through the dashboard even though the API redacts it already.
"""
from __future__ import annotations
import html
from webui.system_health import (
STATUS_DEGRADED,
STATUS_DOWN,
STATUS_OK,
STATUS_SKIPPED,
STATUS_UNPROVEN,
DependencyProbe,
SystemHealthSnapshot,
redact,
)
_STATUS_BADGE_CLASS = {
STATUS_OK: "badge-health-ok",
STATUS_DEGRADED: "badge-health-degraded",
STATUS_DOWN: "badge-health-down",
STATUS_SKIPPED: "badge-health-skipped",
STATUS_UNPROVEN: "badge-health-unproven",
}
def _safe(value: object) -> str:
"""Escape free text for HTML after redacting anything secret-shaped.
Use this for every value that can carry arbitrary text probe details,
reasons, probe errors because those are where a credential could ride
along.
"""
return html.escape(redact(str(value)))
def _esc(value: object) -> str:
"""Escape a structured field for HTML without redacting it.
Commit SHAs, probe names, statuses, and timestamps are enumerated or
machine-generated, never credential-bearing. They must not go through
:func:`redact`: its opaque-token rule matches any 32-plus-character run,
so a 40-character git SHA would render as ``[redacted]`` and the parity
view the one thing an operator reads this page for would be blank.
"""
return html.escape(str(value))
def _status_badge(status: str) -> str:
css = _STATUS_BADGE_CLASS.get(status, "badge-health-unproven")
return f'<span class="badge {css}">{_esc(status)}</span>'
def _reason_list(reasons: tuple[str, ...], *, empty: str) -> str:
if not reasons:
return f"<p class='muted'>{html.escape(empty)}</p>"
items = "".join(f"<li>{_safe(reason)}</li>" for reason in reasons)
return f"<ul class='reasons'>{items}</ul>"
def _readiness_card(snapshot: SystemHealthSnapshot) -> str:
"""Overall readiness.
``ready`` and ``readiness_complete`` are shown separately on purpose: a
snapshot whose required probes never ran is not the same as one that ran
them and passed, and collapsing the two would render an unproven green.
"""
if snapshot.ready and snapshot.readiness_complete:
headline = "Ready"
elif snapshot.ready:
headline = "Ready (incomplete evidence)"
else:
headline = "Not ready"
return (
"<section class='health-card'>"
f"<h3>Readiness {_status_badge(snapshot.status)}</h3>"
f"<p class='health-headline'>{html.escape(headline)}</p>"
"<table class='detail'>"
f"<tr><th>Service</th><td><code>{_esc(snapshot.service)}</code></td></tr>"
f"<tr><th>Mode</th><td>{_esc(snapshot.mode)}</td></tr>"
f"<tr><th>Ready</th><td>{_esc(snapshot.ready)}</td></tr>"
"<tr><th>Readiness evidence complete</th>"
f"<td>{_esc(snapshot.readiness_complete)}</td></tr>"
"<tr><th>Deep probes requested</th>"
f"<td>{_esc(snapshot.deep_probes_requested)}</td></tr>"
f"<tr><th>Observed at</th><td><code>{_esc(snapshot.timestamp)}</code></td></tr>"
"</table>"
"<h4>Readiness reasons</h4>"
f"{_reason_list(snapshot.readiness_reasons, empty='No readiness objections recorded.')}"
"</section>"
)
def _version_card(snapshot: SystemHealthSnapshot) -> str:
version = snapshot.version
uptime_hours = snapshot.uptime_seconds / 3600.0
known = (
"resolved"
if version.known
else "unresolved — version fields could not be read from the checkout"
)
schema = version.control_plane_schema_version
return (
"<section class='health-card'>"
"<h3>Version and uptime</h3>"
"<table class='detail'>"
f"<tr><th>Git SHA</th><td><code>{_esc(version.git_sha or 'unknown')}</code></td></tr>"
"<tr><th>Git describe</th>"
f"<td><code>{_esc(version.git_describe or 'unknown')}</code></td></tr>"
"<tr><th>Control-plane schema</th>"
f"<td>{_esc(schema if schema is not None else 'unknown')}</td></tr>"
f"<tr><th>Python</th><td><code>{_esc(version.python_version)}</code></td></tr>"
f"<tr><th>Version status</th><td>{html.escape(known)}</td></tr>"
f"<tr><th>Started at</th><td><code>{_esc(snapshot.started_at)}</code></td></tr>"
"<tr><th>Uptime</th>"
f"<td>{snapshot.uptime_seconds:.3f}s ({uptime_hours:.2f}h)</td></tr>"
"</table>"
"</section>"
)
def _dependency_rows(probes: tuple[DependencyProbe, ...]) -> str:
if not probes:
return "<p class='muted'>No dependency probes were reported.</p>"
rows = []
for probe in probes:
latency = (
f"{probe.latency_ms:.1f} ms" if probe.latency_ms is not None else "n/a"
)
rows.append(
"<tr>"
f"<td><code>{_esc(probe.name)}</code></td>"
f"<td>{_esc(probe.kind)}</td>"
f"<td>{_status_badge(probe.status)}</td>"
f"<td>{_esc('required' if probe.required else 'optional')}</td>"
f"<td>{html.escape(latency)}</td>"
f"<td>{_safe(probe.detail)}</td>"
"</tr>"
)
return (
"<table class='registry'><thead><tr>"
"<th>Dependency</th><th>Kind</th><th>Status</th><th>Requirement</th>"
"<th>Latency</th><th>Detail</th>"
"</tr></thead><tbody>"
f"{''.join(rows)}</tbody></table>"
)
def _dependency_card(snapshot: SystemHealthSnapshot) -> str:
degraded = [probe for probe in snapshot.dependencies if probe.ran and not probe.healthy]
not_run = [probe for probe in snapshot.dependencies if not probe.ran]
banner = ""
if degraded:
names = ", ".join(sorted(probe.name for probe in degraded))
banner += (
"<div class='stub health-degraded'><p><strong>Degraded dependencies:</strong> "
f"{_esc(names)}</p></div>"
)
if not_run:
names = ", ".join(sorted(probe.name for probe in not_run))
banner += (
"<div class='stub'><p><strong>Not probed:</strong> "
f"{_esc(names)} — these contribute no evidence and are not counted "
"as healthy.</p></div>"
)
return (
"<section class='health-card'>"
"<h3>Dependencies</h3>"
f"{banner}"
f"{_dependency_rows(snapshot.dependencies)}"
"<p class='muted'>Details are redacted at the API boundary and again "
"before rendering; credentials are never displayed.</p>"
"</section>"
)
def _namespace_card(snapshot: SystemHealthSnapshot) -> str:
if not snapshot.mcp_namespaces:
body = "<p class='muted'>No MCP namespaces are declared.</p>"
else:
rows = []
for entry in snapshot.mcp_namespaces:
rows.append(
"<tr>"
f"<td><code>{_esc(entry.get('namespace'))}</code></td>"
f"<td><code>{_esc(entry.get('required_tool'))}</code></td>"
f"<td>{_status_badge(str(entry.get('status') or STATUS_UNPROVEN))}</td>"
f"<td>{_esc(entry.get('ide_namespace_proven'))}</td>"
f"<td>{_safe(entry.get('reason'))}</td>"
"</tr>"
)
body = (
"<table class='registry'><thead><tr>"
"<th>Namespace</th><th>Required tool</th><th>Status</th>"
"<th>IDE-proven</th><th>Reason</th>"
"</tr></thead><tbody>"
f"{''.join(rows)}</tbody></table>"
)
return (
"<section class='health-card'>"
"<h3>MCP namespaces</h3>"
f"{body}"
"<p class='muted'>The web process runs outside the IDE-managed MCP "
"client, so namespace health is reported as unproven rather than "
"guessed (#543).</p>"
"</section>"
)
def _stale_runtime_card(snapshot: SystemHealthSnapshot) -> str:
stale = snapshot.stale_runtime
if stale.stale:
warning = (
"<div class='stub health-stale'><p><strong>Stale runtime:</strong> "
"the running code, the checkout, and the remote-tracking commit "
"disagree. Capability gates may be evaluating obsolete code — "
"do not treat this runtime as mutation-safe.</p></div>"
)
elif not stale.determinable:
warning = (
"<div class='stub health-stale'><p><strong>Staleness "
"indeterminate:</strong> parity could not be proven, so this "
"runtime is not reported as mutation-safe.</p></div>"
)
else:
warning = ""
return (
"<section class='health-card'>"
"<h3>Stale-runtime parity</h3>"
f"{warning}"
"<table class='detail'>"
"<tr><th>Daemon head</th>"
f"<td><code>{_esc(stale.daemon_head or 'unknown')}</code></td></tr>"
"<tr><th>Checkout head</th>"
f"<td><code>{_esc(stale.checkout_head or 'unknown')}</code></td></tr>"
"<tr><th>Remote head</th>"
f"<td><code>{_esc(stale.remote_head or 'unknown')}</code></td></tr>"
f"<tr><th>Stale</th><td>{_esc(stale.stale)}</td></tr>"
f"<tr><th>Determinable</th><td>{_esc(stale.determinable)}</td></tr>"
f"<tr><th>Mutation safe</th><td>{_esc(stale.mutation_safe)}</td></tr>"
"</table>"
f"{_reason_list(stale.reasons, empty='Runtime, checkout, and remote agree.')}"
"</section>"
)
def _probe_error_card(snapshot: SystemHealthSnapshot) -> str:
if not snapshot.probe_errors:
return ""
return (
"<section class='health-card'>"
"<h3>Probe errors</h3>"
f"{_reason_list(snapshot.probe_errors, empty='')}"
"</section>"
)
def _recovery_card() -> str:
"""Sanctioned recovery pointers only — never a manual process kill (#630)."""
return (
"<section class='health-card'>"
"<h3>Recovery</h3>"
"<p class='muted'>This dashboard is read-only. Restart and reload "
"controls arrive in Phase 2 (#642); until then recovery runs through "
"the sanctioned client reconnect / operator restart path.</p>"
"<ul class='reasons'>"
"<li><a href='/runtime'>Runtime and session view</a> — active profile, "
"workflow hashes, and shell health.</li>"
"<li>Reconnect the MCP client from the IDE, then re-run the blocked "
"cycle. Never kill the daemon process manually: unmanaged kills are "
"recorded as runtime contamination (#630).</li>"
"<li>See <code>docs/webui-local-dev.md</code> for the documented "
"recovery sequence.</li>"
"</ul>"
"</section>"
)
def render_system_health_page(snapshot: SystemHealthSnapshot) -> str:
"""Render the full system-health dashboard body."""
return (
"<h2>System health</h2>"
"<p class='meta'>Read-only view of the Phase 1 system-health API "
"(<code>/api/v1/system/health</code>). Reload this page to refresh; "
"nothing here polls or mutates on your behalf.</p>"
f"{_readiness_card(snapshot)}"
f"{_stale_runtime_card(snapshot)}"
f"{_version_card(snapshot)}"
f"{_dependency_card(snapshot)}"
f"{_namespace_card(snapshot)}"
f"{_probe_error_card(snapshot)}"
f"{_recovery_card()}"
)