Compare commits

..
Author SHA1 Message Date
sysadmin 38506a753f Merge remote-tracking branch 'prgs/master' into feat/issue-651-usage-cost-analytics
# Conflicts:
#	docs/webui-authz-audit.md
#	webui/console_authz.py
2026-07-24 08:33:35 -04:00
sysadminandClaude Opus 4.8 b179610e7f fix(webui): remediate PR #876 REQUEST_CHANGES for analytics (#651)
Address review blockers and medium/low findings:

- F1: HTML-escape all dynamic analytics fields (html.escape quote=True)
- F2: Gate POST /api/v1/analytics/usage through console_authz
  record_analytics_usage (fail closed for unauthenticated / Phase 1)
- F3: Enforce usage_events retention (max rows + max age)
- F4: Coerce None remote/org/repo to empty strings in load_analytics

Add tests for XSS escaping, unauthorized ingest deny, retention, and
None scope coercion. Document authz + retention in analytics guide.

Co-Authored-By: Claude Opus 4.8 (1M context) <[email protected]>
2026-07-24 08:11:12 -04:00
jcwalker3 0041542fe2 Merge branch 'master' into feat/issue-651-usage-cost-analytics 2026-07-24 07:03:24 -05:00
jcwalker3 1f144705e2 Merge branch 'master' into feat/issue-651-usage-cost-analytics 2026-07-24 06:55:03 -05:00
sysadmin 5c5c1fdf77 feat(webui): model usage, token cost, latency, and performance analytics (Closes #651) 2026-07-24 07:16:44 -04:00
14 changed files with 1450 additions and 1504 deletions
+231 -1
View File
@@ -31,7 +31,7 @@ from typing import Any, Iterator, Sequence
import dependency_graph
SCHEMA_VERSION = 4
SCHEMA_VERSION = 5
# Assignable work kinds only — raw monitoring incidents are never work items.
WORK_KINDS = frozenset({"issue", "pr"})
@@ -186,6 +186,34 @@ CREATE INDEX IF NOT EXISTS idx_dependency_edges_target
ON dependency_edges(remote, org, repo, target_kind, target_number);
CREATE INDEX IF NOT EXISTS idx_assignments_session ON assignments(session_id, status);
CREATE INDEX IF NOT EXISTS idx_incident_gitea ON incident_links(gitea_org, gitea_repo, gitea_issue_number);
-- Model usage, token cost, latency, and performance events (#651)
CREATE TABLE IF NOT EXISTS usage_events (
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
session_id TEXT,
remote TEXT NOT NULL DEFAULT 'dadeschools',
org TEXT NOT NULL DEFAULT '',
repo TEXT NOT NULL DEFAULT '',
project_id TEXT,
role TEXT NOT NULL DEFAULT 'unknown',
model TEXT NOT NULL DEFAULT 'unknown',
issue_number INTEGER,
pr_number INTEGER,
stage TEXT NOT NULL DEFAULT 'unknown',
input_tokens INTEGER,
output_tokens INTEGER,
total_tokens INTEGER,
estimated_cost_usd REAL,
latency_ms INTEGER,
duration_ms INTEGER,
status TEXT NOT NULL DEFAULT 'success',
metadata TEXT,
created_at TEXT NOT NULL
);
CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo);
CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model);
CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage);
"""
@@ -339,6 +367,7 @@ class ControlPlaneDB:
self._migrate_incident_links_null_scope(conn)
self._migrate_lease_lifecycle_columns(conn)
self._migrate_session_ownership_columns(conn)
self._migrate_usage_events_table(conn)
conn.execute(
"INSERT OR REPLACE INTO schema_meta(key, value) VALUES (?, ?)",
("schema_version", str(SCHEMA_VERSION)),
@@ -518,6 +547,207 @@ class ControlPlaneDB:
f"UPDATE incident_links SET {col} = '' WHERE {col} IS NULL"
)
def _migrate_usage_events_table(self, conn: sqlite3.Connection) -> None:
"""Create usage_events table and indexes if they do not exist (#651)."""
conn.execute("""
CREATE TABLE IF NOT EXISTS usage_events (
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
session_id TEXT,
remote TEXT NOT NULL DEFAULT 'dadeschools',
org TEXT NOT NULL DEFAULT '',
repo TEXT NOT NULL DEFAULT '',
project_id TEXT,
role TEXT NOT NULL DEFAULT 'unknown',
model TEXT NOT NULL DEFAULT 'unknown',
issue_number INTEGER,
pr_number INTEGER,
stage TEXT NOT NULL DEFAULT 'unknown',
input_tokens INTEGER,
output_tokens INTEGER,
total_tokens INTEGER,
estimated_cost_usd REAL,
latency_ms INTEGER,
duration_ms INTEGER,
status TEXT NOT NULL DEFAULT 'success',
metadata TEXT,
created_at TEXT NOT NULL
);
""")
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo);")
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model);")
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage);")
# #651 retention: cap growth so unauthenticated or high-volume ingest
# cannot DoS the control-plane DB (PR #876 F3). Applied after every write.
USAGE_EVENTS_MAX_ROWS = 10_000
USAGE_EVENTS_RETENTION_DAYS = 90
def record_usage_event(
self,
*,
session_id: str | None = None,
remote: str = "dadeschools",
org: str = "",
repo: str = "",
project_id: str | None = None,
role: str = "unknown",
model: str = "unknown",
issue_number: int | None = None,
pr_number: int | None = None,
stage: str = "unknown",
input_tokens: int | None = None,
output_tokens: int | None = None,
total_tokens: int | None = None,
estimated_cost_usd: float | None = None,
latency_ms: int | None = None,
duration_ms: int | None = None,
status: str = "success",
metadata: str | dict[str, Any] | None = None,
created_at: str | None = None,
) -> int:
"""Record a model usage, token cost, latency, or stage performance event (#651)."""
ts = created_at or _ts()
meta_str: str | None = None
if metadata is not None:
from webui import console_redaction
redacted_meta = console_redaction.redact_payload(metadata)
if isinstance(redacted_meta, str):
meta_str = redacted_meta
else:
try:
meta_str = json.dumps(redacted_meta, default=str)
except Exception:
meta_str = str(redacted_meta)
if total_tokens is None and (input_tokens is not None or output_tokens is not None):
total_tokens = (input_tokens or 0) + (output_tokens or 0)
with self._tx(immediate=True) as conn:
cursor = conn.execute(
"""
INSERT INTO usage_events (
session_id, remote, org, repo, project_id, role, model,
issue_number, pr_number, stage, input_tokens, output_tokens,
total_tokens, estimated_cost_usd, latency_ms, duration_ms,
status, metadata, created_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
""",
(
session_id,
remote,
org,
repo,
project_id,
role,
model,
issue_number,
pr_number,
stage,
input_tokens,
output_tokens,
total_tokens,
estimated_cost_usd,
latency_ms,
duration_ms,
status,
meta_str,
ts,
),
)
usage_id = cursor.lastrowid
self._enforce_usage_events_retention(conn)
return usage_id
def _enforce_usage_events_retention(self, conn: sqlite3.Connection) -> None:
"""Drop aged and excess usage_events rows (PR #876 F3)."""
# Age-based: ISO-8601 UTC timestamps compare lexicographically.
cutoff = (
datetime.now(timezone.utc)
- timedelta(days=int(self.USAGE_EVENTS_RETENTION_DAYS))
).strftime("%Y-%m-%dT%H:%M:%SZ")
conn.execute(
"DELETE FROM usage_events WHERE created_at < ?",
(cutoff,),
)
# Count-based: keep the newest USAGE_EVENTS_MAX_ROWS by usage_id.
max_rows = int(self.USAGE_EVENTS_MAX_ROWS)
if max_rows > 0:
conn.execute(
"""
DELETE FROM usage_events
WHERE usage_id NOT IN (
SELECT usage_id FROM usage_events
ORDER BY usage_id DESC
LIMIT ?
)
""",
(max_rows,),
)
def query_usage_events(
self,
*,
remote: str | None = None,
org: str | None = None,
repo: str | None = None,
project_id: str | None = None,
role: str | None = None,
model: str | None = None,
issue_number: int | None = None,
pr_number: int | None = None,
stage: str | None = None,
session_id: str | None = None,
limit: int = 500,
offset: int = 0,
) -> list[dict[str, Any]]:
"""Query stored usage events matching filters (#651)."""
conditions = []
params = []
if remote:
conditions.append("remote = ?")
params.append(remote)
if org:
conditions.append("org = ?")
params.append(org)
if repo:
conditions.append("repo = ?")
params.append(repo)
if project_id:
conditions.append("project_id = ?")
params.append(project_id)
if role:
conditions.append("role = ?")
params.append(role)
if model:
conditions.append("model = ?")
params.append(model)
if issue_number is not None:
conditions.append("issue_number = ?")
params.append(issue_number)
if pr_number is not None:
conditions.append("pr_number = ?")
params.append(pr_number)
if stage:
conditions.append("stage = ?")
params.append(stage)
if session_id:
conditions.append("session_id = ?")
params.append(session_id)
where_clause = f"WHERE {' AND '.join(conditions)}" if conditions else ""
sql = f"""
SELECT * FROM usage_events
{where_clause}
ORDER BY usage_id ASC
LIMIT ? OFFSET ?
"""
params.extend([limit, offset])
with self._tx(immediate=False) as conn:
cursor = conn.execute(sql, params)
rows = cursor.fetchall()
return [dict(row) for row in rows]
# ── sessions ──────────────────────────────────────────────────────────
def upsert_session(
@@ -0,0 +1,143 @@
# Model Usage, Token Cost, Latency, and Workflow Analytics (Phase 4)
- **Tracking Issue:** [#651](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
- **Console Surface:** `/analytics`, `/api/v1/analytics`, `/api/v1/analytics/usage`
## 1. Overview
The Web Console Analytics module provides durable, aggregate visibility into **model usage, token cost, latency percentiles, and workflow-stage performance** across projects, worker roles, AI models, issues, and PRs.
### Non-Goals
- No mandatory client-side telemetry that leaks prompts or secret keys.
- No third-party payment provider or billing integration.
- No automatic model routing changes without controller policy (#647).
---
## 2. Event Schema (`usage_events`)
Usage metrics are stored in the control-plane database under table `usage_events`.
| Column | Type | Description |
|---|---|---|
| `usage_id` | `INTEGER` | Primary key (autoincrement) |
| `session_id` | `TEXT` | Optional active session identifier |
| `remote` | `TEXT` | Known Gitea instance (`dadeschools` or `prgs`) |
| `org` | `TEXT` | Repository owner / organization |
| `repo` | `TEXT` | Repository name |
| `project_id` | `TEXT` | Optional project identifier |
| `role` | `TEXT` | Active worker role (`author`, `reviewer`, `merger`, `reconciler`, `controller`) |
| `model` | `TEXT` | LLM model identifier (e.g. `gemini-3.6-flash`, `claude-3-5-sonnet`) |
| `issue_number` | `INTEGER` | Correlated Gitea issue number (optional) |
| `pr_number` | `INTEGER` | Correlated Gitea PR number (optional) |
| `stage` | `TEXT` | Workflow stage (`preflight`, `implementation`, `review`, `merge`, `reconciliation`) |
| `input_tokens` | `INTEGER` | Input token count (optional / nullable) |
| `output_tokens` | `INTEGER` | Output token count (optional / nullable) |
| `total_tokens` | `INTEGER` | Total token count (optional / nullable) |
| `estimated_cost_usd` | `REAL` | Estimated USD cost (optional / nullable) |
| `latency_ms` | `INTEGER` | Request latency in milliseconds (optional / nullable) |
| `duration_ms` | `INTEGER` | Stage execution duration in milliseconds (optional / nullable) |
| `status` | `TEXT` | Outcome status (`success`, `failure`, `timeout`) |
| `metadata` | `TEXT` | Redacted metadata or summary string |
| `created_at` | `TEXT` | ISO 8601 UTC timestamp |
---
## 3. Handling of Missing Data ("Unknown" vs. Zero Fabrication)
To ensure operational metrics accurately reflect evidence:
- **Untracked or missing metrics are displayed as `Unknown`**, never zero-fabricated.
- If an event omits `estimated_cost_usd`, `latency_ms`, or token counts, the aggregator marks those fields as missing (`None`) rather than defaulting to `0` or `$0.00`.
- Summary tables and KPI cards explicitly indicate when data is unmeasured or partially reported.
---
## 4. Redaction & Security Rules
Per `#633` security policy:
- Free-text fields (`metadata`, `prompt_summary`, `session_id`) are run through `console_redaction.redact_text` before persistence and output serialization.
- Secret tokens, keychain commands, authorization headers, passwords, and JWTs are stripped automatically.
---
## 5. Opt-in Instrumentation Guide
Applications, MCP servers, and background sessions can report usage metrics through either Python API or HTTP ingestion.
### Python Ingestion
```python
from webui.analytics_loader import record_usage
record_usage(
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
role="author",
model="gemini-3.6-flash",
issue_number=651,
stage="implementation",
input_tokens=1420,
output_tokens=380,
total_tokens=1800,
estimated_cost_usd=0.00045,
latency_ms=320,
duration_ms=4500,
status="success",
metadata={"note": "Implementation of analytics module"},
)
```
### HTTP Ingestion API (authorized write)
`POST /api/v1/analytics/usage` is a **gated write**. It runs through
`console_authz` action `record_analytics_usage` (operator+, Phase 2 execution).
Unauthenticated or phase-inactive requests receive **403** and do not write.
Prefer in-process `record_usage` for MCP / session instrumentation.
```http
POST /api/v1/analytics/usage HTTP/1.1
Content-Type: application/json
# Requires authenticated principal with record_analytics_usage execution enabled
{
"remote": "dadeschools",
"org": "Scaled-Tech-Consulting",
"repo": "Gitea-Tools",
"role": "author",
"model": "gemini-3.6-flash",
"issue_number": 651,
"stage": "implementation",
"input_tokens": 1420,
"output_tokens": 380,
"total_tokens": 1800,
"estimated_cost_usd": 0.00045,
"latency_ms": 320,
"duration_ms": 4500,
"status": "success",
"metadata": "Analytics schema landed"
}
```
### Retention
`usage_events` is retained with hard caps applied on every write:
| Limit | Default |
|---|---|
| Max rows | 10,000 (`ControlPlaneDB.USAGE_EVENTS_MAX_ROWS`) |
| Max age | 90 days (`ControlPlaneDB.USAGE_EVENTS_RETENTION_DAYS`) |
Older rows (by `created_at`) and excess oldest rows (by `usage_id`) are deleted
after each insert so unbounded growth / DoS-by-volume cannot fill the DB.
---
## 6. Querying Analytics API
```http
GET /api/v1/analytics?role=author&stage=implementation HTTP/1.1
```
Returns `AnalyticsSnapshot` JSON containing aggregations (`by_model`, `by_stage`, `by_role`, `by_work_item`, `by_project`) and latency percentiles (`p50`, `p90`, `p95`, `p99`).
+1
View File
@@ -91,6 +91,7 @@ already define, and a regression test asserts each mapping matches.
| `close_pr` | controller | privileged | `gitea.pr.close` | Yes | No | No | 3 |
| `merge_pr` | controller | privileged | `gitea.pr.merge` | Yes | **Yes** | **Yes** | 3 |
| `delete_branch` | admin | destructive | `gitea.branch.delete` | Yes | **Yes** | **Yes** | 3 |
| `record_analytics_usage` | operator | gated_write | `runtime.record_analytics_usage` | Yes | No | No | 2 |
| `system.reload_namespace` | controller | privileged | `runtime.reload_namespace` | Yes | No | No | 2 |
| `system.restart_namespace` | admin | destructive | `runtime.restart_namespace` | Yes | **Yes** | **Yes** | 2 |
-47
View File
@@ -349,53 +349,6 @@ health, workflow/schema SHA-256 hashes, and stale-runtime warnings when the
checkout is behind merged safety-gate changes. Restart guidance links to #420;
no tokens or MCP restart actions are exposed.
## Inventory API (#636)
`GET /api/v1/inventory` returns one versioned, read-only snapshot that unifies
what the lease (#433), worktree (#432), and runtime (#430) MVP views each show
separately, so traffic-control and recovery consumers read the same source.
`GET /api/v1/inventory/{section}` returns a single section under the identical
schema (`sessions`, `leases`, `locks`, `worktrees`, `namespaces`); an unknown
section is a `404` with `error: unknown_section`. Both routes are `GET`-only.
Each section carries its own `status` (`ok` / `degraded` / `unavailable`), a
`reason` when not `ok`, and a `scan_ms`. A subsystem that cannot be read
degrades to a reasoned section; it never raises and never emits an empty list
that would read as "nothing is there".
### Field authority
Every section names where its rows came from; authorities are never blended.
| Section | Authority | Source |
|---|---|---|
| `sessions` | `control_plane_db` | #613 control-plane DB (`mode=ro`), authoritative for exclusive ownership (#600/#601) |
| `leases` | `control_plane_db` | #613 control-plane DB; degrades if the `work_items` table is absent |
| `locks` | `filesystem` | durable per-issue lock files (`issue_lock_store`) |
| `worktrees` | `filesystem` | registered git worktrees via the #432 hygiene scanner |
| `namespaces` | `filesystem` | the active profile serving this web process (others are not enumerable) |
The payload restates this map under `field_authority` for machine consumers.
### Ownership safety
`ownership_authority_complete` is true only when every ownership-bearing section
(`sessions`, `leases`, `locks`) read cleanly. While it is false, nothing is
reported as unowned and no collision is asserted from a degraded source —
absence of evidence is reported as absence of evidence, never as free work.
`collisions` surfaces detectable conflicts, each with a `kind` and `severity`:
`lock-without-worktree`, `duplicate-live-lock`, `live-lock-dead-owner` (unexpired
lease, dead pid — a #753 recovery candidate that would read as live to a naive
timestamp check), `stale-lock-dead-owner`, `expired-lock-live-owner` (the
#635/#760 daemon-pid deadlock), `concurrent-active-lease`, `active-lease-past-expiry`,
and `orphan-lease`. Collisions are emitted only from sections that read cleanly.
The control-plane DB is opened through a `mode=ro` URI so a read never creates
or migrates it; paths are collapsed against `$HOME`, URLs lose userinfo and
query strings, and credential-shaped values are redacted at the boundary. Lease
steal/release and worktree deletion are Phase 2+ and have no representation here.
## Workflow-event timeline (#637)
`GET /api/v1/timeline` is a read-only, versioned aggregation of workflow
+9
View File
@@ -510,6 +510,15 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
"permission": "gitea.issue.comment",
"role": "author",
},
# #651 console analytics ingest — control-plane DB write, not a Gitea API
# call. Authority comes from console RBAC (operator+) plus phase gating;
# permission string is a non-Gitea runtime capability so no Gitea profile
# can satisfy it by accident.
"record_analytics_usage": {
"permission": "runtime.record_analytics_usage",
"role": "author",
},
}
+1 -1
View File
@@ -36,7 +36,7 @@ class ControlPlaneDBTest(unittest.TestCase):
rows = dict(conn.execute("SELECT key, value FROM schema_meta").fetchall())
finally:
conn.close()
self.assertEqual(rows["schema_version"], "4")
self.assertEqual(rows["schema_version"], "5")
self.assertIn("DB coordinates", rows["architecture"])
self.assertIn("bridge", rows["architecture"].lower())
+252
View File
@@ -0,0 +1,252 @@
"""Unit and integration tests for Model Usage & Performance Analytics (#651)."""
from __future__ import annotations
import os
import tempfile
import unittest
from starlette.testclient import TestClient
import control_plane_db
from webui.analytics_loader import (
ANALYTICS_SCHEMA_VERSION,
compute_percentile,
load_analytics,
record_usage,
)
from webui.app import create_app
from webui import console_redaction
class AnalyticsLoaderTest(unittest.TestCase):
def setUp(self) -> None:
self.temp_dir = tempfile.TemporaryDirectory()
self.db_path = os.path.join(self.temp_dir.name, "test_control_plane.sqlite3")
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
self.db = control_plane_db.ControlPlaneDB(db_path=self.db_path)
def tearDown(self) -> None:
self.temp_dir.cleanup()
def test_compute_percentile(self) -> None:
self.assertIsNone(compute_percentile([], 50.0))
self.assertEqual(compute_percentile([100], 50.0), 100.0)
# 2 elements: [100, 200]
self.assertEqual(compute_percentile([100, 200], 50.0), 150.0)
# 100 elements: 1..100
vals = list(range(1, 101))
self.assertEqual(compute_percentile(vals, 50.0), 50.5)
self.assertAlmostEqual(compute_percentile(vals, 90.0), 90.1)
def test_record_and_aggregate_usage(self) -> None:
# Record event 1 (complete data)
u1 = record_usage(
db_path=self.db_path,
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
role="author",
model="gemini-3.6-flash",
issue_number=651,
stage="implementation",
input_tokens=1000,
output_tokens=500,
estimated_cost_usd=0.0015,
latency_ms=200,
duration_ms=3000,
metadata={"secret_key": "secret123", "note": "token=secret123"},
)
self.assertGreater(u1, 0)
# Record event 2 (missing tokens and cost -> unknown)
u2 = record_usage(
db_path=self.db_path,
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
role="reviewer",
model="claude-3-5-sonnet",
pr_number=846,
stage="review",
latency_ms=500,
duration_ms=6000,
)
self.assertGreater(u2, u1)
snapshot = load_analytics(
db_path=self.db_path,
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
)
self.assertTrue(snapshot.ok)
self.assertEqual(snapshot.schema_version, ANALYTICS_SCHEMA_VERSION)
self.assertEqual(snapshot.total_events, 2)
# Verify overall summary
summary = snapshot.overall_summary
self.assertEqual(summary.total_events, 2)
self.assertEqual(summary.events_with_tokens, 1)
self.assertEqual(summary.total_tokens, 1500)
self.assertEqual(summary.events_with_cost, 1)
self.assertEqual(summary.estimated_cost_usd, 0.0015)
self.assertEqual(summary.events_with_latency, 2)
self.assertEqual(summary.latency_p50_ms, 350.0)
# Verify missing data handling (AC 3: not zero-fabricated)
reviewer_model = snapshot.by_model.get("claude-3-5-sonnet")
self.assertIsNotNone(reviewer_model)
self.assertEqual(reviewer_model.total_events, 1)
self.assertEqual(reviewer_model.events_with_tokens, 0)
self.assertIsNone(reviewer_model.total_tokens)
self.assertEqual(reviewer_model.display_tokens, "Unknown")
self.assertEqual(reviewer_model.events_with_cost, 0)
self.assertIsNone(reviewer_model.estimated_cost_usd)
self.assertEqual(reviewer_model.display_cost, "Unknown")
# Verify redaction (AC 4)
e1 = [e for e in snapshot.events if e.usage_id == u1][0]
self.assertIsNotNone(e1.metadata)
self.assertNotIn("secret123", e1.metadata)
self.assertIn("[REDACTED]", e1.metadata)
def test_missing_db_fail_soft(self) -> None:
invalid_path = "/nonexistent_path_dir/db.sqlite3"
snapshot = load_analytics(db_path=invalid_path)
self.assertFalse(snapshot.ok)
self.assertIn("control_plane_db_unavailable", snapshot.reason)
self.assertEqual(snapshot.overall_summary.display_tokens, "Unknown")
class AnalyticsWebUITest(unittest.TestCase):
def setUp(self) -> None:
self.temp_dir = tempfile.TemporaryDirectory()
self.db_path = os.path.join(self.temp_dir.name, "test_webui.sqlite3")
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
self.app = create_app()
self.client = TestClient(self.app)
record_usage(
db_path=self.db_path,
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
role="author",
model="gemini-3.6-flash",
issue_number=651,
stage="implementation",
input_tokens=2000,
output_tokens=1000,
estimated_cost_usd=0.003,
latency_ms=150,
duration_ms=2500,
)
def tearDown(self) -> None:
self.temp_dir.cleanup()
def test_analytics_html_route(self) -> None:
response = self.client.get("/analytics")
self.assertEqual(response.status_code, 200)
self.assertIn("Model Usage & Performance Analytics", response.text)
self.assertIn("gemini-3.6-flash", response.text)
self.assertIn("3,000", response.text)
def test_analytics_api_route(self) -> None:
response = self.client.get("/api/v1/analytics")
self.assertEqual(response.status_code, 200)
data = response.json()
self.assertTrue(data["ok"])
self.assertEqual(data["total_events"], 1)
self.assertIn("gemini-3.6-flash", data["by_model"])
def test_analytics_ingest_unauthorized_denied(self) -> None:
"""F2: unauthenticated POST must not write the control-plane DB."""
payload = {
"remote": "dadeschools",
"org": "Scaled-Tech-Consulting",
"repo": "Gitea-Tools",
"role": "reviewer",
"model": "claude-3-5-sonnet",
"pr_number": 846,
"stage": "review",
"input_tokens": 500,
"output_tokens": 100,
"latency_ms": 400,
"metadata": "Review note token=secret456",
}
response = self.client.post("/api/v1/analytics/usage", json=payload)
self.assertEqual(response.status_code, 403)
res_json = response.json()
self.assertFalse(res_json.get("ok", True))
self.assertEqual(res_json.get("error"), "unauthorized")
authorization = res_json.get("authorization") or {}
self.assertFalse(authorization.get("allowed"))
self.assertFalse(authorization.get("execution_enabled"))
# No new row written
res2 = self.client.get("/api/v1/analytics")
self.assertEqual(res2.status_code, 200)
self.assertEqual(res2.json()["total_events"], 1)
def test_html_escapes_script_bearing_model_role_stage(self) -> None:
"""F1: stored XSS — dynamic model/role/stage must render escaped."""
xss = '<script>alert(1)</script>'
record_usage(
db_path=self.db_path,
remote="dadeschools",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
role=xss,
model=xss,
stage=xss,
issue_number=999,
status="success",
)
response = self.client.get("/analytics")
self.assertEqual(response.status_code, 200)
# Raw tag must not appear; escaped form must.
self.assertNotIn("<script>alert(1)</script>", response.text)
self.assertIn("&lt;script&gt;alert(1)&lt;/script&gt;", response.text)
def test_load_analytics_coerces_none_scope(self) -> None:
"""F4: None remote/org/repo become empty strings, never None."""
snapshot = load_analytics(db_path=self.db_path, remote=None, org=None, repo=None)
self.assertIsInstance(snapshot.remote, str)
self.assertIsInstance(snapshot.org, str)
self.assertIsInstance(snapshot.repo, str)
self.assertEqual(snapshot.remote, "")
self.assertEqual(snapshot.org, "")
self.assertEqual(snapshot.repo, "")
def test_usage_events_retention_max_rows(self) -> None:
"""F3: record_usage_event enforces USAGE_EVENTS_MAX_ROWS."""
db = control_plane_db.ControlPlaneDB(db_path=self.db_path)
original_max = db.USAGE_EVENTS_MAX_ROWS
try:
db.USAGE_EVENTS_MAX_ROWS = 3
for i in range(5):
db.record_usage_event(
remote="dadeschools",
org="org",
repo="repo",
role="author",
model=f"model-{i}",
stage="test",
)
rows = db.query_usage_events(limit=100)
self.assertLessEqual(len(rows), 3)
# Newest three retained
models = {r["model"] for r in rows}
self.assertEqual(models, {"model-2", "model-3", "model-4"})
finally:
db.USAGE_EVENTS_MAX_ROWS = original_max
if __name__ == "__main__":
unittest.main()
-468
View File
@@ -1,468 +0,0 @@
"""Tests for the unified web-console inventory API (#636).
Covers the four cases the issue names empty, populated, partial failure, and
the no-false-unowned invariant plus redaction, collision detection, the
resource-split routes, and read-only guarantees against a real control-plane
database and real durable lock files.
"""
import json
import os
import sys
import tempfile
import unittest
from datetime import datetime, timedelta, timezone
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from starlette.testclient import TestClient
import control_plane_db
from webui.app import create_app
from webui import inventory
def _iso(dt: datetime) -> str:
return dt.astimezone(timezone.utc).isoformat()
def _write_lock(lock_dir: str, name: str, payload: dict) -> str:
path = os.path.join(lock_dir, name)
with open(path, "w", encoding="utf-8") as handle:
json.dump(payload, handle)
return path
def _live_lock_payload(
*,
issue_number: int,
branch: str,
worktree_path: str,
pid: int,
username: str = "jcwalker3",
profile: str = "prgs-author",
) -> dict:
now = datetime.now(timezone.utc)
future = now + timedelta(hours=2)
return {
"branch_name": branch,
"issue_number": issue_number,
"org": "Scaled-Tech-Consulting",
"repo": "Gitea-Tools",
"remote": "prgs",
"pid": pid,
"session_pid": pid,
"lock_generation": 1,
"worktree_path": worktree_path,
"claimant": {"username": username, "profile": profile},
"work_lease": {
"branch": branch,
"issue_number": issue_number,
"operation_type": "author_issue_work",
"created_at": _iso(now),
"expires_at": _iso(future),
"last_heartbeat_at": _iso(now),
"claimant": {"username": username, "profile": profile},
},
}
class _FixtureMixin(unittest.TestCase):
def setUp(self) -> None:
self._tmp = tempfile.TemporaryDirectory()
self.tmp = self._tmp.name
self.lock_dir = os.path.join(self.tmp, "locks")
os.makedirs(self.lock_dir, mode=0o700)
self.db_path = os.path.join(self.tmp, "control_plane.db")
self.addCleanup(self._tmp.cleanup)
def _seed_db(self) -> control_plane_db.ControlPlaneDB:
db = control_plane_db.ControlPlaneDB(self.db_path)
db.upsert_session(
session_id="prgs-author-1",
role="author",
profile="prgs-author",
namespace="gitea-author",
pid=os.getpid(),
)
db.upsert_work_item(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
kind="issue",
number=636,
)
db.assign_and_lease(
session_id="prgs-author-1",
role="author",
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
kind="issue",
number=636,
)
return db
class TestRedaction(unittest.TestCase):
def test_redact_path_collapses_home(self):
home = os.path.expanduser("~")
self.assertEqual(
inventory.redact_path(f"{home}/Development/Gitea-Tools"),
"~/Development/Gitea-Tools",
)
def test_redact_url_strips_userinfo_and_query(self):
self.assertEqual(
inventory.redact_url("https://user:[email protected]/api?token=abc"),
"https://gitea.prgs.cc/api",
)
def test_scrub_drops_credential_keys(self):
scrubbed = inventory.scrub(
{"token": "abc123", "api_key": "k", "profile": "prgs-author"}
)
self.assertEqual(scrubbed["token"], "[redacted]")
self.assertEqual(scrubbed["api_key"], "[redacted]")
self.assertEqual(scrubbed["profile"], "prgs-author")
def test_scrub_is_recursive_and_never_raises(self):
class Weird:
def __repr__(self) -> str:
return "weird-obj"
out = inventory.scrub({"nested": [{"password": "p", "obj": Weird()}]})
self.assertEqual(out["nested"][0]["password"], "[redacted]")
self.assertEqual(out["nested"][0]["obj"], "weird-obj")
class TestEmptyInventory(_FixtureMixin):
def test_empty_db_and_locks_degrade_without_raising(self):
# No DB file, no locks: sessions/leases unavailable, locks ok+empty.
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
sessions = snap.section("sessions")
leases = snap.section("leases")
locks = snap.section("locks")
self.assertEqual(sessions.status, inventory.STATUS_UNAVAILABLE)
self.assertEqual(leases.status, inventory.STATUS_UNAVAILABLE)
self.assertEqual(locks.status, inventory.STATUS_OK)
self.assertEqual(len(locks.items), 0)
# Ownership authority is incomplete because the DB is missing.
self.assertFalse(snap.ownership_authority_complete)
self.assertEqual(snap.collisions, ())
def test_empty_db_present_but_unpopulated(self):
control_plane_db.ControlPlaneDB(self.db_path) # creates schema, no rows
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
self.assertEqual(snap.section("sessions").status, inventory.STATUS_OK)
self.assertEqual(len(snap.section("sessions").items), 0)
self.assertEqual(snap.section("leases").status, inventory.STATUS_OK)
self.assertTrue(snap.ownership_authority_complete)
class TestPopulatedInventory(_FixtureMixin):
def test_sections_populated_and_correlated(self):
self._seed_db()
wt = f"{self.tmp}/branches/issue-636-inventory-api"
_write_lock(
self.lock_dir,
"prgs-Scaled-Tech-Consulting-Gitea-Tools-636.json",
_live_lock_payload(
issue_number=636,
branch="feat/issue-636-inventory-api",
worktree_path=wt,
pid=os.getpid(),
),
)
hygiene = _StubHygiene(
entries=(
_StubEntry(
rel_path="branches/issue-636-inventory-api",
branch="feat/issue-636-inventory-api",
classification="active-issue",
),
)
)
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: hygiene,
)
self.assertTrue(snap.ownership_authority_complete)
self.assertEqual(len(snap.section("sessions").items), 1)
self.assertEqual(len(snap.section("leases").items), 1)
self.assertEqual(len(snap.section("locks").items), 1)
self.assertEqual(len(snap.section("worktrees").items), 1)
# The lease, lock, and worktree for #636 correlate onto one row.
row = next(r for r in snap.correlations if r["issue_number"] == 636)
self.assertEqual(row["branch"], "feat/issue-636-inventory-api")
self.assertTrue(row["lock_live"])
self.assertEqual(row["worktree_classification"], "active-issue")
self.assertEqual(len(row["lease_ids"]), 1)
# No collision: live lock, live pid, matching worktree.
self.assertEqual(snap.collisions, ())
def test_serialized_payload_declares_field_authority(self):
self._seed_db()
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
payload = inventory.snapshot_to_dict(snap)
self.assertEqual(payload["api_version"], "v1")
self.assertEqual(payload["schema_version"], 1)
self.assertEqual(payload["field_authority"]["sessions"], "control_plane_db")
self.assertEqual(payload["field_authority"]["locks"], "filesystem")
self.assertIn("sessions", payload["sections"])
class TestPartialFailure(_FixtureMixin):
def test_worktree_scan_failure_degrades_only_that_section(self):
self._seed_db()
def _boom():
raise RuntimeError("git worktree list exploded")
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=_boom,
)
self.assertEqual(
snap.section("worktrees").status, inventory.STATUS_UNAVAILABLE
)
self.assertIn("exploded", snap.section("worktrees").reason)
# DB-backed sections still healthy.
self.assertEqual(snap.section("sessions").status, inventory.STATUS_OK)
self.assertIn("worktrees", snap.degraded_sections)
def test_degraded_ownership_suppresses_unowned_claim(self):
# DB absent → sessions/leases unavailable → ownership incomplete even
# though a lock exists and could look "unclaimed" by the DB alone.
_write_lock(
self.lock_dir,
"prgs-Scaled-Tech-Consulting-Gitea-Tools-636.json",
_live_lock_payload(
issue_number=636,
branch="feat/issue-636-inventory-api",
worktree_path=f"{self.tmp}/wt",
pid=os.getpid(),
),
)
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
self.assertFalse(snap.ownership_authority_complete)
payload = inventory.snapshot_to_dict(snap)
self.assertIn("may be treated as unowned", payload["ownership_note"])
class TestCollisionDetection(_FixtureMixin):
def test_live_lock_dead_owner_flagged(self):
_write_lock(
self.lock_dir,
"prgs-Scaled-Tech-Consulting-Gitea-Tools-700.json",
_live_lock_payload(
issue_number=700,
branch="feat/issue-700-x",
worktree_path=f"{self.tmp}/wt700",
pid=999_999_999, # not a running pid
),
)
control_plane_db.ControlPlaneDB(self.db_path) # empty but present
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
kinds = {c.kind for c in snap.collisions}
self.assertIn("live-lock-dead-owner", kinds)
# Also lock-without-worktree, since no worktree carries the branch.
self.assertIn("lock-without-worktree", kinds)
def test_duplicate_live_lock_on_same_branch(self):
for issue in (800, 801):
_write_lock(
self.lock_dir,
f"prgs-Scaled-Tech-Consulting-Gitea-Tools-{issue}.json",
_live_lock_payload(
issue_number=issue,
branch="feat/issue-800-shared",
worktree_path=f"{self.tmp}/wt{issue}",
pid=os.getpid(),
),
)
control_plane_db.ControlPlaneDB(self.db_path)
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
self.assertIn(
"duplicate-live-lock", {c.kind for c in snap.collisions}
)
def test_no_collision_when_sections_degraded(self):
# locks ok but worktrees unavailable → lock-without-worktree must NOT
# be asserted (a missing scan is not a missing worktree).
_write_lock(
self.lock_dir,
"prgs-Scaled-Tech-Consulting-Gitea-Tools-636.json",
_live_lock_payload(
issue_number=636,
branch="feat/issue-636-inventory-api",
worktree_path=f"{self.tmp}/wt",
pid=os.getpid(),
),
)
control_plane_db.ControlPlaneDB(self.db_path)
def _boom():
raise RuntimeError("scan down")
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
load_hygiene=_boom,
)
self.assertNotIn(
"lock-without-worktree", {c.kind for c in snap.collisions}
)
class TestSectionInclude(_FixtureMixin):
def test_include_restricts_scanned_sections(self):
self._seed_db()
snap = inventory.load_inventory_snapshot(
db_path=self.db_path,
lock_dir=self.lock_dir,
include=("locks",),
)
self.assertIsNotNone(snap.section("locks"))
self.assertIsNone(snap.section("sessions"))
self.assertIsNone(snap.section("worktrees"))
class TestRoutes(_FixtureMixin):
def setUp(self) -> None:
super().setUp()
# Point the loaders at the fixture DB and lock dir via env, and stub
# the worktree scan so the route does not shell out to git.
self._prev_env = {
"GITEA_CONTROL_PLANE_DB": os.environ.get("GITEA_CONTROL_PLANE_DB"),
"GITEA_ISSUE_LOCK_DIR": os.environ.get("GITEA_ISSUE_LOCK_DIR"),
"WEBUI_TEST_OFFLINE": os.environ.get("WEBUI_TEST_OFFLINE"),
}
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir
os.environ["WEBUI_TEST_OFFLINE"] = "1"
self._seed_db()
self.client = TestClient(create_app())
def tearDown(self) -> None:
for key, value in self._prev_env.items():
if value is None:
os.environ.pop(key, None)
else:
os.environ[key] = value
def test_inventory_route_returns_versioned_payload(self):
resp = self.client.get("/api/v1/inventory")
self.assertEqual(resp.status_code, 200)
body = resp.json()
self.assertEqual(body["api_version"], "v1")
self.assertIn("sessions", body["sections"])
self.assertIn("field_authority", body)
def test_section_route_restricts_and_labels(self):
resp = self.client.get("/api/v1/inventory/locks")
self.assertEqual(resp.status_code, 200)
body = resp.json()
self.assertEqual(body["requested_section"], "locks")
self.assertIn("locks", body["sections"])
self.assertNotIn("sessions", body["sections"])
def test_unknown_section_is_404(self):
resp = self.client.get("/api/v1/inventory/bogus")
self.assertEqual(resp.status_code, 404)
self.assertEqual(resp.json()["error"], "unknown_section")
def test_inventory_route_rejects_post(self):
resp = self.client.post("/api/v1/inventory")
self.assertEqual(resp.status_code, 405)
class TestReadOnly(_FixtureMixin):
def test_snapshot_does_not_create_db_file(self):
missing = os.path.join(self.tmp, "does-not-exist.db")
inventory.load_inventory_snapshot(
db_path=missing,
lock_dir=self.lock_dir,
load_hygiene=lambda: _StubHygiene(entries=()),
)
self.assertFalse(os.path.exists(missing))
def test_readonly_connection_refuses_write(self):
self._seed_db()
conn = inventory._open_readonly(self.db_path)
try:
with self.assertRaises(Exception):
conn.execute(
"INSERT INTO sessions(session_id, role, started_at, "
"last_heartbeat_at, status) VALUES ('x','author',"
"'t','t','active')"
)
conn.commit()
finally:
conn.close()
# ── lightweight stand-ins for the #432 hygiene snapshot ──────────────────────
class _StubEntry:
def __init__(
self,
*,
rel_path: str,
branch: str | None = None,
classification: str = "stale-clean",
head_sha: str | None = "abc123",
dirty_tracked: int = 0,
dirty_untracked: bool = False,
detached: bool = False,
registered_worktree: bool = True,
notes: str = "",
) -> None:
self.rel_path = rel_path
self.folder_name = rel_path.split("/", 1)[-1]
self.branch = branch
self.classification = classification
self.head_sha = head_sha
self.dirty_tracked = dirty_tracked
self.dirty_untracked = dirty_untracked
self.detached = detached
self.registered_worktree = registered_worktree
self.notes = notes
class _StubHygiene:
def __init__(self, *, entries=(), scan_error=None) -> None:
self.entries = tuple(entries)
self.scan_error = scan_error
if __name__ == "__main__":
unittest.main()
+433
View File
@@ -0,0 +1,433 @@
"""Model usage, token cost, latency, and workflow-performance analytics (#651, Phase 4).
Ingests session instrumentation metrics, aggregates usage/cost/latency percentiles
by project, role, model, issue/PR, and stage, enforcing secret redaction and
explicitly rendering missing metrics as "Unknown" without zero-fabrication.
"""
from __future__ import annotations
import math
from dataclasses import asdict, dataclass
from typing import Any, Sequence
import control_plane_db
from webui import console_redaction
ANALYTICS_SCHEMA_VERSION = 1
@dataclass(frozen=True)
class UsageEvent:
usage_id: int
session_id: str | None
remote: str
org: str
repo: str
project_id: str | None
role: str
model: str
issue_number: int | None
pr_number: int | None
stage: str
input_tokens: int | None
output_tokens: int | None
total_tokens: int | None
estimated_cost_usd: float | None
latency_ms: int | None
duration_ms: int | None
status: str
metadata: str | None
created_at: str
def to_dict(self) -> dict[str, Any]:
d = asdict(self)
if d["metadata"]:
d["metadata"] = console_redaction.redact_text(d["metadata"])
return d
@dataclass(frozen=True)
class GroupMetrics:
name: str
total_events: int
events_with_tokens: int
input_tokens: int | None
output_tokens: int | None
total_tokens: int | None
events_with_cost: int
estimated_cost_usd: float | None
events_with_latency: int
latency_p50_ms: float | None
latency_p90_ms: float | None
latency_p95_ms: float | None
latency_p99_ms: float | None
latency_avg_ms: float | None
events_with_duration: int
duration_avg_ms: float | None
display_tokens: str
display_cost: str
display_latency_p50: str
display_latency_p90: str
display_duration_avg: str
def to_dict(self) -> dict[str, Any]:
return asdict(self)
@dataclass(frozen=True)
class AnalyticsSnapshot:
ok: bool
reason: str
schema_version: int
remote: str
org: str
repo: str
total_events: int
overall_summary: GroupMetrics
by_project: dict[str, GroupMetrics]
by_role: dict[str, GroupMetrics]
by_model: dict[str, GroupMetrics]
by_work_item: dict[str, GroupMetrics]
by_stage: dict[str, GroupMetrics]
events: tuple[UsageEvent, ...]
def to_dict(self) -> dict[str, Any]:
return {
"ok": self.ok,
"reason": self.reason,
"schema_version": self.schema_version,
"remote": self.remote,
"org": self.org,
"repo": self.repo,
"total_events": self.total_events,
"overall_summary": self.overall_summary.to_dict(),
"by_project": {k: v.to_dict() for k, v in self.by_project.items()},
"by_role": {k: v.to_dict() for k, v in self.by_role.items()},
"by_model": {k: v.to_dict() for k, v in self.by_model.items()},
"by_work_item": {k: v.to_dict() for k, v in self.by_work_item.items()},
"by_stage": {k: v.to_dict() for k, v in self.by_stage.items()},
"events": [e.to_dict() for e in self.events],
}
def compute_percentile(values: Sequence[float | int], percentile: float) -> float | None:
if not values:
return None
sorted_vals = sorted(values)
n = len(sorted_vals)
if n == 1:
return float(sorted_vals[0])
k = (n - 1) * (percentile / 100.0)
f = math.floor(k)
c = math.ceil(k)
if f == c:
return float(sorted_vals[int(f)])
d0 = sorted_vals[int(f)] * (c - k)
d1 = sorted_vals[int(c)] * (k - f)
return float(d0 + d1)
def aggregate_events(group_name: str, events: Sequence[UsageEvent]) -> GroupMetrics:
total_events = len(events)
if total_events == 0:
return GroupMetrics(
name=group_name,
total_events=0,
events_with_tokens=0,
input_tokens=None,
output_tokens=None,
total_tokens=None,
events_with_cost=0,
estimated_cost_usd=None,
events_with_latency=0,
latency_p50_ms=None,
latency_p90_ms=None,
latency_p95_ms=None,
latency_p99_ms=None,
latency_avg_ms=None,
events_with_duration=0,
duration_avg_ms=None,
display_tokens="Unknown",
display_cost="Unknown",
display_latency_p50="Unknown",
display_latency_p90="Unknown",
display_duration_avg="Unknown",
)
token_events = [
e for e in events
if e.total_tokens is not None or e.input_tokens is not None or e.output_tokens is not None
]
events_with_tokens = len(token_events)
if events_with_tokens > 0:
input_tokens = sum(e.input_tokens or 0 for e in token_events)
output_tokens = sum(e.output_tokens or 0 for e in token_events)
total_tokens = sum(
e.total_tokens if e.total_tokens is not None else ((e.input_tokens or 0) + (e.output_tokens or 0))
for e in token_events
)
display_tokens = f"{total_tokens:,}"
else:
input_tokens = None
output_tokens = None
total_tokens = None
display_tokens = "Unknown"
cost_events = [e for e in events if e.estimated_cost_usd is not None]
events_with_cost = len(cost_events)
if events_with_cost > 0:
estimated_cost_usd = round(sum(e.estimated_cost_usd for e in cost_events), 6)
display_cost = f"${estimated_cost_usd:.4f}"
else:
estimated_cost_usd = None
display_cost = "Unknown"
latency_vals = [e.latency_ms for e in events if e.latency_ms is not None]
events_with_latency = len(latency_vals)
if events_with_latency > 0:
latency_p50_ms = compute_percentile(latency_vals, 50.0)
latency_p90_ms = compute_percentile(latency_vals, 90.0)
latency_p95_ms = compute_percentile(latency_vals, 95.0)
latency_p99_ms = compute_percentile(latency_vals, 99.0)
latency_avg_ms = round(sum(latency_vals) / events_with_latency, 2)
display_latency_p50 = f"{round(latency_p50_ms, 1)} ms" if latency_p50_ms is not None else "Unknown"
display_latency_p90 = f"{round(latency_p90_ms, 1)} ms" if latency_p90_ms is not None else "Unknown"
else:
latency_p50_ms = None
latency_p90_ms = None
latency_p95_ms = None
latency_p99_ms = None
latency_avg_ms = None
display_latency_p50 = "Unknown"
display_latency_p90 = "Unknown"
duration_vals = [e.duration_ms for e in events if e.duration_ms is not None]
events_with_duration = len(duration_vals)
if events_with_duration > 0:
duration_avg_ms = round(sum(duration_vals) / events_with_duration, 2)
display_duration_avg = f"{round(duration_avg_ms / 1000.0, 2)} s" if duration_avg_ms >= 1000 else f"{round(duration_avg_ms, 1)} ms"
else:
duration_avg_ms = None
display_duration_avg = "Unknown"
return GroupMetrics(
name=group_name,
total_events=total_events,
events_with_tokens=events_with_tokens,
input_tokens=input_tokens,
output_tokens=output_tokens,
total_tokens=total_tokens,
events_with_cost=events_with_cost,
estimated_cost_usd=estimated_cost_usd,
events_with_latency=events_with_latency,
latency_p50_ms=latency_p50_ms,
latency_p90_ms=latency_p90_ms,
latency_p95_ms=latency_p95_ms,
latency_p99_ms=latency_p99_ms,
latency_avg_ms=latency_avg_ms,
events_with_duration=events_with_duration,
duration_avg_ms=duration_avg_ms,
display_tokens=display_tokens,
display_cost=display_cost,
display_latency_p50=display_latency_p50,
display_latency_p90=display_latency_p90,
display_duration_avg=display_duration_avg,
)
def record_usage(
*,
db_path: str | None = None,
session_id: str | None = None,
remote: str = "dadeschools",
org: str = "",
repo: str = "",
project_id: str | None = None,
role: str = "unknown",
model: str = "unknown",
issue_number: int | None = None,
pr_number: int | None = None,
stage: str = "unknown",
input_tokens: int | None = None,
output_tokens: int | None = None,
total_tokens: int | None = None,
estimated_cost_usd: float | None = None,
latency_ms: int | None = None,
duration_ms: int | None = None,
status: str = "success",
metadata: str | dict[str, Any] | None = None,
created_at: str | None = None,
) -> int:
"""Ingest/record a single usage event with optional metrics."""
db = control_plane_db.ControlPlaneDB(db_path=db_path)
return db.record_usage_event(
session_id=session_id,
remote=remote,
org=org,
repo=repo,
project_id=project_id,
role=role,
model=model,
issue_number=issue_number,
pr_number=pr_number,
stage=stage,
input_tokens=input_tokens,
output_tokens=output_tokens,
total_tokens=total_tokens,
estimated_cost_usd=estimated_cost_usd,
latency_ms=latency_ms,
duration_ms=duration_ms,
status=status,
metadata=metadata,
created_at=created_at,
)
def load_analytics(
*,
db_path: str | None = None,
remote: str | None = None,
org: str | None = None,
repo: str | None = None,
project_id: str | None = None,
role: str | None = None,
model: str | None = None,
stage: str | None = None,
issue_number: int | None = None,
pr_number: int | None = None,
limit: int = 500,
) -> AnalyticsSnapshot:
"""Load analytics snapshot aggregated by project, role, model, issue/PR, and stage."""
remote_filter = (remote or "").strip() or None
org_filter = (org or "").strip() or None
repo_filter = (repo or "").strip() or None
role_filter = (role or "").strip() or None
model_filter = (model or "").strip() or None
stage_filter = (stage or "").strip() or None
# F4: coerce optional scope filters to str so AnalyticsSnapshot never holds None.
scope_remote = (remote or "").strip()
scope_org = (org or "").strip()
scope_repo = (repo or "").strip()
try:
db = control_plane_db.ControlPlaneDB(db_path=db_path)
rows = db.query_usage_events(
remote=remote_filter,
org=org_filter,
repo=repo_filter,
project_id=project_id,
role=role_filter,
model=model_filter,
stage=stage_filter,
issue_number=issue_number,
pr_number=pr_number,
limit=limit,
)
except Exception as exc:
empty_summary = aggregate_events("Overall", [])
return AnalyticsSnapshot(
ok=False,
reason=f"control_plane_db_unavailable: {exc}",
schema_version=ANALYTICS_SCHEMA_VERSION,
remote=scope_remote,
org=scope_org,
repo=scope_repo,
total_events=0,
overall_summary=empty_summary,
by_project={},
by_role={},
by_model={},
by_work_item={},
by_stage={},
events=(),
)
parsed_events: list[UsageEvent] = []
for r in rows:
meta = console_redaction.redact_text(r.get("metadata")) if r.get("metadata") else None
parsed_events.append(
UsageEvent(
usage_id=r["usage_id"],
session_id=r.get("session_id"),
remote=r.get("remote") or scope_remote,
org=r.get("org") or scope_org,
repo=r.get("repo") or scope_repo,
project_id=r.get("project_id"),
role=r.get("role") or "unknown",
model=r.get("model") or "unknown",
issue_number=r.get("issue_number"),
pr_number=r.get("pr_number"),
stage=r.get("stage") or "unknown",
input_tokens=r.get("input_tokens"),
output_tokens=r.get("output_tokens"),
total_tokens=r.get("total_tokens"),
estimated_cost_usd=r.get("estimated_cost_usd"),
latency_ms=r.get("latency_ms"),
duration_ms=r.get("duration_ms"),
status=r.get("status") or "success",
metadata=meta,
created_at=r.get("created_at") or "",
)
)
overall_summary = aggregate_events("Overall", parsed_events)
# Group by project
groups_by_project: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
key = e.project_id or (f"{e.org}/{e.repo}" if e.org and e.repo else "default")
groups_by_project.setdefault(key, []).append(e)
by_project = {k: aggregate_events(k, v) for k, v in groups_by_project.items()}
# Group by role
groups_by_role: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
groups_by_role.setdefault(e.role, []).append(e)
by_role = {k: aggregate_events(k, v) for k, v in groups_by_role.items()}
# Group by model
groups_by_model: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
groups_by_model.setdefault(e.model, []).append(e)
by_model = {k: aggregate_events(k, v) for k, v in groups_by_model.items()}
# Group by work item
groups_by_work_item: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
if e.issue_number:
key = f"issue #{e.issue_number}"
elif e.pr_number:
key = f"pr #{e.pr_number}"
else:
key = "unlinked"
groups_by_work_item.setdefault(key, []).append(e)
by_work_item = {k: aggregate_events(k, v) for k, v in groups_by_work_item.items()}
# Group by stage
groups_by_stage: dict[str, list[UsageEvent]] = {}
for e in parsed_events:
groups_by_stage.setdefault(e.stage, []).append(e)
by_stage = {k: aggregate_events(k, v) for k, v in groups_by_stage.items()}
return AnalyticsSnapshot(
ok=True,
reason="ok",
schema_version=ANALYTICS_SCHEMA_VERSION,
remote=scope_remote,
org=scope_org,
repo=scope_repo,
total_events=len(parsed_events),
overall_summary=overall_summary,
by_project=by_project,
by_role=by_role,
by_model=by_model,
by_work_item=by_work_item,
by_stage=by_stage,
events=tuple(parsed_events),
)
def snapshot_to_dict(snapshot: AnalyticsSnapshot) -> dict[str, Any]:
return snapshot.to_dict()
+248
View File
@@ -0,0 +1,248 @@
"""HTML views for the Model Usage & Performance Analytics console (#651)."""
from __future__ import annotations
import html
from webui.analytics_loader import AnalyticsSnapshot, GroupMetrics, UsageEvent
from webui.layout import render_page
def _escape(text: object) -> str:
"""HTML-escape dynamic analytics fields (mirrors audit_views / project_views)."""
return html.escape(str(text), quote=True)
def _render_badge(text: str, badge_type: str = "muted") -> str:
return f'<span class="badge badge-{_escape(badge_type)}">{_escape(text)}</span>'
def _render_group_table(title: str, groups: dict[str, GroupMetrics], key_header: str = "Group") -> str:
if not groups:
return (
f"<h3>{_escape(title)}</h3>"
'<div class="card"><p class="muted">No telemetry events recorded for this dimension.</p></div>'
)
rows = []
for key, g in sorted(groups.items(), key=lambda x: x[1].total_events, reverse=True):
cost_cell = (
f'<span class="accent">{_escape(g.display_cost)}</span>'
if g.events_with_cost > 0
else _render_badge("Unknown")
)
tokens_cell = (
_escape(g.display_tokens)
if g.events_with_tokens > 0
else _render_badge("Unknown")
)
lat_p50 = (
_escape(g.display_latency_p50)
if g.events_with_latency > 0
else _render_badge("Unknown")
)
lat_p90 = (
_escape(g.display_latency_p90)
if g.events_with_latency > 0
else _render_badge("Unknown")
)
dur_avg = (
_escape(g.display_duration_avg)
if g.events_with_duration > 0
else _render_badge("Unknown")
)
rows.append(
"<tr>"
f"<td><strong>{_escape(key)}</strong></td>"
f"<td>{g.total_events}</td>"
f"<td>{tokens_cell}</td>"
f"<td>{cost_cell}</td>"
f"<td>{lat_p50}</td>"
f"<td>{lat_p90}</td>"
f"<td>{dur_avg}</td>"
"</tr>"
)
rows_html = "".join(rows)
return f"""
<h3>{_escape(title)}</h3>
<div class="card" style="overflow-x: auto;">
<table class="data-table">
<thead>
<tr>
<th>{_escape(key_header)}</th>
<th>Events</th>
<th>Total Tokens</th>
<th>Est. Cost</th>
<th>Latency (p50)</th>
<th>Latency (p90)</th>
<th>Avg Stage Duration</th>
</tr>
</thead>
<tbody>
{rows_html}
</tbody>
</table>
</div>
"""
def _render_events_table(events: tuple[UsageEvent, ...]) -> str:
if not events:
return (
"<h3>Recent Usage & Instrumentation Events</h3>"
'<div class="card"><p class="muted">No individual telemetry events recorded yet. Opt-in instrumentation via session logging or authorized POST /api/v1/analytics/usage.</p></div>'
)
rows = []
for e in list(events)[-50:]: # Display latest 50
if e.issue_number is not None:
work_item = f"issue #{e.issue_number}"
elif e.pr_number is not None:
work_item = f"pr #{e.pr_number}"
else:
work_item = "unlinked"
tokens = (
_escape(f"{e.total_tokens:,}")
if e.total_tokens is not None
else _render_badge("Unknown")
)
cost = (
_escape(f"${e.estimated_cost_usd:.4f}")
if e.estimated_cost_usd is not None
else _render_badge("Unknown")
)
latency = (
_escape(f"{e.latency_ms} ms")
if e.latency_ms is not None
else _render_badge("Unknown")
)
duration = (
_escape(f"{e.duration_ms} ms")
if e.duration_ms is not None
else _render_badge("Unknown")
)
status_badge = _render_badge(
e.status, "success" if e.status == "success" else "danger"
)
rows.append(
"<tr>"
f"<td>#{e.usage_id}</td>"
f"<td><small>{_escape(e.created_at)}</small></td>"
f"<td><span class=\"badge\">{_escape(e.role)}</span></td>"
f"<td><strong>{_escape(e.model)}</strong></td>"
f"<td>{_escape(e.stage)}</td>"
f"<td>{_escape(work_item)}</td>"
f"<td>{tokens}</td>"
f"<td>{cost}</td>"
f"<td>{latency}</td>"
f"<td>{duration}</td>"
f"<td>{status_badge}</td>"
"</tr>"
)
rows_html = "".join(rows)
return f"""
<h3>Recent Telemetry Events</h3>
<div class="card" style="overflow-x: auto;">
<table class="data-table">
<thead>
<tr>
<th>ID</th>
<th>Timestamp</th>
<th>Role</th>
<th>Model</th>
<th>Stage</th>
<th>Work Item</th>
<th>Tokens</th>
<th>Cost</th>
<th>Latency</th>
<th>Duration</th>
<th>Status</th>
</tr>
</thead>
<tbody>
{rows_html}
</tbody>
</table>
</div>
"""
def render_analytics_page(snapshot: AnalyticsSnapshot) -> str:
"""Render the main Model Usage & Performance Analytics console page."""
summary = snapshot.overall_summary
kpi_tokens = (
_escape(summary.display_tokens)
if summary.events_with_tokens > 0
else _render_badge("Unknown")
)
kpi_cost = (
_escape(summary.display_cost)
if summary.events_with_cost > 0
else _render_badge("Unknown")
)
kpi_lat_p50 = (
_escape(summary.display_latency_p50)
if summary.events_with_latency > 0
else _render_badge("Unknown")
)
kpi_dur_avg = (
_escape(summary.display_duration_avg)
if summary.events_with_duration > 0
else _render_badge("Unknown")
)
status_notice = ""
if not snapshot.ok:
status_notice = (
f'<div class="card warning-card"><strong>Degraded Data Source:</strong> '
f'{_escape(snapshot.reason)}</div>'
)
body_html = f"""
<h2>Model Usage & Performance Analytics (Phase 4)</h2>
<p class="muted">
Durable console analytics for model usage, token cost, latency percentiles, and workflow-stage performance correlated to issues, PRs, and worker roles.
</p>
{status_notice}
<div class="notice-card" style="background: rgba(91, 159, 212, 0.1); border: 1px solid var(--border); padding: 0.75rem 1rem; border-radius: 6px; margin-bottom: 1.5rem;">
<small><strong>Note on telemetry fidelity:</strong> Missing data or untracked metrics are explicitly labeled as <em>Unknown</em>. No token costs or latency metrics are zero-fabricated.</small>
</div>
<div class="card-grid" style="display: grid; grid-template-columns: repeat(auto-fit, minmax(180px, 1fr)); gap: 1rem; margin-bottom: 1.5rem;">
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Total Events</span>
<h3 style="margin: 0.25rem 0 0 0;">{summary.total_events}</h3>
</div>
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Total Tokens</span>
<h3 style="margin: 0.25rem 0 0 0;">{kpi_tokens}</h3>
</div>
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Est. Token Cost</span>
<h3 style="margin: 0.25rem 0 0 0;">{kpi_cost}</h3>
</div>
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Latency (p50)</span>
<h3 style="margin: 0.25rem 0 0 0;">{kpi_lat_p50}</h3>
</div>
<div class="card">
<span class="muted" style="font-size: 0.85rem;">Avg Stage Duration</span>
<h3 style="margin: 0.25rem 0 0 0;">{kpi_dur_avg}</h3>
</div>
</div>
{_render_group_table("Usage & Cost by Model", snapshot.by_model, "Model")}
{_render_group_table("Performance by Workflow Stage", snapshot.by_stage, "Stage")}
{_render_group_table("Usage & Cost by Role", snapshot.by_role, "Role")}
{_render_group_table("Work Item Analytics", snapshot.by_work_item, "Work Item")}
{_render_events_table(snapshot.events)}
"""
return render_page(title="Model Usage & Performance Analytics", body_html=body_html)
+118 -35
View File
@@ -46,12 +46,13 @@ from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as wo
from webui.worktree_views import render_worktrees_page
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
from webui.runtime_views import render_runtime_page
from webui.inventory import (
SECTION_NAMES as _INVENTORY_SECTIONS,
load_inventory_snapshot,
snapshot_to_dict as inventory_snapshot_to_dict,
)
from webui.timeline import load_timeline, snapshot_to_dict as timeline_snapshot_to_dict
from webui.analytics_loader import (
load_analytics,
record_usage,
snapshot_to_dict as analytics_snapshot_to_dict,
)
from webui.analytics_views import render_analytics_page
from webui.system_health import (
API_PATH as SYSTEM_HEALTH_API_PATH,
load_system_health,
@@ -465,30 +466,6 @@ async def api_action_attempt(request: Request) -> JSONResponse:
return JSONResponse(result, status_code=status)
async def api_inventory(_request: Request) -> JSONResponse:
"""Unified read-only session/lease/lock/worktree inventory (#636)."""
snapshot = load_inventory_snapshot()
return JSONResponse(inventory_snapshot_to_dict(snapshot))
async def api_inventory_section(request: Request) -> JSONResponse:
"""Resource-split view: one inventory section under the shared schema."""
section = request.path_params["section"]
if section not in _INVENTORY_SECTIONS:
return JSONResponse(
{
"error": "unknown_section",
"detail": f"no inventory section named {section!r}",
"available": sorted(_INVENTORY_SECTIONS),
},
status_code=404,
)
snapshot = load_inventory_snapshot(include=(section,))
payload = inventory_snapshot_to_dict(snapshot)
payload["requested_section"] = section
return JSONResponse(payload)
async def api_console_security_model(_request: Request) -> JSONResponse:
"""Read-only publication of the #633 authorization/redaction/audit model."""
return JSONResponse({
@@ -596,6 +573,114 @@ async def api_v1_timeline(request: Request) -> JSONResponse:
return JSONResponse(timeline_snapshot_to_dict(snapshot), status_code=status_code)
async def analytics(request: Request) -> HTMLResponse:
"""Read-only model usage, token cost, latency, and performance analytics HTML view (#651)."""
snapshot = load_analytics(
remote=request.query_params.get("remote"),
org=request.query_params.get("org"),
repo=request.query_params.get("repo"),
role=request.query_params.get("role"),
model=request.query_params.get("model"),
stage=request.query_params.get("stage"),
issue_number=_query_int(request, "issue"),
pr_number=_query_int(request, "pr"),
limit=_query_int(request, "limit") or 200,
)
return HTMLResponse(render_analytics_page(snapshot))
async def api_v1_analytics(request: Request) -> JSONResponse:
"""Read-only model usage, token cost, latency, and performance analytics API (#651)."""
snapshot = load_analytics(
remote=request.query_params.get("remote"),
org=request.query_params.get("org"),
repo=request.query_params.get("repo"),
role=request.query_params.get("role"),
model=request.query_params.get("model"),
stage=request.query_params.get("stage"),
issue_number=_query_int(request, "issue"),
pr_number=_query_int(request, "pr"),
limit=_query_int(request, "limit") or 500,
)
status_code = 200 if snapshot.ok else 500
return JSONResponse(analytics_snapshot_to_dict(snapshot), status_code=status_code)
async def api_v1_analytics_ingest(request: Request) -> JSONResponse:
"""Optional session instrumentation ingestion endpoint (#651).
Fail-closed write: every request is authorized through console_authz
(``record_analytics_usage``) before any control-plane DB mutation. Phase 1
keeps ``execution_enabled=False`` and denies unauthenticated callers, so
this route cannot be used as an unauthenticated write or XSS injection
vector (PR #876 F2).
"""
try:
body = await request.json()
except Exception:
body = {}
if not isinstance(body, dict):
body = {}
principal = resolve_principal(headers=dict(request.headers))
decision = authorize(
"record_analytics_usage", principal, for_execution=True
)
allowed = bool(decision.allowed and decision.execution_enabled)
console_audit.record_event(
action_id="record_analytics_usage",
result=(
console_audit.RESULT_ALLOWED
if allowed
else console_audit.RESULT_DENIED
),
decision=decision,
principal=principal,
target=_audit_target("record_analytics_usage", body),
request_id=_request_id(),
detail=decision.detail,
)
authorization = decision.to_dict()
if not allowed:
return JSONResponse(
{
"ok": False,
"error": "unauthorized",
"detail": (
"POST /api/v1/analytics/usage requires an authenticated "
"principal with record_analytics_usage execution enabled"
),
"authorization": authorization,
},
status_code=403,
)
usage_id = record_usage(
session_id=body.get("session_id"),
remote=body.get("remote", "dadeschools"),
org=body.get("org", ""),
repo=body.get("repo", ""),
project_id=body.get("project_id"),
role=body.get("role", "unknown"),
model=body.get("model", "unknown"),
issue_number=body.get("issue_number") or body.get("issue"),
pr_number=body.get("pr_number") or body.get("pr"),
stage=body.get("stage", "unknown"),
input_tokens=body.get("input_tokens"),
output_tokens=body.get("output_tokens"),
total_tokens=body.get("total_tokens"),
estimated_cost_usd=body.get("estimated_cost_usd"),
latency_ms=body.get("latency_ms"),
duration_ms=body.get("duration_ms"),
status=body.get("status", "success"),
metadata=body.get("metadata"),
)
return JSONResponse(
{"ok": True, "usage_id": usage_id, "authorization": authorization},
status_code=201,
)
async def method_not_allowed(request: Request, _exc: Exception) -> Response:
path = request.url.path
if path in _AUDIT_MUTATION_PATHS and request.method == "POST":
@@ -637,6 +722,10 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
Route("/runtime", runtime, methods=["GET"]),
Route("/api/runtime", api_runtime, methods=["GET"]),
Route("/api/v1/timeline", api_v1_timeline, methods=["GET"]),
Route("/analytics", analytics, methods=["GET"]),
Route("/api/analytics", api_v1_analytics, methods=["GET"]),
Route("/api/v1/analytics", api_v1_analytics, methods=["GET"]),
Route("/api/v1/analytics/usage", api_v1_analytics_ingest, methods=["POST"]),
Route("/audit", audit, methods=["GET", "POST"]),
Route("/api/audit", api_audit, methods=["GET", "POST"]),
Route("/worktrees", worktrees, methods=["GET"]),
@@ -655,12 +744,6 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
methods=["POST"],
),
Route("/api/leases", api_leases, methods=["GET"]),
Route("/api/v1/inventory", api_inventory, methods=["GET"]),
Route(
"/api/v1/inventory/{section}",
api_inventory_section,
methods=["GET"],
),
Route(
"/api/console/security-model",
api_console_security_model,
+13
View File
@@ -236,6 +236,19 @@ _ACTION_SPECS: tuple[ConsoleAction, ...] = (
phase=3,
summary="Remove a remote feature branch.",
),
# #651 analytics ingest: local control-plane write, not a Gitea mutation.
# Phase 2 gated write so Phase 1 (ACTIVE_PHASE=1) fails closed on execution.
ConsoleAction(
action_id="record_analytics_usage",
task_key="record_analytics_usage",
action_class=CLASS_WRITE,
minimum_role=OPERATOR,
requires_confirmation=True,
dual_control=False,
break_glass=False,
phase=2,
summary="Ingest a model-usage / latency analytics event into the control-plane DB.",
),
# #642: sanctioned daemon lifecycle. These exist so operators have an
# audited path off `pkill -f mcp_server.py` (#630). Restart drops every
# in-flight request on a namespace, so it carries the same dual-control and
-952
View File
@@ -1,952 +0,0 @@
"""Unified session/lease/lock/worktree inventory for the web console (#636).
Leases (#433), worktrees (#432), and runtime (#430) each ship their own MVP
view, each with its own shape and its own idea of what "owned" means. A
traffic-control or recovery operator has to read all three and correlate them
by hand, which is exactly the step that goes wrong under collision pressure.
This module aggregates them into one versioned, read-only snapshot so the
console, and any worker asking "what is safe to do next", read the same
inventory from the same authority.
Field authority is explicit and never blended. Every section declares where its
rows came from:
* ``control_plane_db`` the #613 substrate: sessions, leases, assignments.
Authoritative for *exclusive ownership* (#600/#601).
* ``filesystem`` durable per-issue lock files (:mod:`issue_lock_store`) and
registered git worktrees. Authoritative for *what exists on this machine*.
* ``gitea`` remote issue/PR state, reached only through existing loaders.
Safety invariants:
* **Read-only.** The control-plane database is opened through a ``mode=ro``
URI. :class:`control_plane_db.ControlPlaneDB` creates directories and runs
migrations in its constructor, which an inventory read must never do, so this
module talks to sqlite directly rather than through that class.
* **Fail-soft, never fail-silent.** A subsystem that cannot be read degrades to
a section carrying ``status`` and ``reason``. It never raises, and it never
produces an empty list that reads like "nothing is there".
* **Never invent active ownership.** This is the invariant that matters most.
A degraded ownership source sets ``ownership_authority_complete`` false, and
while that flag is false no work item is reported unowned and no collision is
asserted. Absence of evidence is reported as absence of evidence.
* **Redaction at the boundary.** Absolute paths are collapsed against the home
directory, URLs lose userinfo and query strings, and no credential-shaped
value is emitted. No session token exists in these sources and none is read.
Phase 1 is read-only. Lease steal/release and worktree deletion are Phase 2+
and deliberately have no representation here, not even a disabled one.
"""
from __future__ import annotations
import os
import re
import sqlite3
import time
from dataclasses import dataclass, field
from datetime import datetime, timezone
from typing import Any, Callable
from urllib.parse import urlparse
import control_plane_db
import issue_lock_store
SCHEMA_VERSION = 1
API_VERSION = "v1"
#: Sections whose absence would make an ownership claim unprovable. If any of
#: these is not ``ok``, the snapshot refuses to describe anything as unowned.
OWNERSHIP_SECTIONS = ("sessions", "leases", "locks")
SECTION_NAMES = ("sessions", "leases", "locks", "worktrees", "namespaces")
STATUS_OK = "ok"
STATUS_DEGRADED = "degraded"
STATUS_UNAVAILABLE = "unavailable"
AUTHORITY_CONTROL_PLANE_DB = "control_plane_db"
AUTHORITY_FILESYSTEM = "filesystem"
AUTHORITY_GITEA = "gitea"
_CREDENTIAL_KEY_RE = re.compile(
r"(token|secret|password|passwd|api[_-]?key|authorization|bearer|credential)",
re.IGNORECASE,
)
_REDACTED = "[redacted]"
@dataclass(frozen=True)
class InventorySection:
"""One subsystem's contribution, with its authority and health."""
name: str
authority: str
status: str
items: tuple[dict[str, Any], ...] = ()
reason: str | None = None
scan_ms: float | None = None
@property
def ok(self) -> bool:
return self.status == STATUS_OK
def to_dict(self) -> dict[str, Any]:
return {
"name": self.name,
"authority": self.authority,
"status": self.status,
"count": len(self.items),
"reason": self.reason,
"scan_ms": self.scan_ms,
"items": [dict(item) for item in self.items],
}
@dataclass(frozen=True)
class CollisionSignal:
"""A detected conflict between two ownership records."""
kind: str
message: str
severity: str = "warning"
issue_number: int | None = None
branch: str | None = None
worktree_path: str | None = None
session_ids: tuple[str, ...] = ()
def to_dict(self) -> dict[str, Any]:
return {
"kind": self.kind,
"severity": self.severity,
"message": self.message,
"issue_number": self.issue_number,
"branch": self.branch,
"worktree_path": self.worktree_path,
"session_ids": list(self.session_ids),
}
@dataclass(frozen=True)
class InventorySnapshot:
"""Versioned aggregate of every inventory section."""
generated_at: str
sections: tuple[InventorySection, ...]
collisions: tuple[CollisionSignal, ...] = ()
correlations: tuple[dict[str, Any], ...] = ()
schema_version: int = SCHEMA_VERSION
api_version: str = API_VERSION
scan_ms: float | None = None
_section_index: dict[str, InventorySection] = field(
default_factory=dict, repr=False, compare=False
)
def section(self, name: str) -> InventorySection | None:
return self._section_index.get(name)
@property
def degraded_sections(self) -> tuple[str, ...]:
return tuple(s.name for s in self.sections if not s.ok)
@property
def ownership_authority_complete(self) -> bool:
"""True only when every ownership-bearing section read cleanly.
While this is false the snapshot must not describe any work item as
unowned: a lease the reader could not load is not an absent lease.
"""
for name in OWNERSHIP_SECTIONS:
section = self._section_index.get(name)
if section is None or not section.ok:
return False
return True
@property
def status(self) -> str:
if all(s.ok for s in self.sections):
return STATUS_OK
return STATUS_DEGRADED
# ── redaction ────────────────────────────────────────────────────────────────
def redact_path(path: str | None) -> str | None:
"""Collapse an absolute path against ``$HOME`` for browser display."""
if not path:
return path
text = str(path)
home = os.path.expanduser("~")
if home and home != "/" and text.startswith(home):
return "~" + text[len(home) :]
return text
def redact_url(value: str | None) -> str | None:
"""Strip userinfo and query string from a URL."""
if not value:
return value
text = str(value)
try:
parsed = urlparse(text)
except ValueError:
return _REDACTED
if not parsed.scheme or not parsed.netloc:
return text
netloc = parsed.hostname or ""
if parsed.port:
netloc = f"{netloc}:{parsed.port}"
rebuilt = f"{parsed.scheme}://{netloc}{parsed.path}"
return rebuilt.rstrip("/") or rebuilt
def scrub(value: Any, *, key: str | None = None) -> Any:
"""Recursively drop credential-shaped values and redact paths/URLs.
Never raises: an unexpected object degrades to its ``repr`` rather than
propagating out of a read-only view.
"""
if key and _CREDENTIAL_KEY_RE.search(key):
return _REDACTED
if isinstance(value, dict):
return {str(k): scrub(v, key=str(k)) for k, v in value.items()}
if isinstance(value, (list, tuple)):
return [scrub(v, key=key) for v in value]
if isinstance(value, str):
if value.startswith(("http://", "https://")):
return redact_url(value)
if value.startswith("/") or value.startswith("~"):
return redact_path(value)
return value
if isinstance(value, (int, float, bool)) or value is None:
return value
return repr(value)
# ── control-plane database (read-only) ───────────────────────────────────────
def _open_readonly(db_path: str) -> sqlite3.Connection:
"""Open the control-plane DB without creating or migrating anything."""
conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=5)
conn.row_factory = sqlite3.Row
return conn
def _table_names(conn: sqlite3.Connection) -> set[str]:
rows = conn.execute(
"SELECT name FROM sqlite_master WHERE type = 'table'"
).fetchall()
return {str(row[0]) for row in rows}
def _load_cp_db_sections(
*,
db_path: str | None = None,
limit: int = 200,
) -> tuple[InventorySection, InventorySection]:
"""Return the ``sessions`` and ``leases`` sections from the #613 DB."""
path = (db_path or control_plane_db.default_db_path()).strip()
def _both_unavailable(reason: str) -> tuple[InventorySection, InventorySection]:
return (
InventorySection(
name="sessions",
authority=AUTHORITY_CONTROL_PLANE_DB,
status=STATUS_UNAVAILABLE,
reason=reason,
),
InventorySection(
name="leases",
authority=AUTHORITY_CONTROL_PLANE_DB,
status=STATUS_UNAVAILABLE,
reason=reason,
),
)
if not path:
return _both_unavailable("control-plane database path is not configured")
if not os.path.exists(path):
return _both_unavailable(
f"control-plane database not present at {redact_path(path)}; "
"no session or lease authority available"
)
started = time.perf_counter()
try:
conn = _open_readonly(path)
except sqlite3.Error as exc:
return _both_unavailable(f"control-plane database could not be opened: {exc}")
try:
tables = _table_names(conn)
if "sessions" not in tables or "leases" not in tables:
missing = sorted({"sessions", "leases"} - tables)
return _both_unavailable(
"control-plane database is missing required tables: "
+ ", ".join(missing)
)
session_rows = [
dict(row)
for row in conn.execute(
"SELECT session_id, role, profile, namespace, pid, started_at,"
" last_heartbeat_at, status FROM sessions"
" ORDER BY last_heartbeat_at DESC LIMIT ?",
(max(1, int(limit)),),
).fetchall()
]
has_work_items = "work_items" in tables
if has_work_items:
lease_sql = (
"SELECT l.lease_id, l.session_id, l.role, l.phase, l.status,"
" l.expires_at, w.remote, w.org, w.repo, w.kind AS work_kind,"
" w.number AS work_number, w.state AS work_state,"
" s.pid AS session_pid, s.profile AS session_profile,"
" s.namespace AS session_namespace, s.status AS session_status"
" FROM leases l"
" JOIN work_items w ON w.work_item_id = l.work_item_id"
" LEFT JOIN sessions s ON s.session_id = l.session_id"
" ORDER BY l.expires_at DESC LIMIT ?"
)
else:
lease_sql = (
"SELECT l.lease_id, l.session_id, l.role, l.phase, l.status,"
" l.expires_at FROM leases l"
" ORDER BY l.expires_at DESC LIMIT ?"
)
lease_rows = [
dict(row)
for row in conn.execute(lease_sql, (max(1, int(limit)),)).fetchall()
]
except sqlite3.Error as exc:
return _both_unavailable(f"control-plane database read failed: {exc}")
finally:
conn.close()
elapsed = (time.perf_counter() - started) * 1000.0
now = datetime.now(timezone.utc)
sessions = tuple(
scrub(
{
"session_id": row.get("session_id"),
"role": row.get("role"),
"profile": row.get("profile"),
"namespace": row.get("namespace"),
"pid": row.get("pid"),
"pid_alive": issue_lock_store.is_process_alive(row.get("pid")),
"started_at": row.get("started_at"),
"last_heartbeat_at": row.get("last_heartbeat_at"),
"status": row.get("status"),
}
)
for row in session_rows
)
leases = tuple(
scrub(
{
"lease_id": row.get("lease_id"),
"session_id": row.get("session_id"),
"role": row.get("role"),
"phase": row.get("phase"),
"status": row.get("status"),
"expires_at": row.get("expires_at"),
"expired": _is_expired(row.get("expires_at"), now=now),
"remote": row.get("remote"),
"org": row.get("org"),
"repo": row.get("repo"),
"work_kind": row.get("work_kind"),
"work_number": row.get("work_number"),
"work_state": row.get("work_state"),
"session_pid": row.get("session_pid"),
"session_profile": row.get("session_profile"),
"session_namespace": row.get("session_namespace"),
"session_status": row.get("session_status"),
}
)
for row in lease_rows
)
degraded_reason = (
None
if has_work_items
else "work_items table absent; lease rows carry no work linkage"
)
lease_status = STATUS_OK if has_work_items else STATUS_DEGRADED
return (
InventorySection(
name="sessions",
authority=AUTHORITY_CONTROL_PLANE_DB,
status=STATUS_OK,
items=sessions,
scan_ms=round(elapsed, 3),
),
InventorySection(
name="leases",
authority=AUTHORITY_CONTROL_PLANE_DB,
status=lease_status,
items=leases,
reason=degraded_reason,
scan_ms=round(elapsed, 3),
),
)
def _is_expired(expires_at: str | None, *, now: datetime) -> bool | None:
if not expires_at:
return None
text = str(expires_at).strip().replace("Z", "+00:00")
try:
parsed = datetime.fromisoformat(text)
except ValueError:
return None
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=timezone.utc)
return parsed <= now
# ── durable issue locks (filesystem) ─────────────────────────────────────────
def _load_locks_section(*, lock_dir: str | None = None) -> InventorySection:
started = time.perf_counter()
try:
paths = issue_lock_store.iter_lock_files(lock_dir)
except OSError as exc:
return InventorySection(
name="locks",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_UNAVAILABLE,
reason=f"issue lock directory could not be listed: {exc}",
)
items: list[dict[str, Any]] = []
unreadable = 0
for path in paths:
try:
record = issue_lock_store.read_lock_file(path)
except (OSError, ValueError):
unreadable += 1
continue
if not record:
unreadable += 1
continue
try:
freshness = issue_lock_store.assess_lock_freshness(record)
except Exception: # noqa: BLE001 — a read-only view never raises
freshness = {"status": "unknown", "live": False, "stale": False}
claimant = record.get("claimant") or (
(record.get("work_lease") or {}).get("claimant") or {}
)
items.append(
scrub(
{
"issue_number": record.get("issue_number"),
"branch_name": record.get("branch_name"),
"remote": record.get("remote"),
"org": record.get("org"),
"repo": record.get("repo"),
"worktree_path": record.get("worktree_path"),
"pid": record.get("session_pid") or record.get("pid"),
"pid_alive": issue_lock_store.is_process_alive(
record.get("session_pid") or record.get("pid")
),
"claimant_username": (claimant or {}).get("username"),
"claimant_profile": (claimant or {}).get("profile"),
"lock_generation": record.get("lock_generation"),
"freshness_status": freshness.get("status"),
"live": bool(freshness.get("live")),
"stale": bool(freshness.get("stale")),
"freshness_reason": freshness.get("reason"),
"lock_path": record.get("lock_file_path") or path,
}
)
)
elapsed = (time.perf_counter() - started) * 1000.0
reason = (
f"{unreadable} lock file(s) were unreadable and are not represented"
if unreadable
else None
)
return InventorySection(
name="locks",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_DEGRADED if unreadable else STATUS_OK,
items=tuple(items),
reason=reason,
scan_ms=round(elapsed, 3),
)
# ── worktrees (filesystem, via the #432 scanner) ─────────────────────────────
def _load_worktrees_section(
*, load_hygiene: Callable[[], Any] | None = None
) -> InventorySection:
started = time.perf_counter()
try:
loader = load_hygiene
if loader is None:
from webui.worktree_scanner import load_hygiene_snapshot
loader = load_hygiene_snapshot
snapshot = loader()
except Exception as exc: # noqa: BLE001 — fail soft, never fail the request
return InventorySection(
name="worktrees",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_UNAVAILABLE,
reason=f"worktree scan failed: {exc}",
)
items = tuple(
scrub(
{
"rel_path": entry.rel_path,
"folder_name": entry.folder_name,
"classification": entry.classification,
"branch": entry.branch,
"head_sha": entry.head_sha,
"dirty_tracked": entry.dirty_tracked,
"dirty_untracked": entry.dirty_untracked,
"detached": entry.detached,
"registered_worktree": entry.registered_worktree,
"notes": entry.notes,
}
)
for entry in snapshot.entries
)
scan_error = getattr(snapshot, "scan_error", None)
elapsed = (time.perf_counter() - started) * 1000.0
return InventorySection(
name="worktrees",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_DEGRADED if scan_error else STATUS_OK,
items=items,
reason=scan_error,
scan_ms=round(elapsed, 3),
)
# ── namespaces / capability summary ──────────────────────────────────────────
def _load_namespaces_section() -> InventorySection:
started = time.perf_counter()
try:
from gitea_auth import get_profile
profile = get_profile() or {}
except Exception as exc: # noqa: BLE001
return InventorySection(
name="namespaces",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_UNAVAILABLE,
reason=f"active profile could not be resolved: {exc}",
)
allowed = list(profile.get("allowed_operations") or [])
forbidden = list(profile.get("forbidden_operations") or [])
profile_name = str(profile.get("profile_name") or "")
namespace = None
try:
import role_namespace_gate
namespace = role_namespace_gate.infer_mcp_namespace(profile_name)
except Exception: # noqa: BLE001 — namespace inference is advisory
namespace = None
item = scrub(
{
"profile_name": profile_name,
"role": profile.get("role"),
"mcp_namespace": namespace,
"allowed_operations": sorted(allowed),
"forbidden_operations": sorted(forbidden),
"capability_summary": {
"can_author": "gitea.pr.create" in allowed,
"can_review": "gitea.pr.approve" in allowed,
"can_merge": "gitea.pr.merge" in allowed,
"can_close_pr": "gitea.pr.close" in allowed,
},
"active": True,
}
)
elapsed = (time.perf_counter() - started) * 1000.0
return InventorySection(
name="namespaces",
authority=AUTHORITY_FILESYSTEM,
status=STATUS_OK,
items=(item,),
reason=(
"only the profile serving this web process is observable; other "
"namespaces are not enumerable from here"
),
scan_ms=round(elapsed, 3),
)
# ── correlation and collision detection ──────────────────────────────────────
def _issue_from_branch(branch: str | None) -> int | None:
match = re.search(r"issue-(\d+)", str(branch or ""), re.IGNORECASE)
return int(match.group(1)) if match else None
def correlate(
*,
leases: InventorySection,
locks: InventorySection,
worktrees: InventorySection,
sessions: InventorySection,
) -> tuple[tuple[dict[str, Any], ...], tuple[CollisionSignal, ...]]:
"""Join lease owner ↔ lock ↔ worktree ↔ namespace where evidence allows.
Correlation rows are emitted from whatever sections did load. Collision
signals are only emitted from sections that are ``ok``: a conflict inferred
from a partially-read source would be a false accusation.
"""
correlations: list[dict[str, Any]] = []
collisions: list[CollisionSignal] = []
worktree_by_branch: dict[str, dict[str, Any]] = {}
for entry in worktrees.items:
branch = (entry.get("branch") or "").strip()
if branch:
worktree_by_branch.setdefault(branch, entry)
session_by_id = {
str(s.get("session_id")): s for s in sessions.items if s.get("session_id")
}
# Lock-centred rows: a durable lock names an issue, a branch, and a worktree.
for lock in locks.items:
branch = (lock.get("branch_name") or "").strip()
worktree = worktree_by_branch.get(branch)
matching_leases = [
lease
for lease in leases.items
if lease.get("work_kind") == "issue"
and lease.get("work_number") == lock.get("issue_number")
]
correlations.append(
{
"issue_number": lock.get("issue_number"),
"branch": branch or None,
"lock_live": bool(lock.get("live")),
"lock_claimant": lock.get("claimant_profile"),
"lock_pid": lock.get("pid"),
"lock_pid_alive": lock.get("pid_alive"),
"worktree_rel_path": (worktree or {}).get("rel_path"),
"worktree_classification": (worktree or {}).get("classification"),
"worktree_registered": (worktree or {}).get("registered_worktree"),
"lease_ids": [
lease.get("lease_id")
for lease in matching_leases
if lease.get("lease_id")
],
"lease_sessions": [
lease.get("session_id")
for lease in matching_leases
if lease.get("session_id")
],
}
)
if locks.ok and worktrees.ok:
# A claim whose lease window is still open but has no registered
# worktree is an anomaly regardless of whether its pid is alive; a
# fully time-expired lease is on its way out and is not flagged.
if (
lock.get("freshness_status") != "expired"
and branch
and worktree is None
):
collisions.append(
CollisionSignal(
kind="lock-without-worktree",
severity="warning",
issue_number=lock.get("issue_number"),
branch=branch,
worktree_path=lock.get("worktree_path"),
message=(
f"Live lock on issue #{lock.get('issue_number')} names "
f"branch {branch!r} but no registered worktree carries "
"that branch (#404)"
),
)
)
if locks.ok:
# A lock whose recorded pid is gone is held by nobody: a clean #753
# dead-session recovery candidate. Subclassify by the lease window,
# because the two cases need different operator urgency. When the
# window is still open the lock would read as live to a naive
# timestamp check even though the owner is dead — the more dangerous
# case — so it is flagged distinctly from a fully time-expired lease.
if lock.get("pid_alive") is False and lock.get("stale"):
if lock.get("freshness_status") == "expired":
collisions.append(
CollisionSignal(
kind="stale-lock-dead-owner",
severity="warning",
issue_number=lock.get("issue_number"),
branch=branch or None,
message=(
f"Lock on issue #{lock.get('issue_number')} is stale "
f"and its recorded pid {lock.get('pid')} is not running "
"(#753 dead-session recovery candidate)"
),
)
)
else:
collisions.append(
CollisionSignal(
kind="live-lock-dead-owner",
severity="warning",
issue_number=lock.get("issue_number"),
branch=branch or None,
message=(
f"Lock on issue #{lock.get('issue_number')} has an "
"unexpired lease but its recorded pid "
f"{lock.get('pid')} is not running; it would read as "
"live to a timestamp check (#753 dead-session recovery "
"candidate)"
),
)
)
# The #635 trap: the lease has expired but the recorded pid is a
# still-running daemon, so neither dead-pid reclaim nor exact-owner
# renewal applies. This is the collision an operator must see.
elif (
lock.get("freshness_status") == "expired"
and lock.get("pid_alive") is True
):
collisions.append(
CollisionSignal(
kind="expired-lock-live-owner",
severity="error",
issue_number=lock.get("issue_number"),
branch=branch or None,
message=(
f"Lock on issue #{lock.get('issue_number')} has an expired "
f"lease but its recorded pid {lock.get('pid')} is still "
"running (daemon-pid deadlock; needs an operator decision, "
"#635/#760)"
),
)
)
# Two live locks on one branch, or two active leases on one work item.
if locks.ok:
by_branch: dict[str, list[dict[str, Any]]] = {}
for lock in locks.items:
if not lock.get("live"):
continue
branch = (lock.get("branch_name") or "").strip()
if branch:
by_branch.setdefault(branch, []).append(lock)
for branch, entries in sorted(by_branch.items()):
if len(entries) > 1:
collisions.append(
CollisionSignal(
kind="duplicate-live-lock",
severity="error",
branch=branch,
message=(
f"{len(entries)} live locks name branch {branch!r}: "
"issues "
+ ", ".join(
f"#{e.get('issue_number')}" for e in entries
)
),
)
)
if leases.ok:
by_work: dict[tuple[str, int], list[dict[str, Any]]] = {}
for lease in leases.items:
if str(lease.get("status") or "").lower() != "active":
continue
kind = str(lease.get("work_kind") or "").strip().lower()
number = lease.get("work_number")
if not kind or number is None:
continue
by_work.setdefault((kind, int(number)), []).append(lease)
for (kind, number), entries in sorted(by_work.items()):
sessions_held = {
str(e.get("session_id")) for e in entries if e.get("session_id")
}
if len(sessions_held) > 1:
collisions.append(
CollisionSignal(
kind="concurrent-active-lease",
severity="error",
issue_number=number if kind == "issue" else None,
session_ids=tuple(sorted(sessions_held)),
message=(
f"{len(sessions_held)} sessions hold an active lease on "
f"{kind} #{number}"
),
)
)
for entry in entries:
if entry.get("expired") is True:
collisions.append(
CollisionSignal(
kind="active-lease-past-expiry",
severity="warning",
issue_number=number if kind == "issue" else None,
session_ids=(
(str(entry.get("session_id")),)
if entry.get("session_id")
else ()
),
message=(
f"Lease {entry.get('lease_id')} on {kind} #{number} "
"is still marked active past its expiry"
),
)
)
# A lease whose owning session is gone is an orphan, not free work.
if leases.ok and sessions.ok:
for lease in leases.items:
if str(lease.get("status") or "").lower() != "active":
continue
session_id = str(lease.get("session_id") or "")
if session_id and session_id not in session_by_id:
collisions.append(
CollisionSignal(
kind="orphan-lease",
severity="error",
session_ids=(session_id,),
message=(
f"Active lease {lease.get('lease_id')} names session "
f"{session_id}, which has no session record"
),
)
)
return tuple(correlations), tuple(collisions)
# ── snapshot assembly ────────────────────────────────────────────────────────
def load_inventory_snapshot(
*,
db_path: str | None = None,
lock_dir: str | None = None,
load_hygiene: Callable[[], Any] | None = None,
include: tuple[str, ...] | None = None,
) -> InventorySnapshot:
"""Build the unified inventory snapshot.
Every section is loaded independently and fails soft. *include* restricts
the sections that are scanned; omitted sections are simply absent rather
than reported as empty, so a resource-split request cannot be mistaken for
a whole-inventory answer.
"""
started = time.perf_counter()
wanted = tuple(include) if include else SECTION_NAMES
sections: list[InventorySection] = []
sessions_section: InventorySection | None = None
leases_section: InventorySection | None = None
if "sessions" in wanted or "leases" in wanted:
sessions_section, leases_section = _load_cp_db_sections(db_path=db_path)
if "sessions" in wanted:
sections.append(sessions_section)
if "leases" in wanted:
sections.append(leases_section)
locks_section = (
_load_locks_section(lock_dir=lock_dir)
if "locks" in wanted
else _empty_section("locks", AUTHORITY_FILESYSTEM)
)
if "locks" in wanted:
sections.append(locks_section)
worktrees_section = (
_load_worktrees_section(load_hygiene=load_hygiene)
if "worktrees" in wanted
else _empty_section("worktrees", AUTHORITY_FILESYSTEM)
)
if "worktrees" in wanted:
sections.append(worktrees_section)
if "namespaces" in wanted:
sections.append(_load_namespaces_section())
correlations, collisions = correlate(
leases=leases_section or _empty_section("leases", AUTHORITY_CONTROL_PLANE_DB),
locks=locks_section,
worktrees=worktrees_section,
sessions=sessions_section
or _empty_section("sessions", AUTHORITY_CONTROL_PLANE_DB),
)
elapsed = (time.perf_counter() - started) * 1000.0
index = {section.name: section for section in sections}
return InventorySnapshot(
generated_at=datetime.now(timezone.utc).isoformat(),
sections=tuple(sections),
collisions=collisions,
correlations=correlations,
scan_ms=round(elapsed, 3),
_section_index=index,
)
def _empty_section(name: str, authority: str) -> InventorySection:
"""A section that was not requested — never a claim that it is empty."""
return InventorySection(
name=name,
authority=authority,
status=STATUS_UNAVAILABLE,
reason="section not requested in this scan",
)
def snapshot_to_dict(snapshot: InventorySnapshot) -> dict[str, Any]:
"""Serialize the snapshot for the versioned API."""
return {
"api_version": snapshot.api_version,
"schema_version": snapshot.schema_version,
"generated_at": snapshot.generated_at,
"status": snapshot.status,
"scan_ms": snapshot.scan_ms,
"ownership_authority_complete": snapshot.ownership_authority_complete,
"ownership_note": (
"Every ownership source read cleanly; an item absent from leases "
"and locks is genuinely unclaimed."
if snapshot.ownership_authority_complete
else "One or more ownership sources are degraded; nothing in this "
"snapshot may be treated as unowned. Collisions are reported only "
"from sections that read cleanly."
),
"degraded_sections": list(snapshot.degraded_sections),
"field_authority": {
"sessions": AUTHORITY_CONTROL_PLANE_DB,
"leases": AUTHORITY_CONTROL_PLANE_DB,
"locks": AUTHORITY_FILESYSTEM,
"worktrees": AUTHORITY_FILESYSTEM,
"namespaces": AUTHORITY_FILESYSTEM,
},
"sections": {section.name: section.to_dict() for section in snapshot.sections},
"correlations": [dict(row) for row in snapshot.correlations],
"collisions": [signal.to_dict() for signal in snapshot.collisions],
}
+1
View File
@@ -65,6 +65,7 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
)),
NavGroup("Insights", (
NavItem("/insights", "Insights", "stub"),
NavItem("/analytics", "Analytics"),
NavItem("/audit", "Audit"),
)),
)