Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
38506a753f | ||
|
|
b179610e7f | ||
|
|
0041542fe2 | ||
|
|
1f144705e2 | ||
|
|
5c5c1fdf77 |
+231
-1
@@ -31,7 +31,7 @@ from typing import Any, Iterator, Sequence
|
||||
|
||||
import dependency_graph
|
||||
|
||||
SCHEMA_VERSION = 4
|
||||
SCHEMA_VERSION = 5
|
||||
|
||||
# Assignable work kinds only — raw monitoring incidents are never work items.
|
||||
WORK_KINDS = frozenset({"issue", "pr"})
|
||||
@@ -186,6 +186,34 @@ CREATE INDEX IF NOT EXISTS idx_dependency_edges_target
|
||||
ON dependency_edges(remote, org, repo, target_kind, target_number);
|
||||
CREATE INDEX IF NOT EXISTS idx_assignments_session ON assignments(session_id, status);
|
||||
CREATE INDEX IF NOT EXISTS idx_incident_gitea ON incident_links(gitea_org, gitea_repo, gitea_issue_number);
|
||||
|
||||
-- Model usage, token cost, latency, and performance events (#651)
|
||||
CREATE TABLE IF NOT EXISTS usage_events (
|
||||
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
session_id TEXT,
|
||||
remote TEXT NOT NULL DEFAULT 'dadeschools',
|
||||
org TEXT NOT NULL DEFAULT '',
|
||||
repo TEXT NOT NULL DEFAULT '',
|
||||
project_id TEXT,
|
||||
role TEXT NOT NULL DEFAULT 'unknown',
|
||||
model TEXT NOT NULL DEFAULT 'unknown',
|
||||
issue_number INTEGER,
|
||||
pr_number INTEGER,
|
||||
stage TEXT NOT NULL DEFAULT 'unknown',
|
||||
input_tokens INTEGER,
|
||||
output_tokens INTEGER,
|
||||
total_tokens INTEGER,
|
||||
estimated_cost_usd REAL,
|
||||
latency_ms INTEGER,
|
||||
duration_ms INTEGER,
|
||||
status TEXT NOT NULL DEFAULT 'success',
|
||||
metadata TEXT,
|
||||
created_at TEXT NOT NULL
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo);
|
||||
CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model);
|
||||
CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage);
|
||||
"""
|
||||
|
||||
|
||||
@@ -339,6 +367,7 @@ class ControlPlaneDB:
|
||||
self._migrate_incident_links_null_scope(conn)
|
||||
self._migrate_lease_lifecycle_columns(conn)
|
||||
self._migrate_session_ownership_columns(conn)
|
||||
self._migrate_usage_events_table(conn)
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO schema_meta(key, value) VALUES (?, ?)",
|
||||
("schema_version", str(SCHEMA_VERSION)),
|
||||
@@ -518,6 +547,207 @@ class ControlPlaneDB:
|
||||
f"UPDATE incident_links SET {col} = '' WHERE {col} IS NULL"
|
||||
)
|
||||
|
||||
def _migrate_usage_events_table(self, conn: sqlite3.Connection) -> None:
|
||||
"""Create usage_events table and indexes if they do not exist (#651)."""
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS usage_events (
|
||||
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
session_id TEXT,
|
||||
remote TEXT NOT NULL DEFAULT 'dadeschools',
|
||||
org TEXT NOT NULL DEFAULT '',
|
||||
repo TEXT NOT NULL DEFAULT '',
|
||||
project_id TEXT,
|
||||
role TEXT NOT NULL DEFAULT 'unknown',
|
||||
model TEXT NOT NULL DEFAULT 'unknown',
|
||||
issue_number INTEGER,
|
||||
pr_number INTEGER,
|
||||
stage TEXT NOT NULL DEFAULT 'unknown',
|
||||
input_tokens INTEGER,
|
||||
output_tokens INTEGER,
|
||||
total_tokens INTEGER,
|
||||
estimated_cost_usd REAL,
|
||||
latency_ms INTEGER,
|
||||
duration_ms INTEGER,
|
||||
status TEXT NOT NULL DEFAULT 'success',
|
||||
metadata TEXT,
|
||||
created_at TEXT NOT NULL
|
||||
);
|
||||
""")
|
||||
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo);")
|
||||
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model);")
|
||||
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage);")
|
||||
|
||||
# #651 retention: cap growth so unauthenticated or high-volume ingest
|
||||
# cannot DoS the control-plane DB (PR #876 F3). Applied after every write.
|
||||
USAGE_EVENTS_MAX_ROWS = 10_000
|
||||
USAGE_EVENTS_RETENTION_DAYS = 90
|
||||
|
||||
def record_usage_event(
|
||||
self,
|
||||
*,
|
||||
session_id: str | None = None,
|
||||
remote: str = "dadeschools",
|
||||
org: str = "",
|
||||
repo: str = "",
|
||||
project_id: str | None = None,
|
||||
role: str = "unknown",
|
||||
model: str = "unknown",
|
||||
issue_number: int | None = None,
|
||||
pr_number: int | None = None,
|
||||
stage: str = "unknown",
|
||||
input_tokens: int | None = None,
|
||||
output_tokens: int | None = None,
|
||||
total_tokens: int | None = None,
|
||||
estimated_cost_usd: float | None = None,
|
||||
latency_ms: int | None = None,
|
||||
duration_ms: int | None = None,
|
||||
status: str = "success",
|
||||
metadata: str | dict[str, Any] | None = None,
|
||||
created_at: str | None = None,
|
||||
) -> int:
|
||||
"""Record a model usage, token cost, latency, or stage performance event (#651)."""
|
||||
ts = created_at or _ts()
|
||||
meta_str: str | None = None
|
||||
if metadata is not None:
|
||||
from webui import console_redaction
|
||||
redacted_meta = console_redaction.redact_payload(metadata)
|
||||
if isinstance(redacted_meta, str):
|
||||
meta_str = redacted_meta
|
||||
else:
|
||||
try:
|
||||
meta_str = json.dumps(redacted_meta, default=str)
|
||||
except Exception:
|
||||
meta_str = str(redacted_meta)
|
||||
|
||||
if total_tokens is None and (input_tokens is not None or output_tokens is not None):
|
||||
total_tokens = (input_tokens or 0) + (output_tokens or 0)
|
||||
|
||||
with self._tx(immediate=True) as conn:
|
||||
cursor = conn.execute(
|
||||
"""
|
||||
INSERT INTO usage_events (
|
||||
session_id, remote, org, repo, project_id, role, model,
|
||||
issue_number, pr_number, stage, input_tokens, output_tokens,
|
||||
total_tokens, estimated_cost_usd, latency_ms, duration_ms,
|
||||
status, metadata, created_at
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
session_id,
|
||||
remote,
|
||||
org,
|
||||
repo,
|
||||
project_id,
|
||||
role,
|
||||
model,
|
||||
issue_number,
|
||||
pr_number,
|
||||
stage,
|
||||
input_tokens,
|
||||
output_tokens,
|
||||
total_tokens,
|
||||
estimated_cost_usd,
|
||||
latency_ms,
|
||||
duration_ms,
|
||||
status,
|
||||
meta_str,
|
||||
ts,
|
||||
),
|
||||
)
|
||||
usage_id = cursor.lastrowid
|
||||
self._enforce_usage_events_retention(conn)
|
||||
return usage_id
|
||||
|
||||
def _enforce_usage_events_retention(self, conn: sqlite3.Connection) -> None:
|
||||
"""Drop aged and excess usage_events rows (PR #876 F3)."""
|
||||
# Age-based: ISO-8601 UTC timestamps compare lexicographically.
|
||||
cutoff = (
|
||||
datetime.now(timezone.utc)
|
||||
- timedelta(days=int(self.USAGE_EVENTS_RETENTION_DAYS))
|
||||
).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
conn.execute(
|
||||
"DELETE FROM usage_events WHERE created_at < ?",
|
||||
(cutoff,),
|
||||
)
|
||||
# Count-based: keep the newest USAGE_EVENTS_MAX_ROWS by usage_id.
|
||||
max_rows = int(self.USAGE_EVENTS_MAX_ROWS)
|
||||
if max_rows > 0:
|
||||
conn.execute(
|
||||
"""
|
||||
DELETE FROM usage_events
|
||||
WHERE usage_id NOT IN (
|
||||
SELECT usage_id FROM usage_events
|
||||
ORDER BY usage_id DESC
|
||||
LIMIT ?
|
||||
)
|
||||
""",
|
||||
(max_rows,),
|
||||
)
|
||||
|
||||
def query_usage_events(
|
||||
self,
|
||||
*,
|
||||
remote: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
project_id: str | None = None,
|
||||
role: str | None = None,
|
||||
model: str | None = None,
|
||||
issue_number: int | None = None,
|
||||
pr_number: int | None = None,
|
||||
stage: str | None = None,
|
||||
session_id: str | None = None,
|
||||
limit: int = 500,
|
||||
offset: int = 0,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Query stored usage events matching filters (#651)."""
|
||||
conditions = []
|
||||
params = []
|
||||
if remote:
|
||||
conditions.append("remote = ?")
|
||||
params.append(remote)
|
||||
if org:
|
||||
conditions.append("org = ?")
|
||||
params.append(org)
|
||||
if repo:
|
||||
conditions.append("repo = ?")
|
||||
params.append(repo)
|
||||
if project_id:
|
||||
conditions.append("project_id = ?")
|
||||
params.append(project_id)
|
||||
if role:
|
||||
conditions.append("role = ?")
|
||||
params.append(role)
|
||||
if model:
|
||||
conditions.append("model = ?")
|
||||
params.append(model)
|
||||
if issue_number is not None:
|
||||
conditions.append("issue_number = ?")
|
||||
params.append(issue_number)
|
||||
if pr_number is not None:
|
||||
conditions.append("pr_number = ?")
|
||||
params.append(pr_number)
|
||||
if stage:
|
||||
conditions.append("stage = ?")
|
||||
params.append(stage)
|
||||
if session_id:
|
||||
conditions.append("session_id = ?")
|
||||
params.append(session_id)
|
||||
|
||||
where_clause = f"WHERE {' AND '.join(conditions)}" if conditions else ""
|
||||
sql = f"""
|
||||
SELECT * FROM usage_events
|
||||
{where_clause}
|
||||
ORDER BY usage_id ASC
|
||||
LIMIT ? OFFSET ?
|
||||
"""
|
||||
params.extend([limit, offset])
|
||||
|
||||
with self._tx(immediate=False) as conn:
|
||||
cursor = conn.execute(sql, params)
|
||||
rows = cursor.fetchall()
|
||||
return [dict(row) for row in rows]
|
||||
|
||||
# ── sessions ──────────────────────────────────────────────────────────
|
||||
|
||||
def upsert_session(
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
# Model Usage, Token Cost, Latency, and Workflow Analytics (Phase 4)
|
||||
|
||||
- **Tracking Issue:** [#651](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
|
||||
- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
|
||||
- **Console Surface:** `/analytics`, `/api/v1/analytics`, `/api/v1/analytics/usage`
|
||||
|
||||
## 1. Overview
|
||||
|
||||
The Web Console Analytics module provides durable, aggregate visibility into **model usage, token cost, latency percentiles, and workflow-stage performance** across projects, worker roles, AI models, issues, and PRs.
|
||||
|
||||
### Non-Goals
|
||||
- No mandatory client-side telemetry that leaks prompts or secret keys.
|
||||
- No third-party payment provider or billing integration.
|
||||
- No automatic model routing changes without controller policy (#647).
|
||||
|
||||
---
|
||||
|
||||
## 2. Event Schema (`usage_events`)
|
||||
|
||||
Usage metrics are stored in the control-plane database under table `usage_events`.
|
||||
|
||||
| Column | Type | Description |
|
||||
|---|---|---|
|
||||
| `usage_id` | `INTEGER` | Primary key (autoincrement) |
|
||||
| `session_id` | `TEXT` | Optional active session identifier |
|
||||
| `remote` | `TEXT` | Known Gitea instance (`dadeschools` or `prgs`) |
|
||||
| `org` | `TEXT` | Repository owner / organization |
|
||||
| `repo` | `TEXT` | Repository name |
|
||||
| `project_id` | `TEXT` | Optional project identifier |
|
||||
| `role` | `TEXT` | Active worker role (`author`, `reviewer`, `merger`, `reconciler`, `controller`) |
|
||||
| `model` | `TEXT` | LLM model identifier (e.g. `gemini-3.6-flash`, `claude-3-5-sonnet`) |
|
||||
| `issue_number` | `INTEGER` | Correlated Gitea issue number (optional) |
|
||||
| `pr_number` | `INTEGER` | Correlated Gitea PR number (optional) |
|
||||
| `stage` | `TEXT` | Workflow stage (`preflight`, `implementation`, `review`, `merge`, `reconciliation`) |
|
||||
| `input_tokens` | `INTEGER` | Input token count (optional / nullable) |
|
||||
| `output_tokens` | `INTEGER` | Output token count (optional / nullable) |
|
||||
| `total_tokens` | `INTEGER` | Total token count (optional / nullable) |
|
||||
| `estimated_cost_usd` | `REAL` | Estimated USD cost (optional / nullable) |
|
||||
| `latency_ms` | `INTEGER` | Request latency in milliseconds (optional / nullable) |
|
||||
| `duration_ms` | `INTEGER` | Stage execution duration in milliseconds (optional / nullable) |
|
||||
| `status` | `TEXT` | Outcome status (`success`, `failure`, `timeout`) |
|
||||
| `metadata` | `TEXT` | Redacted metadata or summary string |
|
||||
| `created_at` | `TEXT` | ISO 8601 UTC timestamp |
|
||||
|
||||
---
|
||||
|
||||
## 3. Handling of Missing Data ("Unknown" vs. Zero Fabrication)
|
||||
|
||||
To ensure operational metrics accurately reflect evidence:
|
||||
- **Untracked or missing metrics are displayed as `Unknown`**, never zero-fabricated.
|
||||
- If an event omits `estimated_cost_usd`, `latency_ms`, or token counts, the aggregator marks those fields as missing (`None`) rather than defaulting to `0` or `$0.00`.
|
||||
- Summary tables and KPI cards explicitly indicate when data is unmeasured or partially reported.
|
||||
|
||||
---
|
||||
|
||||
## 4. Redaction & Security Rules
|
||||
|
||||
Per `#633` security policy:
|
||||
- Free-text fields (`metadata`, `prompt_summary`, `session_id`) are run through `console_redaction.redact_text` before persistence and output serialization.
|
||||
- Secret tokens, keychain commands, authorization headers, passwords, and JWTs are stripped automatically.
|
||||
|
||||
---
|
||||
|
||||
## 5. Opt-in Instrumentation Guide
|
||||
|
||||
Applications, MCP servers, and background sessions can report usage metrics through either Python API or HTTP ingestion.
|
||||
|
||||
### Python Ingestion
|
||||
|
||||
```python
|
||||
from webui.analytics_loader import record_usage
|
||||
|
||||
record_usage(
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role="author",
|
||||
model="gemini-3.6-flash",
|
||||
issue_number=651,
|
||||
stage="implementation",
|
||||
input_tokens=1420,
|
||||
output_tokens=380,
|
||||
total_tokens=1800,
|
||||
estimated_cost_usd=0.00045,
|
||||
latency_ms=320,
|
||||
duration_ms=4500,
|
||||
status="success",
|
||||
metadata={"note": "Implementation of analytics module"},
|
||||
)
|
||||
```
|
||||
|
||||
### HTTP Ingestion API (authorized write)
|
||||
|
||||
`POST /api/v1/analytics/usage` is a **gated write**. It runs through
|
||||
`console_authz` action `record_analytics_usage` (operator+, Phase 2 execution).
|
||||
Unauthenticated or phase-inactive requests receive **403** and do not write.
|
||||
Prefer in-process `record_usage` for MCP / session instrumentation.
|
||||
|
||||
```http
|
||||
POST /api/v1/analytics/usage HTTP/1.1
|
||||
Content-Type: application/json
|
||||
# Requires authenticated principal with record_analytics_usage execution enabled
|
||||
|
||||
{
|
||||
"remote": "dadeschools",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"role": "author",
|
||||
"model": "gemini-3.6-flash",
|
||||
"issue_number": 651,
|
||||
"stage": "implementation",
|
||||
"input_tokens": 1420,
|
||||
"output_tokens": 380,
|
||||
"total_tokens": 1800,
|
||||
"estimated_cost_usd": 0.00045,
|
||||
"latency_ms": 320,
|
||||
"duration_ms": 4500,
|
||||
"status": "success",
|
||||
"metadata": "Analytics schema landed"
|
||||
}
|
||||
```
|
||||
|
||||
### Retention
|
||||
|
||||
`usage_events` is retained with hard caps applied on every write:
|
||||
|
||||
| Limit | Default |
|
||||
|---|---|
|
||||
| Max rows | 10,000 (`ControlPlaneDB.USAGE_EVENTS_MAX_ROWS`) |
|
||||
| Max age | 90 days (`ControlPlaneDB.USAGE_EVENTS_RETENTION_DAYS`) |
|
||||
|
||||
Older rows (by `created_at`) and excess oldest rows (by `usage_id`) are deleted
|
||||
after each insert so unbounded growth / DoS-by-volume cannot fill the DB.
|
||||
|
||||
---
|
||||
|
||||
## 6. Querying Analytics API
|
||||
|
||||
```http
|
||||
GET /api/v1/analytics?role=author&stage=implementation HTTP/1.1
|
||||
```
|
||||
|
||||
Returns `AnalyticsSnapshot` JSON containing aggregations (`by_model`, `by_stage`, `by_role`, `by_work_item`, `by_project`) and latency percentiles (`p50`, `p90`, `p95`, `p99`).
|
||||
@@ -91,6 +91,7 @@ already define, and a regression test asserts each mapping matches.
|
||||
| `close_pr` | controller | privileged | `gitea.pr.close` | Yes | No | No | 3 |
|
||||
| `merge_pr` | controller | privileged | `gitea.pr.merge` | Yes | **Yes** | **Yes** | 3 |
|
||||
| `delete_branch` | admin | destructive | `gitea.branch.delete` | Yes | **Yes** | **Yes** | 3 |
|
||||
| `record_analytics_usage` | operator | gated_write | `runtime.record_analytics_usage` | Yes | No | No | 2 |
|
||||
| `system.reload_namespace` | controller | privileged | `runtime.reload_namespace` | Yes | No | No | 2 |
|
||||
| `system.restart_namespace` | admin | destructive | `runtime.restart_namespace` | Yes | **Yes** | **Yes** | 2 |
|
||||
|
||||
|
||||
@@ -349,53 +349,6 @@ health, workflow/schema SHA-256 hashes, and stale-runtime warnings when the
|
||||
checkout is behind merged safety-gate changes. Restart guidance links to #420;
|
||||
no tokens or MCP restart actions are exposed.
|
||||
|
||||
## Inventory API (#636)
|
||||
|
||||
`GET /api/v1/inventory` returns one versioned, read-only snapshot that unifies
|
||||
what the lease (#433), worktree (#432), and runtime (#430) MVP views each show
|
||||
separately, so traffic-control and recovery consumers read the same source.
|
||||
`GET /api/v1/inventory/{section}` returns a single section under the identical
|
||||
schema (`sessions`, `leases`, `locks`, `worktrees`, `namespaces`); an unknown
|
||||
section is a `404` with `error: unknown_section`. Both routes are `GET`-only.
|
||||
|
||||
Each section carries its own `status` (`ok` / `degraded` / `unavailable`), a
|
||||
`reason` when not `ok`, and a `scan_ms`. A subsystem that cannot be read
|
||||
degrades to a reasoned section; it never raises and never emits an empty list
|
||||
that would read as "nothing is there".
|
||||
|
||||
### Field authority
|
||||
|
||||
Every section names where its rows came from; authorities are never blended.
|
||||
|
||||
| Section | Authority | Source |
|
||||
|---|---|---|
|
||||
| `sessions` | `control_plane_db` | #613 control-plane DB (`mode=ro`), authoritative for exclusive ownership (#600/#601) |
|
||||
| `leases` | `control_plane_db` | #613 control-plane DB; degrades if the `work_items` table is absent |
|
||||
| `locks` | `filesystem` | durable per-issue lock files (`issue_lock_store`) |
|
||||
| `worktrees` | `filesystem` | registered git worktrees via the #432 hygiene scanner |
|
||||
| `namespaces` | `filesystem` | the active profile serving this web process (others are not enumerable) |
|
||||
|
||||
The payload restates this map under `field_authority` for machine consumers.
|
||||
|
||||
### Ownership safety
|
||||
|
||||
`ownership_authority_complete` is true only when every ownership-bearing section
|
||||
(`sessions`, `leases`, `locks`) read cleanly. While it is false, nothing is
|
||||
reported as unowned and no collision is asserted from a degraded source —
|
||||
absence of evidence is reported as absence of evidence, never as free work.
|
||||
|
||||
`collisions` surfaces detectable conflicts, each with a `kind` and `severity`:
|
||||
`lock-without-worktree`, `duplicate-live-lock`, `live-lock-dead-owner` (unexpired
|
||||
lease, dead pid — a #753 recovery candidate that would read as live to a naive
|
||||
timestamp check), `stale-lock-dead-owner`, `expired-lock-live-owner` (the
|
||||
#635/#760 daemon-pid deadlock), `concurrent-active-lease`, `active-lease-past-expiry`,
|
||||
and `orphan-lease`. Collisions are emitted only from sections that read cleanly.
|
||||
|
||||
The control-plane DB is opened through a `mode=ro` URI so a read never creates
|
||||
or migrates it; paths are collapsed against `$HOME`, URLs lose userinfo and
|
||||
query strings, and credential-shaped values are redacted at the boundary. Lease
|
||||
steal/release and worktree deletion are Phase 2+ and have no representation here.
|
||||
|
||||
## Workflow-event timeline (#637)
|
||||
|
||||
`GET /api/v1/timeline` is a read-only, versioned aggregation of workflow
|
||||
|
||||
@@ -510,6 +510,15 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
},
|
||||
|
||||
# #651 console analytics ingest — control-plane DB write, not a Gitea API
|
||||
# call. Authority comes from console RBAC (operator+) plus phase gating;
|
||||
# permission string is a non-Gitea runtime capability so no Gitea profile
|
||||
# can satisfy it by accident.
|
||||
"record_analytics_usage": {
|
||||
"permission": "runtime.record_analytics_usage",
|
||||
"role": "author",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -36,7 +36,7 @@ class ControlPlaneDBTest(unittest.TestCase):
|
||||
rows = dict(conn.execute("SELECT key, value FROM schema_meta").fetchall())
|
||||
finally:
|
||||
conn.close()
|
||||
self.assertEqual(rows["schema_version"], "4")
|
||||
self.assertEqual(rows["schema_version"], "5")
|
||||
self.assertIn("DB coordinates", rows["architecture"])
|
||||
self.assertIn("bridge", rows["architecture"].lower())
|
||||
|
||||
|
||||
@@ -0,0 +1,252 @@
|
||||
"""Unit and integration tests for Model Usage & Performance Analytics (#651)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
import control_plane_db
|
||||
from webui.analytics_loader import (
|
||||
ANALYTICS_SCHEMA_VERSION,
|
||||
compute_percentile,
|
||||
load_analytics,
|
||||
record_usage,
|
||||
)
|
||||
from webui.app import create_app
|
||||
from webui import console_redaction
|
||||
|
||||
|
||||
class AnalyticsLoaderTest(unittest.TestCase):
|
||||
|
||||
def setUp(self) -> None:
|
||||
self.temp_dir = tempfile.TemporaryDirectory()
|
||||
self.db_path = os.path.join(self.temp_dir.name, "test_control_plane.sqlite3")
|
||||
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
|
||||
self.db = control_plane_db.ControlPlaneDB(db_path=self.db_path)
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self.temp_dir.cleanup()
|
||||
|
||||
def test_compute_percentile(self) -> None:
|
||||
self.assertIsNone(compute_percentile([], 50.0))
|
||||
self.assertEqual(compute_percentile([100], 50.0), 100.0)
|
||||
|
||||
# 2 elements: [100, 200]
|
||||
self.assertEqual(compute_percentile([100, 200], 50.0), 150.0)
|
||||
|
||||
# 100 elements: 1..100
|
||||
vals = list(range(1, 101))
|
||||
self.assertEqual(compute_percentile(vals, 50.0), 50.5)
|
||||
self.assertAlmostEqual(compute_percentile(vals, 90.0), 90.1)
|
||||
|
||||
def test_record_and_aggregate_usage(self) -> None:
|
||||
# Record event 1 (complete data)
|
||||
u1 = record_usage(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role="author",
|
||||
model="gemini-3.6-flash",
|
||||
issue_number=651,
|
||||
stage="implementation",
|
||||
input_tokens=1000,
|
||||
output_tokens=500,
|
||||
estimated_cost_usd=0.0015,
|
||||
latency_ms=200,
|
||||
duration_ms=3000,
|
||||
metadata={"secret_key": "secret123", "note": "token=secret123"},
|
||||
)
|
||||
self.assertGreater(u1, 0)
|
||||
|
||||
# Record event 2 (missing tokens and cost -> unknown)
|
||||
u2 = record_usage(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role="reviewer",
|
||||
model="claude-3-5-sonnet",
|
||||
pr_number=846,
|
||||
stage="review",
|
||||
latency_ms=500,
|
||||
duration_ms=6000,
|
||||
)
|
||||
self.assertGreater(u2, u1)
|
||||
|
||||
snapshot = load_analytics(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
)
|
||||
|
||||
self.assertTrue(snapshot.ok)
|
||||
self.assertEqual(snapshot.schema_version, ANALYTICS_SCHEMA_VERSION)
|
||||
self.assertEqual(snapshot.total_events, 2)
|
||||
|
||||
# Verify overall summary
|
||||
summary = snapshot.overall_summary
|
||||
self.assertEqual(summary.total_events, 2)
|
||||
self.assertEqual(summary.events_with_tokens, 1)
|
||||
self.assertEqual(summary.total_tokens, 1500)
|
||||
self.assertEqual(summary.events_with_cost, 1)
|
||||
self.assertEqual(summary.estimated_cost_usd, 0.0015)
|
||||
self.assertEqual(summary.events_with_latency, 2)
|
||||
self.assertEqual(summary.latency_p50_ms, 350.0)
|
||||
|
||||
# Verify missing data handling (AC 3: not zero-fabricated)
|
||||
reviewer_model = snapshot.by_model.get("claude-3-5-sonnet")
|
||||
self.assertIsNotNone(reviewer_model)
|
||||
self.assertEqual(reviewer_model.total_events, 1)
|
||||
self.assertEqual(reviewer_model.events_with_tokens, 0)
|
||||
self.assertIsNone(reviewer_model.total_tokens)
|
||||
self.assertEqual(reviewer_model.display_tokens, "Unknown")
|
||||
self.assertEqual(reviewer_model.events_with_cost, 0)
|
||||
self.assertIsNone(reviewer_model.estimated_cost_usd)
|
||||
self.assertEqual(reviewer_model.display_cost, "Unknown")
|
||||
|
||||
# Verify redaction (AC 4)
|
||||
e1 = [e for e in snapshot.events if e.usage_id == u1][0]
|
||||
self.assertIsNotNone(e1.metadata)
|
||||
self.assertNotIn("secret123", e1.metadata)
|
||||
self.assertIn("[REDACTED]", e1.metadata)
|
||||
|
||||
def test_missing_db_fail_soft(self) -> None:
|
||||
invalid_path = "/nonexistent_path_dir/db.sqlite3"
|
||||
snapshot = load_analytics(db_path=invalid_path)
|
||||
self.assertFalse(snapshot.ok)
|
||||
self.assertIn("control_plane_db_unavailable", snapshot.reason)
|
||||
self.assertEqual(snapshot.overall_summary.display_tokens, "Unknown")
|
||||
|
||||
|
||||
class AnalyticsWebUITest(unittest.TestCase):
|
||||
|
||||
def setUp(self) -> None:
|
||||
self.temp_dir = tempfile.TemporaryDirectory()
|
||||
self.db_path = os.path.join(self.temp_dir.name, "test_webui.sqlite3")
|
||||
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
|
||||
self.app = create_app()
|
||||
self.client = TestClient(self.app)
|
||||
|
||||
record_usage(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role="author",
|
||||
model="gemini-3.6-flash",
|
||||
issue_number=651,
|
||||
stage="implementation",
|
||||
input_tokens=2000,
|
||||
output_tokens=1000,
|
||||
estimated_cost_usd=0.003,
|
||||
latency_ms=150,
|
||||
duration_ms=2500,
|
||||
)
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self.temp_dir.cleanup()
|
||||
|
||||
def test_analytics_html_route(self) -> None:
|
||||
response = self.client.get("/analytics")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertIn("Model Usage & Performance Analytics", response.text)
|
||||
self.assertIn("gemini-3.6-flash", response.text)
|
||||
self.assertIn("3,000", response.text)
|
||||
|
||||
def test_analytics_api_route(self) -> None:
|
||||
response = self.client.get("/api/v1/analytics")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
data = response.json()
|
||||
self.assertTrue(data["ok"])
|
||||
self.assertEqual(data["total_events"], 1)
|
||||
self.assertIn("gemini-3.6-flash", data["by_model"])
|
||||
|
||||
def test_analytics_ingest_unauthorized_denied(self) -> None:
|
||||
"""F2: unauthenticated POST must not write the control-plane DB."""
|
||||
payload = {
|
||||
"remote": "dadeschools",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"role": "reviewer",
|
||||
"model": "claude-3-5-sonnet",
|
||||
"pr_number": 846,
|
||||
"stage": "review",
|
||||
"input_tokens": 500,
|
||||
"output_tokens": 100,
|
||||
"latency_ms": 400,
|
||||
"metadata": "Review note token=secret456",
|
||||
}
|
||||
response = self.client.post("/api/v1/analytics/usage", json=payload)
|
||||
self.assertEqual(response.status_code, 403)
|
||||
res_json = response.json()
|
||||
self.assertFalse(res_json.get("ok", True))
|
||||
self.assertEqual(res_json.get("error"), "unauthorized")
|
||||
authorization = res_json.get("authorization") or {}
|
||||
self.assertFalse(authorization.get("allowed"))
|
||||
self.assertFalse(authorization.get("execution_enabled"))
|
||||
|
||||
# No new row written
|
||||
res2 = self.client.get("/api/v1/analytics")
|
||||
self.assertEqual(res2.status_code, 200)
|
||||
self.assertEqual(res2.json()["total_events"], 1)
|
||||
|
||||
def test_html_escapes_script_bearing_model_role_stage(self) -> None:
|
||||
"""F1: stored XSS — dynamic model/role/stage must render escaped."""
|
||||
xss = '<script>alert(1)</script>'
|
||||
record_usage(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role=xss,
|
||||
model=xss,
|
||||
stage=xss,
|
||||
issue_number=999,
|
||||
status="success",
|
||||
)
|
||||
response = self.client.get("/analytics")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
# Raw tag must not appear; escaped form must.
|
||||
self.assertNotIn("<script>alert(1)</script>", response.text)
|
||||
self.assertIn("<script>alert(1)</script>", response.text)
|
||||
|
||||
def test_load_analytics_coerces_none_scope(self) -> None:
|
||||
"""F4: None remote/org/repo become empty strings, never None."""
|
||||
snapshot = load_analytics(db_path=self.db_path, remote=None, org=None, repo=None)
|
||||
self.assertIsInstance(snapshot.remote, str)
|
||||
self.assertIsInstance(snapshot.org, str)
|
||||
self.assertIsInstance(snapshot.repo, str)
|
||||
self.assertEqual(snapshot.remote, "")
|
||||
self.assertEqual(snapshot.org, "")
|
||||
self.assertEqual(snapshot.repo, "")
|
||||
|
||||
def test_usage_events_retention_max_rows(self) -> None:
|
||||
"""F3: record_usage_event enforces USAGE_EVENTS_MAX_ROWS."""
|
||||
db = control_plane_db.ControlPlaneDB(db_path=self.db_path)
|
||||
original_max = db.USAGE_EVENTS_MAX_ROWS
|
||||
try:
|
||||
db.USAGE_EVENTS_MAX_ROWS = 3
|
||||
for i in range(5):
|
||||
db.record_usage_event(
|
||||
remote="dadeschools",
|
||||
org="org",
|
||||
repo="repo",
|
||||
role="author",
|
||||
model=f"model-{i}",
|
||||
stage="test",
|
||||
)
|
||||
rows = db.query_usage_events(limit=100)
|
||||
self.assertLessEqual(len(rows), 3)
|
||||
# Newest three retained
|
||||
models = {r["model"] for r in rows}
|
||||
self.assertEqual(models, {"model-2", "model-3", "model-4"})
|
||||
finally:
|
||||
db.USAGE_EVENTS_MAX_ROWS = original_max
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -1,468 +0,0 @@
|
||||
"""Tests for the unified web-console inventory API (#636).
|
||||
|
||||
Covers the four cases the issue names — empty, populated, partial failure, and
|
||||
the no-false-unowned invariant — plus redaction, collision detection, the
|
||||
resource-split routes, and read-only guarantees against a real control-plane
|
||||
database and real durable lock files.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
import control_plane_db
|
||||
from webui.app import create_app
|
||||
from webui import inventory
|
||||
|
||||
|
||||
def _iso(dt: datetime) -> str:
|
||||
return dt.astimezone(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def _write_lock(lock_dir: str, name: str, payload: dict) -> str:
|
||||
path = os.path.join(lock_dir, name)
|
||||
with open(path, "w", encoding="utf-8") as handle:
|
||||
json.dump(payload, handle)
|
||||
return path
|
||||
|
||||
|
||||
def _live_lock_payload(
|
||||
*,
|
||||
issue_number: int,
|
||||
branch: str,
|
||||
worktree_path: str,
|
||||
pid: int,
|
||||
username: str = "jcwalker3",
|
||||
profile: str = "prgs-author",
|
||||
) -> dict:
|
||||
now = datetime.now(timezone.utc)
|
||||
future = now + timedelta(hours=2)
|
||||
return {
|
||||
"branch_name": branch,
|
||||
"issue_number": issue_number,
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"remote": "prgs",
|
||||
"pid": pid,
|
||||
"session_pid": pid,
|
||||
"lock_generation": 1,
|
||||
"worktree_path": worktree_path,
|
||||
"claimant": {"username": username, "profile": profile},
|
||||
"work_lease": {
|
||||
"branch": branch,
|
||||
"issue_number": issue_number,
|
||||
"operation_type": "author_issue_work",
|
||||
"created_at": _iso(now),
|
||||
"expires_at": _iso(future),
|
||||
"last_heartbeat_at": _iso(now),
|
||||
"claimant": {"username": username, "profile": profile},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
class _FixtureMixin(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.tmp = self._tmp.name
|
||||
self.lock_dir = os.path.join(self.tmp, "locks")
|
||||
os.makedirs(self.lock_dir, mode=0o700)
|
||||
self.db_path = os.path.join(self.tmp, "control_plane.db")
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
|
||||
def _seed_db(self) -> control_plane_db.ControlPlaneDB:
|
||||
db = control_plane_db.ControlPlaneDB(self.db_path)
|
||||
db.upsert_session(
|
||||
session_id="prgs-author-1",
|
||||
role="author",
|
||||
profile="prgs-author",
|
||||
namespace="gitea-author",
|
||||
pid=os.getpid(),
|
||||
)
|
||||
db.upsert_work_item(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
kind="issue",
|
||||
number=636,
|
||||
)
|
||||
db.assign_and_lease(
|
||||
session_id="prgs-author-1",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
kind="issue",
|
||||
number=636,
|
||||
)
|
||||
return db
|
||||
|
||||
|
||||
class TestRedaction(unittest.TestCase):
|
||||
def test_redact_path_collapses_home(self):
|
||||
home = os.path.expanduser("~")
|
||||
self.assertEqual(
|
||||
inventory.redact_path(f"{home}/Development/Gitea-Tools"),
|
||||
"~/Development/Gitea-Tools",
|
||||
)
|
||||
|
||||
def test_redact_url_strips_userinfo_and_query(self):
|
||||
self.assertEqual(
|
||||
inventory.redact_url("https://user:[email protected]/api?token=abc"),
|
||||
"https://gitea.prgs.cc/api",
|
||||
)
|
||||
|
||||
def test_scrub_drops_credential_keys(self):
|
||||
scrubbed = inventory.scrub(
|
||||
{"token": "abc123", "api_key": "k", "profile": "prgs-author"}
|
||||
)
|
||||
self.assertEqual(scrubbed["token"], "[redacted]")
|
||||
self.assertEqual(scrubbed["api_key"], "[redacted]")
|
||||
self.assertEqual(scrubbed["profile"], "prgs-author")
|
||||
|
||||
def test_scrub_is_recursive_and_never_raises(self):
|
||||
class Weird:
|
||||
def __repr__(self) -> str:
|
||||
return "weird-obj"
|
||||
|
||||
out = inventory.scrub({"nested": [{"password": "p", "obj": Weird()}]})
|
||||
self.assertEqual(out["nested"][0]["password"], "[redacted]")
|
||||
self.assertEqual(out["nested"][0]["obj"], "weird-obj")
|
||||
|
||||
|
||||
class TestEmptyInventory(_FixtureMixin):
|
||||
def test_empty_db_and_locks_degrade_without_raising(self):
|
||||
# No DB file, no locks: sessions/leases unavailable, locks ok+empty.
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=lambda: _StubHygiene(entries=()),
|
||||
)
|
||||
sessions = snap.section("sessions")
|
||||
leases = snap.section("leases")
|
||||
locks = snap.section("locks")
|
||||
self.assertEqual(sessions.status, inventory.STATUS_UNAVAILABLE)
|
||||
self.assertEqual(leases.status, inventory.STATUS_UNAVAILABLE)
|
||||
self.assertEqual(locks.status, inventory.STATUS_OK)
|
||||
self.assertEqual(len(locks.items), 0)
|
||||
# Ownership authority is incomplete because the DB is missing.
|
||||
self.assertFalse(snap.ownership_authority_complete)
|
||||
self.assertEqual(snap.collisions, ())
|
||||
|
||||
def test_empty_db_present_but_unpopulated(self):
|
||||
control_plane_db.ControlPlaneDB(self.db_path) # creates schema, no rows
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=lambda: _StubHygiene(entries=()),
|
||||
)
|
||||
self.assertEqual(snap.section("sessions").status, inventory.STATUS_OK)
|
||||
self.assertEqual(len(snap.section("sessions").items), 0)
|
||||
self.assertEqual(snap.section("leases").status, inventory.STATUS_OK)
|
||||
self.assertTrue(snap.ownership_authority_complete)
|
||||
|
||||
|
||||
class TestPopulatedInventory(_FixtureMixin):
|
||||
def test_sections_populated_and_correlated(self):
|
||||
self._seed_db()
|
||||
wt = f"{self.tmp}/branches/issue-636-inventory-api"
|
||||
_write_lock(
|
||||
self.lock_dir,
|
||||
"prgs-Scaled-Tech-Consulting-Gitea-Tools-636.json",
|
||||
_live_lock_payload(
|
||||
issue_number=636,
|
||||
branch="feat/issue-636-inventory-api",
|
||||
worktree_path=wt,
|
||||
pid=os.getpid(),
|
||||
),
|
||||
)
|
||||
hygiene = _StubHygiene(
|
||||
entries=(
|
||||
_StubEntry(
|
||||
rel_path="branches/issue-636-inventory-api",
|
||||
branch="feat/issue-636-inventory-api",
|
||||
classification="active-issue",
|
||||
),
|
||||
)
|
||||
)
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=lambda: hygiene,
|
||||
)
|
||||
self.assertTrue(snap.ownership_authority_complete)
|
||||
self.assertEqual(len(snap.section("sessions").items), 1)
|
||||
self.assertEqual(len(snap.section("leases").items), 1)
|
||||
self.assertEqual(len(snap.section("locks").items), 1)
|
||||
self.assertEqual(len(snap.section("worktrees").items), 1)
|
||||
|
||||
# The lease, lock, and worktree for #636 correlate onto one row.
|
||||
row = next(r for r in snap.correlations if r["issue_number"] == 636)
|
||||
self.assertEqual(row["branch"], "feat/issue-636-inventory-api")
|
||||
self.assertTrue(row["lock_live"])
|
||||
self.assertEqual(row["worktree_classification"], "active-issue")
|
||||
self.assertEqual(len(row["lease_ids"]), 1)
|
||||
# No collision: live lock, live pid, matching worktree.
|
||||
self.assertEqual(snap.collisions, ())
|
||||
|
||||
def test_serialized_payload_declares_field_authority(self):
|
||||
self._seed_db()
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=lambda: _StubHygiene(entries=()),
|
||||
)
|
||||
payload = inventory.snapshot_to_dict(snap)
|
||||
self.assertEqual(payload["api_version"], "v1")
|
||||
self.assertEqual(payload["schema_version"], 1)
|
||||
self.assertEqual(payload["field_authority"]["sessions"], "control_plane_db")
|
||||
self.assertEqual(payload["field_authority"]["locks"], "filesystem")
|
||||
self.assertIn("sessions", payload["sections"])
|
||||
|
||||
|
||||
class TestPartialFailure(_FixtureMixin):
|
||||
def test_worktree_scan_failure_degrades_only_that_section(self):
|
||||
self._seed_db()
|
||||
|
||||
def _boom():
|
||||
raise RuntimeError("git worktree list exploded")
|
||||
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=_boom,
|
||||
)
|
||||
self.assertEqual(
|
||||
snap.section("worktrees").status, inventory.STATUS_UNAVAILABLE
|
||||
)
|
||||
self.assertIn("exploded", snap.section("worktrees").reason)
|
||||
# DB-backed sections still healthy.
|
||||
self.assertEqual(snap.section("sessions").status, inventory.STATUS_OK)
|
||||
self.assertIn("worktrees", snap.degraded_sections)
|
||||
|
||||
def test_degraded_ownership_suppresses_unowned_claim(self):
|
||||
# DB absent → sessions/leases unavailable → ownership incomplete even
|
||||
# though a lock exists and could look "unclaimed" by the DB alone.
|
||||
_write_lock(
|
||||
self.lock_dir,
|
||||
"prgs-Scaled-Tech-Consulting-Gitea-Tools-636.json",
|
||||
_live_lock_payload(
|
||||
issue_number=636,
|
||||
branch="feat/issue-636-inventory-api",
|
||||
worktree_path=f"{self.tmp}/wt",
|
||||
pid=os.getpid(),
|
||||
),
|
||||
)
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=lambda: _StubHygiene(entries=()),
|
||||
)
|
||||
self.assertFalse(snap.ownership_authority_complete)
|
||||
payload = inventory.snapshot_to_dict(snap)
|
||||
self.assertIn("may be treated as unowned", payload["ownership_note"])
|
||||
|
||||
|
||||
class TestCollisionDetection(_FixtureMixin):
|
||||
def test_live_lock_dead_owner_flagged(self):
|
||||
_write_lock(
|
||||
self.lock_dir,
|
||||
"prgs-Scaled-Tech-Consulting-Gitea-Tools-700.json",
|
||||
_live_lock_payload(
|
||||
issue_number=700,
|
||||
branch="feat/issue-700-x",
|
||||
worktree_path=f"{self.tmp}/wt700",
|
||||
pid=999_999_999, # not a running pid
|
||||
),
|
||||
)
|
||||
control_plane_db.ControlPlaneDB(self.db_path) # empty but present
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=lambda: _StubHygiene(entries=()),
|
||||
)
|
||||
kinds = {c.kind for c in snap.collisions}
|
||||
self.assertIn("live-lock-dead-owner", kinds)
|
||||
# Also lock-without-worktree, since no worktree carries the branch.
|
||||
self.assertIn("lock-without-worktree", kinds)
|
||||
|
||||
def test_duplicate_live_lock_on_same_branch(self):
|
||||
for issue in (800, 801):
|
||||
_write_lock(
|
||||
self.lock_dir,
|
||||
f"prgs-Scaled-Tech-Consulting-Gitea-Tools-{issue}.json",
|
||||
_live_lock_payload(
|
||||
issue_number=issue,
|
||||
branch="feat/issue-800-shared",
|
||||
worktree_path=f"{self.tmp}/wt{issue}",
|
||||
pid=os.getpid(),
|
||||
),
|
||||
)
|
||||
control_plane_db.ControlPlaneDB(self.db_path)
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=lambda: _StubHygiene(entries=()),
|
||||
)
|
||||
self.assertIn(
|
||||
"duplicate-live-lock", {c.kind for c in snap.collisions}
|
||||
)
|
||||
|
||||
def test_no_collision_when_sections_degraded(self):
|
||||
# locks ok but worktrees unavailable → lock-without-worktree must NOT
|
||||
# be asserted (a missing scan is not a missing worktree).
|
||||
_write_lock(
|
||||
self.lock_dir,
|
||||
"prgs-Scaled-Tech-Consulting-Gitea-Tools-636.json",
|
||||
_live_lock_payload(
|
||||
issue_number=636,
|
||||
branch="feat/issue-636-inventory-api",
|
||||
worktree_path=f"{self.tmp}/wt",
|
||||
pid=os.getpid(),
|
||||
),
|
||||
)
|
||||
control_plane_db.ControlPlaneDB(self.db_path)
|
||||
|
||||
def _boom():
|
||||
raise RuntimeError("scan down")
|
||||
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=_boom,
|
||||
)
|
||||
self.assertNotIn(
|
||||
"lock-without-worktree", {c.kind for c in snap.collisions}
|
||||
)
|
||||
|
||||
|
||||
class TestSectionInclude(_FixtureMixin):
|
||||
def test_include_restricts_scanned_sections(self):
|
||||
self._seed_db()
|
||||
snap = inventory.load_inventory_snapshot(
|
||||
db_path=self.db_path,
|
||||
lock_dir=self.lock_dir,
|
||||
include=("locks",),
|
||||
)
|
||||
self.assertIsNotNone(snap.section("locks"))
|
||||
self.assertIsNone(snap.section("sessions"))
|
||||
self.assertIsNone(snap.section("worktrees"))
|
||||
|
||||
|
||||
class TestRoutes(_FixtureMixin):
|
||||
def setUp(self) -> None:
|
||||
super().setUp()
|
||||
# Point the loaders at the fixture DB and lock dir via env, and stub
|
||||
# the worktree scan so the route does not shell out to git.
|
||||
self._prev_env = {
|
||||
"GITEA_CONTROL_PLANE_DB": os.environ.get("GITEA_CONTROL_PLANE_DB"),
|
||||
"GITEA_ISSUE_LOCK_DIR": os.environ.get("GITEA_ISSUE_LOCK_DIR"),
|
||||
"WEBUI_TEST_OFFLINE": os.environ.get("WEBUI_TEST_OFFLINE"),
|
||||
}
|
||||
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
|
||||
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir
|
||||
os.environ["WEBUI_TEST_OFFLINE"] = "1"
|
||||
self._seed_db()
|
||||
self.client = TestClient(create_app())
|
||||
|
||||
def tearDown(self) -> None:
|
||||
for key, value in self._prev_env.items():
|
||||
if value is None:
|
||||
os.environ.pop(key, None)
|
||||
else:
|
||||
os.environ[key] = value
|
||||
|
||||
def test_inventory_route_returns_versioned_payload(self):
|
||||
resp = self.client.get("/api/v1/inventory")
|
||||
self.assertEqual(resp.status_code, 200)
|
||||
body = resp.json()
|
||||
self.assertEqual(body["api_version"], "v1")
|
||||
self.assertIn("sessions", body["sections"])
|
||||
self.assertIn("field_authority", body)
|
||||
|
||||
def test_section_route_restricts_and_labels(self):
|
||||
resp = self.client.get("/api/v1/inventory/locks")
|
||||
self.assertEqual(resp.status_code, 200)
|
||||
body = resp.json()
|
||||
self.assertEqual(body["requested_section"], "locks")
|
||||
self.assertIn("locks", body["sections"])
|
||||
self.assertNotIn("sessions", body["sections"])
|
||||
|
||||
def test_unknown_section_is_404(self):
|
||||
resp = self.client.get("/api/v1/inventory/bogus")
|
||||
self.assertEqual(resp.status_code, 404)
|
||||
self.assertEqual(resp.json()["error"], "unknown_section")
|
||||
|
||||
def test_inventory_route_rejects_post(self):
|
||||
resp = self.client.post("/api/v1/inventory")
|
||||
self.assertEqual(resp.status_code, 405)
|
||||
|
||||
|
||||
class TestReadOnly(_FixtureMixin):
|
||||
def test_snapshot_does_not_create_db_file(self):
|
||||
missing = os.path.join(self.tmp, "does-not-exist.db")
|
||||
inventory.load_inventory_snapshot(
|
||||
db_path=missing,
|
||||
lock_dir=self.lock_dir,
|
||||
load_hygiene=lambda: _StubHygiene(entries=()),
|
||||
)
|
||||
self.assertFalse(os.path.exists(missing))
|
||||
|
||||
def test_readonly_connection_refuses_write(self):
|
||||
self._seed_db()
|
||||
conn = inventory._open_readonly(self.db_path)
|
||||
try:
|
||||
with self.assertRaises(Exception):
|
||||
conn.execute(
|
||||
"INSERT INTO sessions(session_id, role, started_at, "
|
||||
"last_heartbeat_at, status) VALUES ('x','author',"
|
||||
"'t','t','active')"
|
||||
)
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
# ── lightweight stand-ins for the #432 hygiene snapshot ──────────────────────
|
||||
|
||||
|
||||
class _StubEntry:
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
rel_path: str,
|
||||
branch: str | None = None,
|
||||
classification: str = "stale-clean",
|
||||
head_sha: str | None = "abc123",
|
||||
dirty_tracked: int = 0,
|
||||
dirty_untracked: bool = False,
|
||||
detached: bool = False,
|
||||
registered_worktree: bool = True,
|
||||
notes: str = "",
|
||||
) -> None:
|
||||
self.rel_path = rel_path
|
||||
self.folder_name = rel_path.split("/", 1)[-1]
|
||||
self.branch = branch
|
||||
self.classification = classification
|
||||
self.head_sha = head_sha
|
||||
self.dirty_tracked = dirty_tracked
|
||||
self.dirty_untracked = dirty_untracked
|
||||
self.detached = detached
|
||||
self.registered_worktree = registered_worktree
|
||||
self.notes = notes
|
||||
|
||||
|
||||
class _StubHygiene:
|
||||
def __init__(self, *, entries=(), scan_error=None) -> None:
|
||||
self.entries = tuple(entries)
|
||||
self.scan_error = scan_error
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,433 @@
|
||||
"""Model usage, token cost, latency, and workflow-performance analytics (#651, Phase 4).
|
||||
|
||||
Ingests session instrumentation metrics, aggregates usage/cost/latency percentiles
|
||||
by project, role, model, issue/PR, and stage, enforcing secret redaction and
|
||||
explicitly rendering missing metrics as "Unknown" without zero-fabrication.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Any, Sequence
|
||||
|
||||
import control_plane_db
|
||||
from webui import console_redaction
|
||||
|
||||
ANALYTICS_SCHEMA_VERSION = 1
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class UsageEvent:
|
||||
usage_id: int
|
||||
session_id: str | None
|
||||
remote: str
|
||||
org: str
|
||||
repo: str
|
||||
project_id: str | None
|
||||
role: str
|
||||
model: str
|
||||
issue_number: int | None
|
||||
pr_number: int | None
|
||||
stage: str
|
||||
input_tokens: int | None
|
||||
output_tokens: int | None
|
||||
total_tokens: int | None
|
||||
estimated_cost_usd: float | None
|
||||
latency_ms: int | None
|
||||
duration_ms: int | None
|
||||
status: str
|
||||
metadata: str | None
|
||||
created_at: str
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
d = asdict(self)
|
||||
if d["metadata"]:
|
||||
d["metadata"] = console_redaction.redact_text(d["metadata"])
|
||||
return d
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class GroupMetrics:
|
||||
name: str
|
||||
total_events: int
|
||||
events_with_tokens: int
|
||||
input_tokens: int | None
|
||||
output_tokens: int | None
|
||||
total_tokens: int | None
|
||||
events_with_cost: int
|
||||
estimated_cost_usd: float | None
|
||||
events_with_latency: int
|
||||
latency_p50_ms: float | None
|
||||
latency_p90_ms: float | None
|
||||
latency_p95_ms: float | None
|
||||
latency_p99_ms: float | None
|
||||
latency_avg_ms: float | None
|
||||
events_with_duration: int
|
||||
duration_avg_ms: float | None
|
||||
display_tokens: str
|
||||
display_cost: str
|
||||
display_latency_p50: str
|
||||
display_latency_p90: str
|
||||
display_duration_avg: str
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AnalyticsSnapshot:
|
||||
ok: bool
|
||||
reason: str
|
||||
schema_version: int
|
||||
remote: str
|
||||
org: str
|
||||
repo: str
|
||||
total_events: int
|
||||
overall_summary: GroupMetrics
|
||||
by_project: dict[str, GroupMetrics]
|
||||
by_role: dict[str, GroupMetrics]
|
||||
by_model: dict[str, GroupMetrics]
|
||||
by_work_item: dict[str, GroupMetrics]
|
||||
by_stage: dict[str, GroupMetrics]
|
||||
events: tuple[UsageEvent, ...]
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"ok": self.ok,
|
||||
"reason": self.reason,
|
||||
"schema_version": self.schema_version,
|
||||
"remote": self.remote,
|
||||
"org": self.org,
|
||||
"repo": self.repo,
|
||||
"total_events": self.total_events,
|
||||
"overall_summary": self.overall_summary.to_dict(),
|
||||
"by_project": {k: v.to_dict() for k, v in self.by_project.items()},
|
||||
"by_role": {k: v.to_dict() for k, v in self.by_role.items()},
|
||||
"by_model": {k: v.to_dict() for k, v in self.by_model.items()},
|
||||
"by_work_item": {k: v.to_dict() for k, v in self.by_work_item.items()},
|
||||
"by_stage": {k: v.to_dict() for k, v in self.by_stage.items()},
|
||||
"events": [e.to_dict() for e in self.events],
|
||||
}
|
||||
|
||||
|
||||
def compute_percentile(values: Sequence[float | int], percentile: float) -> float | None:
|
||||
if not values:
|
||||
return None
|
||||
sorted_vals = sorted(values)
|
||||
n = len(sorted_vals)
|
||||
if n == 1:
|
||||
return float(sorted_vals[0])
|
||||
k = (n - 1) * (percentile / 100.0)
|
||||
f = math.floor(k)
|
||||
c = math.ceil(k)
|
||||
if f == c:
|
||||
return float(sorted_vals[int(f)])
|
||||
d0 = sorted_vals[int(f)] * (c - k)
|
||||
d1 = sorted_vals[int(c)] * (k - f)
|
||||
return float(d0 + d1)
|
||||
|
||||
|
||||
def aggregate_events(group_name: str, events: Sequence[UsageEvent]) -> GroupMetrics:
|
||||
total_events = len(events)
|
||||
if total_events == 0:
|
||||
return GroupMetrics(
|
||||
name=group_name,
|
||||
total_events=0,
|
||||
events_with_tokens=0,
|
||||
input_tokens=None,
|
||||
output_tokens=None,
|
||||
total_tokens=None,
|
||||
events_with_cost=0,
|
||||
estimated_cost_usd=None,
|
||||
events_with_latency=0,
|
||||
latency_p50_ms=None,
|
||||
latency_p90_ms=None,
|
||||
latency_p95_ms=None,
|
||||
latency_p99_ms=None,
|
||||
latency_avg_ms=None,
|
||||
events_with_duration=0,
|
||||
duration_avg_ms=None,
|
||||
display_tokens="Unknown",
|
||||
display_cost="Unknown",
|
||||
display_latency_p50="Unknown",
|
||||
display_latency_p90="Unknown",
|
||||
display_duration_avg="Unknown",
|
||||
)
|
||||
|
||||
token_events = [
|
||||
e for e in events
|
||||
if e.total_tokens is not None or e.input_tokens is not None or e.output_tokens is not None
|
||||
]
|
||||
events_with_tokens = len(token_events)
|
||||
if events_with_tokens > 0:
|
||||
input_tokens = sum(e.input_tokens or 0 for e in token_events)
|
||||
output_tokens = sum(e.output_tokens or 0 for e in token_events)
|
||||
total_tokens = sum(
|
||||
e.total_tokens if e.total_tokens is not None else ((e.input_tokens or 0) + (e.output_tokens or 0))
|
||||
for e in token_events
|
||||
)
|
||||
display_tokens = f"{total_tokens:,}"
|
||||
else:
|
||||
input_tokens = None
|
||||
output_tokens = None
|
||||
total_tokens = None
|
||||
display_tokens = "Unknown"
|
||||
|
||||
cost_events = [e for e in events if e.estimated_cost_usd is not None]
|
||||
events_with_cost = len(cost_events)
|
||||
if events_with_cost > 0:
|
||||
estimated_cost_usd = round(sum(e.estimated_cost_usd for e in cost_events), 6)
|
||||
display_cost = f"${estimated_cost_usd:.4f}"
|
||||
else:
|
||||
estimated_cost_usd = None
|
||||
display_cost = "Unknown"
|
||||
|
||||
latency_vals = [e.latency_ms for e in events if e.latency_ms is not None]
|
||||
events_with_latency = len(latency_vals)
|
||||
if events_with_latency > 0:
|
||||
latency_p50_ms = compute_percentile(latency_vals, 50.0)
|
||||
latency_p90_ms = compute_percentile(latency_vals, 90.0)
|
||||
latency_p95_ms = compute_percentile(latency_vals, 95.0)
|
||||
latency_p99_ms = compute_percentile(latency_vals, 99.0)
|
||||
latency_avg_ms = round(sum(latency_vals) / events_with_latency, 2)
|
||||
display_latency_p50 = f"{round(latency_p50_ms, 1)} ms" if latency_p50_ms is not None else "Unknown"
|
||||
display_latency_p90 = f"{round(latency_p90_ms, 1)} ms" if latency_p90_ms is not None else "Unknown"
|
||||
else:
|
||||
latency_p50_ms = None
|
||||
latency_p90_ms = None
|
||||
latency_p95_ms = None
|
||||
latency_p99_ms = None
|
||||
latency_avg_ms = None
|
||||
display_latency_p50 = "Unknown"
|
||||
display_latency_p90 = "Unknown"
|
||||
|
||||
duration_vals = [e.duration_ms for e in events if e.duration_ms is not None]
|
||||
events_with_duration = len(duration_vals)
|
||||
if events_with_duration > 0:
|
||||
duration_avg_ms = round(sum(duration_vals) / events_with_duration, 2)
|
||||
display_duration_avg = f"{round(duration_avg_ms / 1000.0, 2)} s" if duration_avg_ms >= 1000 else f"{round(duration_avg_ms, 1)} ms"
|
||||
else:
|
||||
duration_avg_ms = None
|
||||
display_duration_avg = "Unknown"
|
||||
|
||||
return GroupMetrics(
|
||||
name=group_name,
|
||||
total_events=total_events,
|
||||
events_with_tokens=events_with_tokens,
|
||||
input_tokens=input_tokens,
|
||||
output_tokens=output_tokens,
|
||||
total_tokens=total_tokens,
|
||||
events_with_cost=events_with_cost,
|
||||
estimated_cost_usd=estimated_cost_usd,
|
||||
events_with_latency=events_with_latency,
|
||||
latency_p50_ms=latency_p50_ms,
|
||||
latency_p90_ms=latency_p90_ms,
|
||||
latency_p95_ms=latency_p95_ms,
|
||||
latency_p99_ms=latency_p99_ms,
|
||||
latency_avg_ms=latency_avg_ms,
|
||||
events_with_duration=events_with_duration,
|
||||
duration_avg_ms=duration_avg_ms,
|
||||
display_tokens=display_tokens,
|
||||
display_cost=display_cost,
|
||||
display_latency_p50=display_latency_p50,
|
||||
display_latency_p90=display_latency_p90,
|
||||
display_duration_avg=display_duration_avg,
|
||||
)
|
||||
|
||||
|
||||
def record_usage(
|
||||
*,
|
||||
db_path: str | None = None,
|
||||
session_id: str | None = None,
|
||||
remote: str = "dadeschools",
|
||||
org: str = "",
|
||||
repo: str = "",
|
||||
project_id: str | None = None,
|
||||
role: str = "unknown",
|
||||
model: str = "unknown",
|
||||
issue_number: int | None = None,
|
||||
pr_number: int | None = None,
|
||||
stage: str = "unknown",
|
||||
input_tokens: int | None = None,
|
||||
output_tokens: int | None = None,
|
||||
total_tokens: int | None = None,
|
||||
estimated_cost_usd: float | None = None,
|
||||
latency_ms: int | None = None,
|
||||
duration_ms: int | None = None,
|
||||
status: str = "success",
|
||||
metadata: str | dict[str, Any] | None = None,
|
||||
created_at: str | None = None,
|
||||
) -> int:
|
||||
"""Ingest/record a single usage event with optional metrics."""
|
||||
db = control_plane_db.ControlPlaneDB(db_path=db_path)
|
||||
return db.record_usage_event(
|
||||
session_id=session_id,
|
||||
remote=remote,
|
||||
org=org,
|
||||
repo=repo,
|
||||
project_id=project_id,
|
||||
role=role,
|
||||
model=model,
|
||||
issue_number=issue_number,
|
||||
pr_number=pr_number,
|
||||
stage=stage,
|
||||
input_tokens=input_tokens,
|
||||
output_tokens=output_tokens,
|
||||
total_tokens=total_tokens,
|
||||
estimated_cost_usd=estimated_cost_usd,
|
||||
latency_ms=latency_ms,
|
||||
duration_ms=duration_ms,
|
||||
status=status,
|
||||
metadata=metadata,
|
||||
created_at=created_at,
|
||||
)
|
||||
|
||||
|
||||
def load_analytics(
|
||||
*,
|
||||
db_path: str | None = None,
|
||||
remote: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
project_id: str | None = None,
|
||||
role: str | None = None,
|
||||
model: str | None = None,
|
||||
stage: str | None = None,
|
||||
issue_number: int | None = None,
|
||||
pr_number: int | None = None,
|
||||
limit: int = 500,
|
||||
) -> AnalyticsSnapshot:
|
||||
"""Load analytics snapshot aggregated by project, role, model, issue/PR, and stage."""
|
||||
remote_filter = (remote or "").strip() or None
|
||||
org_filter = (org or "").strip() or None
|
||||
repo_filter = (repo or "").strip() or None
|
||||
role_filter = (role or "").strip() or None
|
||||
model_filter = (model or "").strip() or None
|
||||
stage_filter = (stage or "").strip() or None
|
||||
|
||||
# F4: coerce optional scope filters to str so AnalyticsSnapshot never holds None.
|
||||
scope_remote = (remote or "").strip()
|
||||
scope_org = (org or "").strip()
|
||||
scope_repo = (repo or "").strip()
|
||||
|
||||
try:
|
||||
db = control_plane_db.ControlPlaneDB(db_path=db_path)
|
||||
rows = db.query_usage_events(
|
||||
remote=remote_filter,
|
||||
org=org_filter,
|
||||
repo=repo_filter,
|
||||
project_id=project_id,
|
||||
role=role_filter,
|
||||
model=model_filter,
|
||||
stage=stage_filter,
|
||||
issue_number=issue_number,
|
||||
pr_number=pr_number,
|
||||
limit=limit,
|
||||
)
|
||||
except Exception as exc:
|
||||
empty_summary = aggregate_events("Overall", [])
|
||||
return AnalyticsSnapshot(
|
||||
ok=False,
|
||||
reason=f"control_plane_db_unavailable: {exc}",
|
||||
schema_version=ANALYTICS_SCHEMA_VERSION,
|
||||
remote=scope_remote,
|
||||
org=scope_org,
|
||||
repo=scope_repo,
|
||||
total_events=0,
|
||||
overall_summary=empty_summary,
|
||||
by_project={},
|
||||
by_role={},
|
||||
by_model={},
|
||||
by_work_item={},
|
||||
by_stage={},
|
||||
events=(),
|
||||
)
|
||||
|
||||
parsed_events: list[UsageEvent] = []
|
||||
for r in rows:
|
||||
meta = console_redaction.redact_text(r.get("metadata")) if r.get("metadata") else None
|
||||
parsed_events.append(
|
||||
UsageEvent(
|
||||
usage_id=r["usage_id"],
|
||||
session_id=r.get("session_id"),
|
||||
remote=r.get("remote") or scope_remote,
|
||||
org=r.get("org") or scope_org,
|
||||
repo=r.get("repo") or scope_repo,
|
||||
project_id=r.get("project_id"),
|
||||
role=r.get("role") or "unknown",
|
||||
model=r.get("model") or "unknown",
|
||||
issue_number=r.get("issue_number"),
|
||||
pr_number=r.get("pr_number"),
|
||||
stage=r.get("stage") or "unknown",
|
||||
input_tokens=r.get("input_tokens"),
|
||||
output_tokens=r.get("output_tokens"),
|
||||
total_tokens=r.get("total_tokens"),
|
||||
estimated_cost_usd=r.get("estimated_cost_usd"),
|
||||
latency_ms=r.get("latency_ms"),
|
||||
duration_ms=r.get("duration_ms"),
|
||||
status=r.get("status") or "success",
|
||||
metadata=meta,
|
||||
created_at=r.get("created_at") or "",
|
||||
)
|
||||
)
|
||||
|
||||
overall_summary = aggregate_events("Overall", parsed_events)
|
||||
|
||||
# Group by project
|
||||
groups_by_project: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
key = e.project_id or (f"{e.org}/{e.repo}" if e.org and e.repo else "default")
|
||||
groups_by_project.setdefault(key, []).append(e)
|
||||
by_project = {k: aggregate_events(k, v) for k, v in groups_by_project.items()}
|
||||
|
||||
# Group by role
|
||||
groups_by_role: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
groups_by_role.setdefault(e.role, []).append(e)
|
||||
by_role = {k: aggregate_events(k, v) for k, v in groups_by_role.items()}
|
||||
|
||||
# Group by model
|
||||
groups_by_model: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
groups_by_model.setdefault(e.model, []).append(e)
|
||||
by_model = {k: aggregate_events(k, v) for k, v in groups_by_model.items()}
|
||||
|
||||
# Group by work item
|
||||
groups_by_work_item: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
if e.issue_number:
|
||||
key = f"issue #{e.issue_number}"
|
||||
elif e.pr_number:
|
||||
key = f"pr #{e.pr_number}"
|
||||
else:
|
||||
key = "unlinked"
|
||||
groups_by_work_item.setdefault(key, []).append(e)
|
||||
by_work_item = {k: aggregate_events(k, v) for k, v in groups_by_work_item.items()}
|
||||
|
||||
# Group by stage
|
||||
groups_by_stage: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
groups_by_stage.setdefault(e.stage, []).append(e)
|
||||
by_stage = {k: aggregate_events(k, v) for k, v in groups_by_stage.items()}
|
||||
|
||||
return AnalyticsSnapshot(
|
||||
ok=True,
|
||||
reason="ok",
|
||||
schema_version=ANALYTICS_SCHEMA_VERSION,
|
||||
remote=scope_remote,
|
||||
org=scope_org,
|
||||
repo=scope_repo,
|
||||
total_events=len(parsed_events),
|
||||
overall_summary=overall_summary,
|
||||
by_project=by_project,
|
||||
by_role=by_role,
|
||||
by_model=by_model,
|
||||
by_work_item=by_work_item,
|
||||
by_stage=by_stage,
|
||||
events=tuple(parsed_events),
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: AnalyticsSnapshot) -> dict[str, Any]:
|
||||
return snapshot.to_dict()
|
||||
@@ -0,0 +1,248 @@
|
||||
"""HTML views for the Model Usage & Performance Analytics console (#651)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
|
||||
from webui.analytics_loader import AnalyticsSnapshot, GroupMetrics, UsageEvent
|
||||
from webui.layout import render_page
|
||||
|
||||
|
||||
def _escape(text: object) -> str:
|
||||
"""HTML-escape dynamic analytics fields (mirrors audit_views / project_views)."""
|
||||
return html.escape(str(text), quote=True)
|
||||
|
||||
|
||||
def _render_badge(text: str, badge_type: str = "muted") -> str:
|
||||
return f'<span class="badge badge-{_escape(badge_type)}">{_escape(text)}</span>'
|
||||
|
||||
|
||||
def _render_group_table(title: str, groups: dict[str, GroupMetrics], key_header: str = "Group") -> str:
|
||||
if not groups:
|
||||
return (
|
||||
f"<h3>{_escape(title)}</h3>"
|
||||
'<div class="card"><p class="muted">No telemetry events recorded for this dimension.</p></div>'
|
||||
)
|
||||
|
||||
rows = []
|
||||
for key, g in sorted(groups.items(), key=lambda x: x[1].total_events, reverse=True):
|
||||
cost_cell = (
|
||||
f'<span class="accent">{_escape(g.display_cost)}</span>'
|
||||
if g.events_with_cost > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
tokens_cell = (
|
||||
_escape(g.display_tokens)
|
||||
if g.events_with_tokens > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
lat_p50 = (
|
||||
_escape(g.display_latency_p50)
|
||||
if g.events_with_latency > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
lat_p90 = (
|
||||
_escape(g.display_latency_p90)
|
||||
if g.events_with_latency > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
dur_avg = (
|
||||
_escape(g.display_duration_avg)
|
||||
if g.events_with_duration > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td><strong>{_escape(key)}</strong></td>"
|
||||
f"<td>{g.total_events}</td>"
|
||||
f"<td>{tokens_cell}</td>"
|
||||
f"<td>{cost_cell}</td>"
|
||||
f"<td>{lat_p50}</td>"
|
||||
f"<td>{lat_p90}</td>"
|
||||
f"<td>{dur_avg}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
|
||||
rows_html = "".join(rows)
|
||||
return f"""
|
||||
<h3>{_escape(title)}</h3>
|
||||
<div class="card" style="overflow-x: auto;">
|
||||
<table class="data-table">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>{_escape(key_header)}</th>
|
||||
<th>Events</th>
|
||||
<th>Total Tokens</th>
|
||||
<th>Est. Cost</th>
|
||||
<th>Latency (p50)</th>
|
||||
<th>Latency (p90)</th>
|
||||
<th>Avg Stage Duration</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{rows_html}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
"""
|
||||
|
||||
|
||||
def _render_events_table(events: tuple[UsageEvent, ...]) -> str:
|
||||
if not events:
|
||||
return (
|
||||
"<h3>Recent Usage & Instrumentation Events</h3>"
|
||||
'<div class="card"><p class="muted">No individual telemetry events recorded yet. Opt-in instrumentation via session logging or authorized POST /api/v1/analytics/usage.</p></div>'
|
||||
)
|
||||
|
||||
rows = []
|
||||
for e in list(events)[-50:]: # Display latest 50
|
||||
if e.issue_number is not None:
|
||||
work_item = f"issue #{e.issue_number}"
|
||||
elif e.pr_number is not None:
|
||||
work_item = f"pr #{e.pr_number}"
|
||||
else:
|
||||
work_item = "unlinked"
|
||||
tokens = (
|
||||
_escape(f"{e.total_tokens:,}")
|
||||
if e.total_tokens is not None
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
cost = (
|
||||
_escape(f"${e.estimated_cost_usd:.4f}")
|
||||
if e.estimated_cost_usd is not None
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
latency = (
|
||||
_escape(f"{e.latency_ms} ms")
|
||||
if e.latency_ms is not None
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
duration = (
|
||||
_escape(f"{e.duration_ms} ms")
|
||||
if e.duration_ms is not None
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
status_badge = _render_badge(
|
||||
e.status, "success" if e.status == "success" else "danger"
|
||||
)
|
||||
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td>#{e.usage_id}</td>"
|
||||
f"<td><small>{_escape(e.created_at)}</small></td>"
|
||||
f"<td><span class=\"badge\">{_escape(e.role)}</span></td>"
|
||||
f"<td><strong>{_escape(e.model)}</strong></td>"
|
||||
f"<td>{_escape(e.stage)}</td>"
|
||||
f"<td>{_escape(work_item)}</td>"
|
||||
f"<td>{tokens}</td>"
|
||||
f"<td>{cost}</td>"
|
||||
f"<td>{latency}</td>"
|
||||
f"<td>{duration}</td>"
|
||||
f"<td>{status_badge}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
|
||||
rows_html = "".join(rows)
|
||||
return f"""
|
||||
<h3>Recent Telemetry Events</h3>
|
||||
<div class="card" style="overflow-x: auto;">
|
||||
<table class="data-table">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>ID</th>
|
||||
<th>Timestamp</th>
|
||||
<th>Role</th>
|
||||
<th>Model</th>
|
||||
<th>Stage</th>
|
||||
<th>Work Item</th>
|
||||
<th>Tokens</th>
|
||||
<th>Cost</th>
|
||||
<th>Latency</th>
|
||||
<th>Duration</th>
|
||||
<th>Status</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{rows_html}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
"""
|
||||
|
||||
|
||||
def render_analytics_page(snapshot: AnalyticsSnapshot) -> str:
|
||||
"""Render the main Model Usage & Performance Analytics console page."""
|
||||
summary = snapshot.overall_summary
|
||||
|
||||
kpi_tokens = (
|
||||
_escape(summary.display_tokens)
|
||||
if summary.events_with_tokens > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
kpi_cost = (
|
||||
_escape(summary.display_cost)
|
||||
if summary.events_with_cost > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
kpi_lat_p50 = (
|
||||
_escape(summary.display_latency_p50)
|
||||
if summary.events_with_latency > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
kpi_dur_avg = (
|
||||
_escape(summary.display_duration_avg)
|
||||
if summary.events_with_duration > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
|
||||
status_notice = ""
|
||||
if not snapshot.ok:
|
||||
status_notice = (
|
||||
f'<div class="card warning-card"><strong>Degraded Data Source:</strong> '
|
||||
f'{_escape(snapshot.reason)}</div>'
|
||||
)
|
||||
|
||||
body_html = f"""
|
||||
<h2>Model Usage & Performance Analytics (Phase 4)</h2>
|
||||
<p class="muted">
|
||||
Durable console analytics for model usage, token cost, latency percentiles, and workflow-stage performance correlated to issues, PRs, and worker roles.
|
||||
</p>
|
||||
|
||||
{status_notice}
|
||||
|
||||
<div class="notice-card" style="background: rgba(91, 159, 212, 0.1); border: 1px solid var(--border); padding: 0.75rem 1rem; border-radius: 6px; margin-bottom: 1.5rem;">
|
||||
<small><strong>Note on telemetry fidelity:</strong> Missing data or untracked metrics are explicitly labeled as <em>Unknown</em>. No token costs or latency metrics are zero-fabricated.</small>
|
||||
</div>
|
||||
|
||||
<div class="card-grid" style="display: grid; grid-template-columns: repeat(auto-fit, minmax(180px, 1fr)); gap: 1rem; margin-bottom: 1.5rem;">
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Total Events</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{summary.total_events}</h3>
|
||||
</div>
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Total Tokens</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{kpi_tokens}</h3>
|
||||
</div>
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Est. Token Cost</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{kpi_cost}</h3>
|
||||
</div>
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Latency (p50)</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{kpi_lat_p50}</h3>
|
||||
</div>
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Avg Stage Duration</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{kpi_dur_avg}</h3>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{_render_group_table("Usage & Cost by Model", snapshot.by_model, "Model")}
|
||||
{_render_group_table("Performance by Workflow Stage", snapshot.by_stage, "Stage")}
|
||||
{_render_group_table("Usage & Cost by Role", snapshot.by_role, "Role")}
|
||||
{_render_group_table("Work Item Analytics", snapshot.by_work_item, "Work Item")}
|
||||
{_render_events_table(snapshot.events)}
|
||||
"""
|
||||
|
||||
return render_page(title="Model Usage & Performance Analytics", body_html=body_html)
|
||||
+118
-35
@@ -46,12 +46,13 @@ from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as wo
|
||||
from webui.worktree_views import render_worktrees_page
|
||||
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
|
||||
from webui.runtime_views import render_runtime_page
|
||||
from webui.inventory import (
|
||||
SECTION_NAMES as _INVENTORY_SECTIONS,
|
||||
load_inventory_snapshot,
|
||||
snapshot_to_dict as inventory_snapshot_to_dict,
|
||||
)
|
||||
from webui.timeline import load_timeline, snapshot_to_dict as timeline_snapshot_to_dict
|
||||
from webui.analytics_loader import (
|
||||
load_analytics,
|
||||
record_usage,
|
||||
snapshot_to_dict as analytics_snapshot_to_dict,
|
||||
)
|
||||
from webui.analytics_views import render_analytics_page
|
||||
from webui.system_health import (
|
||||
API_PATH as SYSTEM_HEALTH_API_PATH,
|
||||
load_system_health,
|
||||
@@ -465,30 +466,6 @@ async def api_action_attempt(request: Request) -> JSONResponse:
|
||||
return JSONResponse(result, status_code=status)
|
||||
|
||||
|
||||
async def api_inventory(_request: Request) -> JSONResponse:
|
||||
"""Unified read-only session/lease/lock/worktree inventory (#636)."""
|
||||
snapshot = load_inventory_snapshot()
|
||||
return JSONResponse(inventory_snapshot_to_dict(snapshot))
|
||||
|
||||
|
||||
async def api_inventory_section(request: Request) -> JSONResponse:
|
||||
"""Resource-split view: one inventory section under the shared schema."""
|
||||
section = request.path_params["section"]
|
||||
if section not in _INVENTORY_SECTIONS:
|
||||
return JSONResponse(
|
||||
{
|
||||
"error": "unknown_section",
|
||||
"detail": f"no inventory section named {section!r}",
|
||||
"available": sorted(_INVENTORY_SECTIONS),
|
||||
},
|
||||
status_code=404,
|
||||
)
|
||||
snapshot = load_inventory_snapshot(include=(section,))
|
||||
payload = inventory_snapshot_to_dict(snapshot)
|
||||
payload["requested_section"] = section
|
||||
return JSONResponse(payload)
|
||||
|
||||
|
||||
async def api_console_security_model(_request: Request) -> JSONResponse:
|
||||
"""Read-only publication of the #633 authorization/redaction/audit model."""
|
||||
return JSONResponse({
|
||||
@@ -596,6 +573,114 @@ async def api_v1_timeline(request: Request) -> JSONResponse:
|
||||
return JSONResponse(timeline_snapshot_to_dict(snapshot), status_code=status_code)
|
||||
|
||||
|
||||
async def analytics(request: Request) -> HTMLResponse:
|
||||
"""Read-only model usage, token cost, latency, and performance analytics HTML view (#651)."""
|
||||
snapshot = load_analytics(
|
||||
remote=request.query_params.get("remote"),
|
||||
org=request.query_params.get("org"),
|
||||
repo=request.query_params.get("repo"),
|
||||
role=request.query_params.get("role"),
|
||||
model=request.query_params.get("model"),
|
||||
stage=request.query_params.get("stage"),
|
||||
issue_number=_query_int(request, "issue"),
|
||||
pr_number=_query_int(request, "pr"),
|
||||
limit=_query_int(request, "limit") or 200,
|
||||
)
|
||||
return HTMLResponse(render_analytics_page(snapshot))
|
||||
|
||||
|
||||
async def api_v1_analytics(request: Request) -> JSONResponse:
|
||||
"""Read-only model usage, token cost, latency, and performance analytics API (#651)."""
|
||||
snapshot = load_analytics(
|
||||
remote=request.query_params.get("remote"),
|
||||
org=request.query_params.get("org"),
|
||||
repo=request.query_params.get("repo"),
|
||||
role=request.query_params.get("role"),
|
||||
model=request.query_params.get("model"),
|
||||
stage=request.query_params.get("stage"),
|
||||
issue_number=_query_int(request, "issue"),
|
||||
pr_number=_query_int(request, "pr"),
|
||||
limit=_query_int(request, "limit") or 500,
|
||||
)
|
||||
status_code = 200 if snapshot.ok else 500
|
||||
return JSONResponse(analytics_snapshot_to_dict(snapshot), status_code=status_code)
|
||||
|
||||
|
||||
async def api_v1_analytics_ingest(request: Request) -> JSONResponse:
|
||||
"""Optional session instrumentation ingestion endpoint (#651).
|
||||
|
||||
Fail-closed write: every request is authorized through console_authz
|
||||
(``record_analytics_usage``) before any control-plane DB mutation. Phase 1
|
||||
keeps ``execution_enabled=False`` and denies unauthenticated callers, so
|
||||
this route cannot be used as an unauthenticated write or XSS injection
|
||||
vector (PR #876 F2).
|
||||
"""
|
||||
try:
|
||||
body = await request.json()
|
||||
except Exception:
|
||||
body = {}
|
||||
if not isinstance(body, dict):
|
||||
body = {}
|
||||
|
||||
principal = resolve_principal(headers=dict(request.headers))
|
||||
decision = authorize(
|
||||
"record_analytics_usage", principal, for_execution=True
|
||||
)
|
||||
allowed = bool(decision.allowed and decision.execution_enabled)
|
||||
console_audit.record_event(
|
||||
action_id="record_analytics_usage",
|
||||
result=(
|
||||
console_audit.RESULT_ALLOWED
|
||||
if allowed
|
||||
else console_audit.RESULT_DENIED
|
||||
),
|
||||
decision=decision,
|
||||
principal=principal,
|
||||
target=_audit_target("record_analytics_usage", body),
|
||||
request_id=_request_id(),
|
||||
detail=decision.detail,
|
||||
)
|
||||
authorization = decision.to_dict()
|
||||
if not allowed:
|
||||
return JSONResponse(
|
||||
{
|
||||
"ok": False,
|
||||
"error": "unauthorized",
|
||||
"detail": (
|
||||
"POST /api/v1/analytics/usage requires an authenticated "
|
||||
"principal with record_analytics_usage execution enabled"
|
||||
),
|
||||
"authorization": authorization,
|
||||
},
|
||||
status_code=403,
|
||||
)
|
||||
|
||||
usage_id = record_usage(
|
||||
session_id=body.get("session_id"),
|
||||
remote=body.get("remote", "dadeschools"),
|
||||
org=body.get("org", ""),
|
||||
repo=body.get("repo", ""),
|
||||
project_id=body.get("project_id"),
|
||||
role=body.get("role", "unknown"),
|
||||
model=body.get("model", "unknown"),
|
||||
issue_number=body.get("issue_number") or body.get("issue"),
|
||||
pr_number=body.get("pr_number") or body.get("pr"),
|
||||
stage=body.get("stage", "unknown"),
|
||||
input_tokens=body.get("input_tokens"),
|
||||
output_tokens=body.get("output_tokens"),
|
||||
total_tokens=body.get("total_tokens"),
|
||||
estimated_cost_usd=body.get("estimated_cost_usd"),
|
||||
latency_ms=body.get("latency_ms"),
|
||||
duration_ms=body.get("duration_ms"),
|
||||
status=body.get("status", "success"),
|
||||
metadata=body.get("metadata"),
|
||||
)
|
||||
return JSONResponse(
|
||||
{"ok": True, "usage_id": usage_id, "authorization": authorization},
|
||||
status_code=201,
|
||||
)
|
||||
|
||||
|
||||
async def method_not_allowed(request: Request, _exc: Exception) -> Response:
|
||||
path = request.url.path
|
||||
if path in _AUDIT_MUTATION_PATHS and request.method == "POST":
|
||||
@@ -637,6 +722,10 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/runtime", runtime, methods=["GET"]),
|
||||
Route("/api/runtime", api_runtime, methods=["GET"]),
|
||||
Route("/api/v1/timeline", api_v1_timeline, methods=["GET"]),
|
||||
Route("/analytics", analytics, methods=["GET"]),
|
||||
Route("/api/analytics", api_v1_analytics, methods=["GET"]),
|
||||
Route("/api/v1/analytics", api_v1_analytics, methods=["GET"]),
|
||||
Route("/api/v1/analytics/usage", api_v1_analytics_ingest, methods=["POST"]),
|
||||
Route("/audit", audit, methods=["GET", "POST"]),
|
||||
Route("/api/audit", api_audit, methods=["GET", "POST"]),
|
||||
Route("/worktrees", worktrees, methods=["GET"]),
|
||||
@@ -655,12 +744,6 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
methods=["POST"],
|
||||
),
|
||||
Route("/api/leases", api_leases, methods=["GET"]),
|
||||
Route("/api/v1/inventory", api_inventory, methods=["GET"]),
|
||||
Route(
|
||||
"/api/v1/inventory/{section}",
|
||||
api_inventory_section,
|
||||
methods=["GET"],
|
||||
),
|
||||
Route(
|
||||
"/api/console/security-model",
|
||||
api_console_security_model,
|
||||
|
||||
@@ -236,6 +236,19 @@ _ACTION_SPECS: tuple[ConsoleAction, ...] = (
|
||||
phase=3,
|
||||
summary="Remove a remote feature branch.",
|
||||
),
|
||||
# #651 analytics ingest: local control-plane write, not a Gitea mutation.
|
||||
# Phase 2 gated write so Phase 1 (ACTIVE_PHASE=1) fails closed on execution.
|
||||
ConsoleAction(
|
||||
action_id="record_analytics_usage",
|
||||
task_key="record_analytics_usage",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Ingest a model-usage / latency analytics event into the control-plane DB.",
|
||||
),
|
||||
# #642: sanctioned daemon lifecycle. These exist so operators have an
|
||||
# audited path off `pkill -f mcp_server.py` (#630). Restart drops every
|
||||
# in-flight request on a namespace, so it carries the same dual-control and
|
||||
|
||||
@@ -1,952 +0,0 @@
|
||||
"""Unified session/lease/lock/worktree inventory for the web console (#636).
|
||||
|
||||
Leases (#433), worktrees (#432), and runtime (#430) each ship their own MVP
|
||||
view, each with its own shape and its own idea of what "owned" means. A
|
||||
traffic-control or recovery operator has to read all three and correlate them
|
||||
by hand, which is exactly the step that goes wrong under collision pressure.
|
||||
|
||||
This module aggregates them into one versioned, read-only snapshot so the
|
||||
console, and any worker asking "what is safe to do next", read the same
|
||||
inventory from the same authority.
|
||||
|
||||
Field authority is explicit and never blended. Every section declares where its
|
||||
rows came from:
|
||||
|
||||
* ``control_plane_db`` — the #613 substrate: sessions, leases, assignments.
|
||||
Authoritative for *exclusive ownership* (#600/#601).
|
||||
* ``filesystem`` — durable per-issue lock files (:mod:`issue_lock_store`) and
|
||||
registered git worktrees. Authoritative for *what exists on this machine*.
|
||||
* ``gitea`` — remote issue/PR state, reached only through existing loaders.
|
||||
|
||||
Safety invariants:
|
||||
|
||||
* **Read-only.** The control-plane database is opened through a ``mode=ro``
|
||||
URI. :class:`control_plane_db.ControlPlaneDB` creates directories and runs
|
||||
migrations in its constructor, which an inventory read must never do, so this
|
||||
module talks to sqlite directly rather than through that class.
|
||||
* **Fail-soft, never fail-silent.** A subsystem that cannot be read degrades to
|
||||
a section carrying ``status`` and ``reason``. It never raises, and it never
|
||||
produces an empty list that reads like "nothing is there".
|
||||
* **Never invent active ownership.** This is the invariant that matters most.
|
||||
A degraded ownership source sets ``ownership_authority_complete`` false, and
|
||||
while that flag is false no work item is reported unowned and no collision is
|
||||
asserted. Absence of evidence is reported as absence of evidence.
|
||||
* **Redaction at the boundary.** Absolute paths are collapsed against the home
|
||||
directory, URLs lose userinfo and query strings, and no credential-shaped
|
||||
value is emitted. No session token exists in these sources and none is read.
|
||||
|
||||
Phase 1 is read-only. Lease steal/release and worktree deletion are Phase 2+
|
||||
and deliberately have no representation here, not even a disabled one.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import sqlite3
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import control_plane_db
|
||||
import issue_lock_store
|
||||
|
||||
SCHEMA_VERSION = 1
|
||||
API_VERSION = "v1"
|
||||
|
||||
#: Sections whose absence would make an ownership claim unprovable. If any of
|
||||
#: these is not ``ok``, the snapshot refuses to describe anything as unowned.
|
||||
OWNERSHIP_SECTIONS = ("sessions", "leases", "locks")
|
||||
|
||||
SECTION_NAMES = ("sessions", "leases", "locks", "worktrees", "namespaces")
|
||||
|
||||
STATUS_OK = "ok"
|
||||
STATUS_DEGRADED = "degraded"
|
||||
STATUS_UNAVAILABLE = "unavailable"
|
||||
|
||||
AUTHORITY_CONTROL_PLANE_DB = "control_plane_db"
|
||||
AUTHORITY_FILESYSTEM = "filesystem"
|
||||
AUTHORITY_GITEA = "gitea"
|
||||
|
||||
_CREDENTIAL_KEY_RE = re.compile(
|
||||
r"(token|secret|password|passwd|api[_-]?key|authorization|bearer|credential)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_REDACTED = "[redacted]"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class InventorySection:
|
||||
"""One subsystem's contribution, with its authority and health."""
|
||||
|
||||
name: str
|
||||
authority: str
|
||||
status: str
|
||||
items: tuple[dict[str, Any], ...] = ()
|
||||
reason: str | None = None
|
||||
scan_ms: float | None = None
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
return self.status == STATUS_OK
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"name": self.name,
|
||||
"authority": self.authority,
|
||||
"status": self.status,
|
||||
"count": len(self.items),
|
||||
"reason": self.reason,
|
||||
"scan_ms": self.scan_ms,
|
||||
"items": [dict(item) for item in self.items],
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CollisionSignal:
|
||||
"""A detected conflict between two ownership records."""
|
||||
|
||||
kind: str
|
||||
message: str
|
||||
severity: str = "warning"
|
||||
issue_number: int | None = None
|
||||
branch: str | None = None
|
||||
worktree_path: str | None = None
|
||||
session_ids: tuple[str, ...] = ()
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"kind": self.kind,
|
||||
"severity": self.severity,
|
||||
"message": self.message,
|
||||
"issue_number": self.issue_number,
|
||||
"branch": self.branch,
|
||||
"worktree_path": self.worktree_path,
|
||||
"session_ids": list(self.session_ids),
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class InventorySnapshot:
|
||||
"""Versioned aggregate of every inventory section."""
|
||||
|
||||
generated_at: str
|
||||
sections: tuple[InventorySection, ...]
|
||||
collisions: tuple[CollisionSignal, ...] = ()
|
||||
correlations: tuple[dict[str, Any], ...] = ()
|
||||
schema_version: int = SCHEMA_VERSION
|
||||
api_version: str = API_VERSION
|
||||
scan_ms: float | None = None
|
||||
_section_index: dict[str, InventorySection] = field(
|
||||
default_factory=dict, repr=False, compare=False
|
||||
)
|
||||
|
||||
def section(self, name: str) -> InventorySection | None:
|
||||
return self._section_index.get(name)
|
||||
|
||||
@property
|
||||
def degraded_sections(self) -> tuple[str, ...]:
|
||||
return tuple(s.name for s in self.sections if not s.ok)
|
||||
|
||||
@property
|
||||
def ownership_authority_complete(self) -> bool:
|
||||
"""True only when every ownership-bearing section read cleanly.
|
||||
|
||||
While this is false the snapshot must not describe any work item as
|
||||
unowned: a lease the reader could not load is not an absent lease.
|
||||
"""
|
||||
for name in OWNERSHIP_SECTIONS:
|
||||
section = self._section_index.get(name)
|
||||
if section is None or not section.ok:
|
||||
return False
|
||||
return True
|
||||
|
||||
@property
|
||||
def status(self) -> str:
|
||||
if all(s.ok for s in self.sections):
|
||||
return STATUS_OK
|
||||
return STATUS_DEGRADED
|
||||
|
||||
|
||||
# ── redaction ────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def redact_path(path: str | None) -> str | None:
|
||||
"""Collapse an absolute path against ``$HOME`` for browser display."""
|
||||
if not path:
|
||||
return path
|
||||
text = str(path)
|
||||
home = os.path.expanduser("~")
|
||||
if home and home != "/" and text.startswith(home):
|
||||
return "~" + text[len(home) :]
|
||||
return text
|
||||
|
||||
|
||||
def redact_url(value: str | None) -> str | None:
|
||||
"""Strip userinfo and query string from a URL."""
|
||||
if not value:
|
||||
return value
|
||||
text = str(value)
|
||||
try:
|
||||
parsed = urlparse(text)
|
||||
except ValueError:
|
||||
return _REDACTED
|
||||
if not parsed.scheme or not parsed.netloc:
|
||||
return text
|
||||
netloc = parsed.hostname or ""
|
||||
if parsed.port:
|
||||
netloc = f"{netloc}:{parsed.port}"
|
||||
rebuilt = f"{parsed.scheme}://{netloc}{parsed.path}"
|
||||
return rebuilt.rstrip("/") or rebuilt
|
||||
|
||||
|
||||
def scrub(value: Any, *, key: str | None = None) -> Any:
|
||||
"""Recursively drop credential-shaped values and redact paths/URLs.
|
||||
|
||||
Never raises: an unexpected object degrades to its ``repr`` rather than
|
||||
propagating out of a read-only view.
|
||||
"""
|
||||
if key and _CREDENTIAL_KEY_RE.search(key):
|
||||
return _REDACTED
|
||||
if isinstance(value, dict):
|
||||
return {str(k): scrub(v, key=str(k)) for k, v in value.items()}
|
||||
if isinstance(value, (list, tuple)):
|
||||
return [scrub(v, key=key) for v in value]
|
||||
if isinstance(value, str):
|
||||
if value.startswith(("http://", "https://")):
|
||||
return redact_url(value)
|
||||
if value.startswith("/") or value.startswith("~"):
|
||||
return redact_path(value)
|
||||
return value
|
||||
if isinstance(value, (int, float, bool)) or value is None:
|
||||
return value
|
||||
return repr(value)
|
||||
|
||||
|
||||
# ── control-plane database (read-only) ───────────────────────────────────────
|
||||
|
||||
|
||||
def _open_readonly(db_path: str) -> sqlite3.Connection:
|
||||
"""Open the control-plane DB without creating or migrating anything."""
|
||||
conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=5)
|
||||
conn.row_factory = sqlite3.Row
|
||||
return conn
|
||||
|
||||
|
||||
def _table_names(conn: sqlite3.Connection) -> set[str]:
|
||||
rows = conn.execute(
|
||||
"SELECT name FROM sqlite_master WHERE type = 'table'"
|
||||
).fetchall()
|
||||
return {str(row[0]) for row in rows}
|
||||
|
||||
|
||||
def _load_cp_db_sections(
|
||||
*,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
) -> tuple[InventorySection, InventorySection]:
|
||||
"""Return the ``sessions`` and ``leases`` sections from the #613 DB."""
|
||||
path = (db_path or control_plane_db.default_db_path()).strip()
|
||||
|
||||
def _both_unavailable(reason: str) -> tuple[InventorySection, InventorySection]:
|
||||
return (
|
||||
InventorySection(
|
||||
name="sessions",
|
||||
authority=AUTHORITY_CONTROL_PLANE_DB,
|
||||
status=STATUS_UNAVAILABLE,
|
||||
reason=reason,
|
||||
),
|
||||
InventorySection(
|
||||
name="leases",
|
||||
authority=AUTHORITY_CONTROL_PLANE_DB,
|
||||
status=STATUS_UNAVAILABLE,
|
||||
reason=reason,
|
||||
),
|
||||
)
|
||||
|
||||
if not path:
|
||||
return _both_unavailable("control-plane database path is not configured")
|
||||
if not os.path.exists(path):
|
||||
return _both_unavailable(
|
||||
f"control-plane database not present at {redact_path(path)}; "
|
||||
"no session or lease authority available"
|
||||
)
|
||||
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
conn = _open_readonly(path)
|
||||
except sqlite3.Error as exc:
|
||||
return _both_unavailable(f"control-plane database could not be opened: {exc}")
|
||||
|
||||
try:
|
||||
tables = _table_names(conn)
|
||||
if "sessions" not in tables or "leases" not in tables:
|
||||
missing = sorted({"sessions", "leases"} - tables)
|
||||
return _both_unavailable(
|
||||
"control-plane database is missing required tables: "
|
||||
+ ", ".join(missing)
|
||||
)
|
||||
|
||||
session_rows = [
|
||||
dict(row)
|
||||
for row in conn.execute(
|
||||
"SELECT session_id, role, profile, namespace, pid, started_at,"
|
||||
" last_heartbeat_at, status FROM sessions"
|
||||
" ORDER BY last_heartbeat_at DESC LIMIT ?",
|
||||
(max(1, int(limit)),),
|
||||
).fetchall()
|
||||
]
|
||||
|
||||
has_work_items = "work_items" in tables
|
||||
if has_work_items:
|
||||
lease_sql = (
|
||||
"SELECT l.lease_id, l.session_id, l.role, l.phase, l.status,"
|
||||
" l.expires_at, w.remote, w.org, w.repo, w.kind AS work_kind,"
|
||||
" w.number AS work_number, w.state AS work_state,"
|
||||
" s.pid AS session_pid, s.profile AS session_profile,"
|
||||
" s.namespace AS session_namespace, s.status AS session_status"
|
||||
" FROM leases l"
|
||||
" JOIN work_items w ON w.work_item_id = l.work_item_id"
|
||||
" LEFT JOIN sessions s ON s.session_id = l.session_id"
|
||||
" ORDER BY l.expires_at DESC LIMIT ?"
|
||||
)
|
||||
else:
|
||||
lease_sql = (
|
||||
"SELECT l.lease_id, l.session_id, l.role, l.phase, l.status,"
|
||||
" l.expires_at FROM leases l"
|
||||
" ORDER BY l.expires_at DESC LIMIT ?"
|
||||
)
|
||||
lease_rows = [
|
||||
dict(row)
|
||||
for row in conn.execute(lease_sql, (max(1, int(limit)),)).fetchall()
|
||||
]
|
||||
except sqlite3.Error as exc:
|
||||
return _both_unavailable(f"control-plane database read failed: {exc}")
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
elapsed = (time.perf_counter() - started) * 1000.0
|
||||
now = datetime.now(timezone.utc)
|
||||
|
||||
sessions = tuple(
|
||||
scrub(
|
||||
{
|
||||
"session_id": row.get("session_id"),
|
||||
"role": row.get("role"),
|
||||
"profile": row.get("profile"),
|
||||
"namespace": row.get("namespace"),
|
||||
"pid": row.get("pid"),
|
||||
"pid_alive": issue_lock_store.is_process_alive(row.get("pid")),
|
||||
"started_at": row.get("started_at"),
|
||||
"last_heartbeat_at": row.get("last_heartbeat_at"),
|
||||
"status": row.get("status"),
|
||||
}
|
||||
)
|
||||
for row in session_rows
|
||||
)
|
||||
|
||||
leases = tuple(
|
||||
scrub(
|
||||
{
|
||||
"lease_id": row.get("lease_id"),
|
||||
"session_id": row.get("session_id"),
|
||||
"role": row.get("role"),
|
||||
"phase": row.get("phase"),
|
||||
"status": row.get("status"),
|
||||
"expires_at": row.get("expires_at"),
|
||||
"expired": _is_expired(row.get("expires_at"), now=now),
|
||||
"remote": row.get("remote"),
|
||||
"org": row.get("org"),
|
||||
"repo": row.get("repo"),
|
||||
"work_kind": row.get("work_kind"),
|
||||
"work_number": row.get("work_number"),
|
||||
"work_state": row.get("work_state"),
|
||||
"session_pid": row.get("session_pid"),
|
||||
"session_profile": row.get("session_profile"),
|
||||
"session_namespace": row.get("session_namespace"),
|
||||
"session_status": row.get("session_status"),
|
||||
}
|
||||
)
|
||||
for row in lease_rows
|
||||
)
|
||||
|
||||
degraded_reason = (
|
||||
None
|
||||
if has_work_items
|
||||
else "work_items table absent; lease rows carry no work linkage"
|
||||
)
|
||||
lease_status = STATUS_OK if has_work_items else STATUS_DEGRADED
|
||||
|
||||
return (
|
||||
InventorySection(
|
||||
name="sessions",
|
||||
authority=AUTHORITY_CONTROL_PLANE_DB,
|
||||
status=STATUS_OK,
|
||||
items=sessions,
|
||||
scan_ms=round(elapsed, 3),
|
||||
),
|
||||
InventorySection(
|
||||
name="leases",
|
||||
authority=AUTHORITY_CONTROL_PLANE_DB,
|
||||
status=lease_status,
|
||||
items=leases,
|
||||
reason=degraded_reason,
|
||||
scan_ms=round(elapsed, 3),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _is_expired(expires_at: str | None, *, now: datetime) -> bool | None:
|
||||
if not expires_at:
|
||||
return None
|
||||
text = str(expires_at).strip().replace("Z", "+00:00")
|
||||
try:
|
||||
parsed = datetime.fromisoformat(text)
|
||||
except ValueError:
|
||||
return None
|
||||
if parsed.tzinfo is None:
|
||||
parsed = parsed.replace(tzinfo=timezone.utc)
|
||||
return parsed <= now
|
||||
|
||||
|
||||
# ── durable issue locks (filesystem) ─────────────────────────────────────────
|
||||
|
||||
|
||||
def _load_locks_section(*, lock_dir: str | None = None) -> InventorySection:
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
paths = issue_lock_store.iter_lock_files(lock_dir)
|
||||
except OSError as exc:
|
||||
return InventorySection(
|
||||
name="locks",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=STATUS_UNAVAILABLE,
|
||||
reason=f"issue lock directory could not be listed: {exc}",
|
||||
)
|
||||
|
||||
items: list[dict[str, Any]] = []
|
||||
unreadable = 0
|
||||
for path in paths:
|
||||
try:
|
||||
record = issue_lock_store.read_lock_file(path)
|
||||
except (OSError, ValueError):
|
||||
unreadable += 1
|
||||
continue
|
||||
if not record:
|
||||
unreadable += 1
|
||||
continue
|
||||
try:
|
||||
freshness = issue_lock_store.assess_lock_freshness(record)
|
||||
except Exception: # noqa: BLE001 — a read-only view never raises
|
||||
freshness = {"status": "unknown", "live": False, "stale": False}
|
||||
claimant = record.get("claimant") or (
|
||||
(record.get("work_lease") or {}).get("claimant") or {}
|
||||
)
|
||||
items.append(
|
||||
scrub(
|
||||
{
|
||||
"issue_number": record.get("issue_number"),
|
||||
"branch_name": record.get("branch_name"),
|
||||
"remote": record.get("remote"),
|
||||
"org": record.get("org"),
|
||||
"repo": record.get("repo"),
|
||||
"worktree_path": record.get("worktree_path"),
|
||||
"pid": record.get("session_pid") or record.get("pid"),
|
||||
"pid_alive": issue_lock_store.is_process_alive(
|
||||
record.get("session_pid") or record.get("pid")
|
||||
),
|
||||
"claimant_username": (claimant or {}).get("username"),
|
||||
"claimant_profile": (claimant or {}).get("profile"),
|
||||
"lock_generation": record.get("lock_generation"),
|
||||
"freshness_status": freshness.get("status"),
|
||||
"live": bool(freshness.get("live")),
|
||||
"stale": bool(freshness.get("stale")),
|
||||
"freshness_reason": freshness.get("reason"),
|
||||
"lock_path": record.get("lock_file_path") or path,
|
||||
}
|
||||
)
|
||||
)
|
||||
|
||||
elapsed = (time.perf_counter() - started) * 1000.0
|
||||
reason = (
|
||||
f"{unreadable} lock file(s) were unreadable and are not represented"
|
||||
if unreadable
|
||||
else None
|
||||
)
|
||||
return InventorySection(
|
||||
name="locks",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=STATUS_DEGRADED if unreadable else STATUS_OK,
|
||||
items=tuple(items),
|
||||
reason=reason,
|
||||
scan_ms=round(elapsed, 3),
|
||||
)
|
||||
|
||||
|
||||
# ── worktrees (filesystem, via the #432 scanner) ─────────────────────────────
|
||||
|
||||
|
||||
def _load_worktrees_section(
|
||||
*, load_hygiene: Callable[[], Any] | None = None
|
||||
) -> InventorySection:
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
loader = load_hygiene
|
||||
if loader is None:
|
||||
from webui.worktree_scanner import load_hygiene_snapshot
|
||||
|
||||
loader = load_hygiene_snapshot
|
||||
snapshot = loader()
|
||||
except Exception as exc: # noqa: BLE001 — fail soft, never fail the request
|
||||
return InventorySection(
|
||||
name="worktrees",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=STATUS_UNAVAILABLE,
|
||||
reason=f"worktree scan failed: {exc}",
|
||||
)
|
||||
|
||||
items = tuple(
|
||||
scrub(
|
||||
{
|
||||
"rel_path": entry.rel_path,
|
||||
"folder_name": entry.folder_name,
|
||||
"classification": entry.classification,
|
||||
"branch": entry.branch,
|
||||
"head_sha": entry.head_sha,
|
||||
"dirty_tracked": entry.dirty_tracked,
|
||||
"dirty_untracked": entry.dirty_untracked,
|
||||
"detached": entry.detached,
|
||||
"registered_worktree": entry.registered_worktree,
|
||||
"notes": entry.notes,
|
||||
}
|
||||
)
|
||||
for entry in snapshot.entries
|
||||
)
|
||||
scan_error = getattr(snapshot, "scan_error", None)
|
||||
elapsed = (time.perf_counter() - started) * 1000.0
|
||||
return InventorySection(
|
||||
name="worktrees",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=STATUS_DEGRADED if scan_error else STATUS_OK,
|
||||
items=items,
|
||||
reason=scan_error,
|
||||
scan_ms=round(elapsed, 3),
|
||||
)
|
||||
|
||||
|
||||
# ── namespaces / capability summary ──────────────────────────────────────────
|
||||
|
||||
|
||||
def _load_namespaces_section() -> InventorySection:
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
from gitea_auth import get_profile
|
||||
|
||||
profile = get_profile() or {}
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return InventorySection(
|
||||
name="namespaces",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=STATUS_UNAVAILABLE,
|
||||
reason=f"active profile could not be resolved: {exc}",
|
||||
)
|
||||
|
||||
allowed = list(profile.get("allowed_operations") or [])
|
||||
forbidden = list(profile.get("forbidden_operations") or [])
|
||||
profile_name = str(profile.get("profile_name") or "")
|
||||
|
||||
namespace = None
|
||||
try:
|
||||
import role_namespace_gate
|
||||
|
||||
namespace = role_namespace_gate.infer_mcp_namespace(profile_name)
|
||||
except Exception: # noqa: BLE001 — namespace inference is advisory
|
||||
namespace = None
|
||||
|
||||
item = scrub(
|
||||
{
|
||||
"profile_name": profile_name,
|
||||
"role": profile.get("role"),
|
||||
"mcp_namespace": namespace,
|
||||
"allowed_operations": sorted(allowed),
|
||||
"forbidden_operations": sorted(forbidden),
|
||||
"capability_summary": {
|
||||
"can_author": "gitea.pr.create" in allowed,
|
||||
"can_review": "gitea.pr.approve" in allowed,
|
||||
"can_merge": "gitea.pr.merge" in allowed,
|
||||
"can_close_pr": "gitea.pr.close" in allowed,
|
||||
},
|
||||
"active": True,
|
||||
}
|
||||
)
|
||||
elapsed = (time.perf_counter() - started) * 1000.0
|
||||
return InventorySection(
|
||||
name="namespaces",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=STATUS_OK,
|
||||
items=(item,),
|
||||
reason=(
|
||||
"only the profile serving this web process is observable; other "
|
||||
"namespaces are not enumerable from here"
|
||||
),
|
||||
scan_ms=round(elapsed, 3),
|
||||
)
|
||||
|
||||
|
||||
# ── correlation and collision detection ──────────────────────────────────────
|
||||
|
||||
|
||||
def _issue_from_branch(branch: str | None) -> int | None:
|
||||
match = re.search(r"issue-(\d+)", str(branch or ""), re.IGNORECASE)
|
||||
return int(match.group(1)) if match else None
|
||||
|
||||
|
||||
def correlate(
|
||||
*,
|
||||
leases: InventorySection,
|
||||
locks: InventorySection,
|
||||
worktrees: InventorySection,
|
||||
sessions: InventorySection,
|
||||
) -> tuple[tuple[dict[str, Any], ...], tuple[CollisionSignal, ...]]:
|
||||
"""Join lease owner ↔ lock ↔ worktree ↔ namespace where evidence allows.
|
||||
|
||||
Correlation rows are emitted from whatever sections did load. Collision
|
||||
signals are only emitted from sections that are ``ok``: a conflict inferred
|
||||
from a partially-read source would be a false accusation.
|
||||
"""
|
||||
correlations: list[dict[str, Any]] = []
|
||||
collisions: list[CollisionSignal] = []
|
||||
|
||||
worktree_by_branch: dict[str, dict[str, Any]] = {}
|
||||
for entry in worktrees.items:
|
||||
branch = (entry.get("branch") or "").strip()
|
||||
if branch:
|
||||
worktree_by_branch.setdefault(branch, entry)
|
||||
|
||||
session_by_id = {
|
||||
str(s.get("session_id")): s for s in sessions.items if s.get("session_id")
|
||||
}
|
||||
|
||||
# Lock-centred rows: a durable lock names an issue, a branch, and a worktree.
|
||||
for lock in locks.items:
|
||||
branch = (lock.get("branch_name") or "").strip()
|
||||
worktree = worktree_by_branch.get(branch)
|
||||
matching_leases = [
|
||||
lease
|
||||
for lease in leases.items
|
||||
if lease.get("work_kind") == "issue"
|
||||
and lease.get("work_number") == lock.get("issue_number")
|
||||
]
|
||||
correlations.append(
|
||||
{
|
||||
"issue_number": lock.get("issue_number"),
|
||||
"branch": branch or None,
|
||||
"lock_live": bool(lock.get("live")),
|
||||
"lock_claimant": lock.get("claimant_profile"),
|
||||
"lock_pid": lock.get("pid"),
|
||||
"lock_pid_alive": lock.get("pid_alive"),
|
||||
"worktree_rel_path": (worktree or {}).get("rel_path"),
|
||||
"worktree_classification": (worktree or {}).get("classification"),
|
||||
"worktree_registered": (worktree or {}).get("registered_worktree"),
|
||||
"lease_ids": [
|
||||
lease.get("lease_id")
|
||||
for lease in matching_leases
|
||||
if lease.get("lease_id")
|
||||
],
|
||||
"lease_sessions": [
|
||||
lease.get("session_id")
|
||||
for lease in matching_leases
|
||||
if lease.get("session_id")
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
if locks.ok and worktrees.ok:
|
||||
# A claim whose lease window is still open but has no registered
|
||||
# worktree is an anomaly regardless of whether its pid is alive; a
|
||||
# fully time-expired lease is on its way out and is not flagged.
|
||||
if (
|
||||
lock.get("freshness_status") != "expired"
|
||||
and branch
|
||||
and worktree is None
|
||||
):
|
||||
collisions.append(
|
||||
CollisionSignal(
|
||||
kind="lock-without-worktree",
|
||||
severity="warning",
|
||||
issue_number=lock.get("issue_number"),
|
||||
branch=branch,
|
||||
worktree_path=lock.get("worktree_path"),
|
||||
message=(
|
||||
f"Live lock on issue #{lock.get('issue_number')} names "
|
||||
f"branch {branch!r} but no registered worktree carries "
|
||||
"that branch (#404)"
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
if locks.ok:
|
||||
# A lock whose recorded pid is gone is held by nobody: a clean #753
|
||||
# dead-session recovery candidate. Subclassify by the lease window,
|
||||
# because the two cases need different operator urgency. When the
|
||||
# window is still open the lock would read as live to a naive
|
||||
# timestamp check even though the owner is dead — the more dangerous
|
||||
# case — so it is flagged distinctly from a fully time-expired lease.
|
||||
if lock.get("pid_alive") is False and lock.get("stale"):
|
||||
if lock.get("freshness_status") == "expired":
|
||||
collisions.append(
|
||||
CollisionSignal(
|
||||
kind="stale-lock-dead-owner",
|
||||
severity="warning",
|
||||
issue_number=lock.get("issue_number"),
|
||||
branch=branch or None,
|
||||
message=(
|
||||
f"Lock on issue #{lock.get('issue_number')} is stale "
|
||||
f"and its recorded pid {lock.get('pid')} is not running "
|
||||
"(#753 dead-session recovery candidate)"
|
||||
),
|
||||
)
|
||||
)
|
||||
else:
|
||||
collisions.append(
|
||||
CollisionSignal(
|
||||
kind="live-lock-dead-owner",
|
||||
severity="warning",
|
||||
issue_number=lock.get("issue_number"),
|
||||
branch=branch or None,
|
||||
message=(
|
||||
f"Lock on issue #{lock.get('issue_number')} has an "
|
||||
"unexpired lease but its recorded pid "
|
||||
f"{lock.get('pid')} is not running; it would read as "
|
||||
"live to a timestamp check (#753 dead-session recovery "
|
||||
"candidate)"
|
||||
),
|
||||
)
|
||||
)
|
||||
# The #635 trap: the lease has expired but the recorded pid is a
|
||||
# still-running daemon, so neither dead-pid reclaim nor exact-owner
|
||||
# renewal applies. This is the collision an operator must see.
|
||||
elif (
|
||||
lock.get("freshness_status") == "expired"
|
||||
and lock.get("pid_alive") is True
|
||||
):
|
||||
collisions.append(
|
||||
CollisionSignal(
|
||||
kind="expired-lock-live-owner",
|
||||
severity="error",
|
||||
issue_number=lock.get("issue_number"),
|
||||
branch=branch or None,
|
||||
message=(
|
||||
f"Lock on issue #{lock.get('issue_number')} has an expired "
|
||||
f"lease but its recorded pid {lock.get('pid')} is still "
|
||||
"running (daemon-pid deadlock; needs an operator decision, "
|
||||
"#635/#760)"
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
# Two live locks on one branch, or two active leases on one work item.
|
||||
if locks.ok:
|
||||
by_branch: dict[str, list[dict[str, Any]]] = {}
|
||||
for lock in locks.items:
|
||||
if not lock.get("live"):
|
||||
continue
|
||||
branch = (lock.get("branch_name") or "").strip()
|
||||
if branch:
|
||||
by_branch.setdefault(branch, []).append(lock)
|
||||
for branch, entries in sorted(by_branch.items()):
|
||||
if len(entries) > 1:
|
||||
collisions.append(
|
||||
CollisionSignal(
|
||||
kind="duplicate-live-lock",
|
||||
severity="error",
|
||||
branch=branch,
|
||||
message=(
|
||||
f"{len(entries)} live locks name branch {branch!r}: "
|
||||
"issues "
|
||||
+ ", ".join(
|
||||
f"#{e.get('issue_number')}" for e in entries
|
||||
)
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
if leases.ok:
|
||||
by_work: dict[tuple[str, int], list[dict[str, Any]]] = {}
|
||||
for lease in leases.items:
|
||||
if str(lease.get("status") or "").lower() != "active":
|
||||
continue
|
||||
kind = str(lease.get("work_kind") or "").strip().lower()
|
||||
number = lease.get("work_number")
|
||||
if not kind or number is None:
|
||||
continue
|
||||
by_work.setdefault((kind, int(number)), []).append(lease)
|
||||
for (kind, number), entries in sorted(by_work.items()):
|
||||
sessions_held = {
|
||||
str(e.get("session_id")) for e in entries if e.get("session_id")
|
||||
}
|
||||
if len(sessions_held) > 1:
|
||||
collisions.append(
|
||||
CollisionSignal(
|
||||
kind="concurrent-active-lease",
|
||||
severity="error",
|
||||
issue_number=number if kind == "issue" else None,
|
||||
session_ids=tuple(sorted(sessions_held)),
|
||||
message=(
|
||||
f"{len(sessions_held)} sessions hold an active lease on "
|
||||
f"{kind} #{number}"
|
||||
),
|
||||
)
|
||||
)
|
||||
for entry in entries:
|
||||
if entry.get("expired") is True:
|
||||
collisions.append(
|
||||
CollisionSignal(
|
||||
kind="active-lease-past-expiry",
|
||||
severity="warning",
|
||||
issue_number=number if kind == "issue" else None,
|
||||
session_ids=(
|
||||
(str(entry.get("session_id")),)
|
||||
if entry.get("session_id")
|
||||
else ()
|
||||
),
|
||||
message=(
|
||||
f"Lease {entry.get('lease_id')} on {kind} #{number} "
|
||||
"is still marked active past its expiry"
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
# A lease whose owning session is gone is an orphan, not free work.
|
||||
if leases.ok and sessions.ok:
|
||||
for lease in leases.items:
|
||||
if str(lease.get("status") or "").lower() != "active":
|
||||
continue
|
||||
session_id = str(lease.get("session_id") or "")
|
||||
if session_id and session_id not in session_by_id:
|
||||
collisions.append(
|
||||
CollisionSignal(
|
||||
kind="orphan-lease",
|
||||
severity="error",
|
||||
session_ids=(session_id,),
|
||||
message=(
|
||||
f"Active lease {lease.get('lease_id')} names session "
|
||||
f"{session_id}, which has no session record"
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
return tuple(correlations), tuple(collisions)
|
||||
|
||||
|
||||
# ── snapshot assembly ────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def load_inventory_snapshot(
|
||||
*,
|
||||
db_path: str | None = None,
|
||||
lock_dir: str | None = None,
|
||||
load_hygiene: Callable[[], Any] | None = None,
|
||||
include: tuple[str, ...] | None = None,
|
||||
) -> InventorySnapshot:
|
||||
"""Build the unified inventory snapshot.
|
||||
|
||||
Every section is loaded independently and fails soft. *include* restricts
|
||||
the sections that are scanned; omitted sections are simply absent rather
|
||||
than reported as empty, so a resource-split request cannot be mistaken for
|
||||
a whole-inventory answer.
|
||||
"""
|
||||
started = time.perf_counter()
|
||||
wanted = tuple(include) if include else SECTION_NAMES
|
||||
|
||||
sections: list[InventorySection] = []
|
||||
sessions_section: InventorySection | None = None
|
||||
leases_section: InventorySection | None = None
|
||||
|
||||
if "sessions" in wanted or "leases" in wanted:
|
||||
sessions_section, leases_section = _load_cp_db_sections(db_path=db_path)
|
||||
if "sessions" in wanted:
|
||||
sections.append(sessions_section)
|
||||
if "leases" in wanted:
|
||||
sections.append(leases_section)
|
||||
|
||||
locks_section = (
|
||||
_load_locks_section(lock_dir=lock_dir)
|
||||
if "locks" in wanted
|
||||
else _empty_section("locks", AUTHORITY_FILESYSTEM)
|
||||
)
|
||||
if "locks" in wanted:
|
||||
sections.append(locks_section)
|
||||
|
||||
worktrees_section = (
|
||||
_load_worktrees_section(load_hygiene=load_hygiene)
|
||||
if "worktrees" in wanted
|
||||
else _empty_section("worktrees", AUTHORITY_FILESYSTEM)
|
||||
)
|
||||
if "worktrees" in wanted:
|
||||
sections.append(worktrees_section)
|
||||
|
||||
if "namespaces" in wanted:
|
||||
sections.append(_load_namespaces_section())
|
||||
|
||||
correlations, collisions = correlate(
|
||||
leases=leases_section or _empty_section("leases", AUTHORITY_CONTROL_PLANE_DB),
|
||||
locks=locks_section,
|
||||
worktrees=worktrees_section,
|
||||
sessions=sessions_section
|
||||
or _empty_section("sessions", AUTHORITY_CONTROL_PLANE_DB),
|
||||
)
|
||||
|
||||
elapsed = (time.perf_counter() - started) * 1000.0
|
||||
index = {section.name: section for section in sections}
|
||||
return InventorySnapshot(
|
||||
generated_at=datetime.now(timezone.utc).isoformat(),
|
||||
sections=tuple(sections),
|
||||
collisions=collisions,
|
||||
correlations=correlations,
|
||||
scan_ms=round(elapsed, 3),
|
||||
_section_index=index,
|
||||
)
|
||||
|
||||
|
||||
def _empty_section(name: str, authority: str) -> InventorySection:
|
||||
"""A section that was not requested — never a claim that it is empty."""
|
||||
return InventorySection(
|
||||
name=name,
|
||||
authority=authority,
|
||||
status=STATUS_UNAVAILABLE,
|
||||
reason="section not requested in this scan",
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: InventorySnapshot) -> dict[str, Any]:
|
||||
"""Serialize the snapshot for the versioned API."""
|
||||
return {
|
||||
"api_version": snapshot.api_version,
|
||||
"schema_version": snapshot.schema_version,
|
||||
"generated_at": snapshot.generated_at,
|
||||
"status": snapshot.status,
|
||||
"scan_ms": snapshot.scan_ms,
|
||||
"ownership_authority_complete": snapshot.ownership_authority_complete,
|
||||
"ownership_note": (
|
||||
"Every ownership source read cleanly; an item absent from leases "
|
||||
"and locks is genuinely unclaimed."
|
||||
if snapshot.ownership_authority_complete
|
||||
else "One or more ownership sources are degraded; nothing in this "
|
||||
"snapshot may be treated as unowned. Collisions are reported only "
|
||||
"from sections that read cleanly."
|
||||
),
|
||||
"degraded_sections": list(snapshot.degraded_sections),
|
||||
"field_authority": {
|
||||
"sessions": AUTHORITY_CONTROL_PLANE_DB,
|
||||
"leases": AUTHORITY_CONTROL_PLANE_DB,
|
||||
"locks": AUTHORITY_FILESYSTEM,
|
||||
"worktrees": AUTHORITY_FILESYSTEM,
|
||||
"namespaces": AUTHORITY_FILESYSTEM,
|
||||
},
|
||||
"sections": {section.name: section.to_dict() for section in snapshot.sections},
|
||||
"correlations": [dict(row) for row in snapshot.correlations],
|
||||
"collisions": [signal.to_dict() for signal in snapshot.collisions],
|
||||
}
|
||||
@@ -65,6 +65,7 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
)),
|
||||
NavGroup("Insights", (
|
||||
NavItem("/insights", "Insights", "stub"),
|
||||
NavItem("/analytics", "Analytics"),
|
||||
NavItem("/audit", "Audit"),
|
||||
)),
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user