Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b179610e7f | ||
|
|
0041542fe2 | ||
|
|
1f144705e2 | ||
|
|
5c5c1fdf77 |
+231
-1
@@ -31,7 +31,7 @@ from typing import Any, Iterator, Sequence
|
||||
|
||||
import dependency_graph
|
||||
|
||||
SCHEMA_VERSION = 4
|
||||
SCHEMA_VERSION = 5
|
||||
|
||||
# Assignable work kinds only — raw monitoring incidents are never work items.
|
||||
WORK_KINDS = frozenset({"issue", "pr"})
|
||||
@@ -186,6 +186,34 @@ CREATE INDEX IF NOT EXISTS idx_dependency_edges_target
|
||||
ON dependency_edges(remote, org, repo, target_kind, target_number);
|
||||
CREATE INDEX IF NOT EXISTS idx_assignments_session ON assignments(session_id, status);
|
||||
CREATE INDEX IF NOT EXISTS idx_incident_gitea ON incident_links(gitea_org, gitea_repo, gitea_issue_number);
|
||||
|
||||
-- Model usage, token cost, latency, and performance events (#651)
|
||||
CREATE TABLE IF NOT EXISTS usage_events (
|
||||
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
session_id TEXT,
|
||||
remote TEXT NOT NULL DEFAULT 'dadeschools',
|
||||
org TEXT NOT NULL DEFAULT '',
|
||||
repo TEXT NOT NULL DEFAULT '',
|
||||
project_id TEXT,
|
||||
role TEXT NOT NULL DEFAULT 'unknown',
|
||||
model TEXT NOT NULL DEFAULT 'unknown',
|
||||
issue_number INTEGER,
|
||||
pr_number INTEGER,
|
||||
stage TEXT NOT NULL DEFAULT 'unknown',
|
||||
input_tokens INTEGER,
|
||||
output_tokens INTEGER,
|
||||
total_tokens INTEGER,
|
||||
estimated_cost_usd REAL,
|
||||
latency_ms INTEGER,
|
||||
duration_ms INTEGER,
|
||||
status TEXT NOT NULL DEFAULT 'success',
|
||||
metadata TEXT,
|
||||
created_at TEXT NOT NULL
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo);
|
||||
CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model);
|
||||
CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage);
|
||||
"""
|
||||
|
||||
|
||||
@@ -339,6 +367,7 @@ class ControlPlaneDB:
|
||||
self._migrate_incident_links_null_scope(conn)
|
||||
self._migrate_lease_lifecycle_columns(conn)
|
||||
self._migrate_session_ownership_columns(conn)
|
||||
self._migrate_usage_events_table(conn)
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO schema_meta(key, value) VALUES (?, ?)",
|
||||
("schema_version", str(SCHEMA_VERSION)),
|
||||
@@ -518,6 +547,207 @@ class ControlPlaneDB:
|
||||
f"UPDATE incident_links SET {col} = '' WHERE {col} IS NULL"
|
||||
)
|
||||
|
||||
def _migrate_usage_events_table(self, conn: sqlite3.Connection) -> None:
|
||||
"""Create usage_events table and indexes if they do not exist (#651)."""
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS usage_events (
|
||||
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
session_id TEXT,
|
||||
remote TEXT NOT NULL DEFAULT 'dadeschools',
|
||||
org TEXT NOT NULL DEFAULT '',
|
||||
repo TEXT NOT NULL DEFAULT '',
|
||||
project_id TEXT,
|
||||
role TEXT NOT NULL DEFAULT 'unknown',
|
||||
model TEXT NOT NULL DEFAULT 'unknown',
|
||||
issue_number INTEGER,
|
||||
pr_number INTEGER,
|
||||
stage TEXT NOT NULL DEFAULT 'unknown',
|
||||
input_tokens INTEGER,
|
||||
output_tokens INTEGER,
|
||||
total_tokens INTEGER,
|
||||
estimated_cost_usd REAL,
|
||||
latency_ms INTEGER,
|
||||
duration_ms INTEGER,
|
||||
status TEXT NOT NULL DEFAULT 'success',
|
||||
metadata TEXT,
|
||||
created_at TEXT NOT NULL
|
||||
);
|
||||
""")
|
||||
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_scope ON usage_events(remote, org, repo);")
|
||||
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_role_model ON usage_events(role, model);")
|
||||
conn.execute("CREATE INDEX IF NOT EXISTS idx_usage_events_stage ON usage_events(stage);")
|
||||
|
||||
# #651 retention: cap growth so unauthenticated or high-volume ingest
|
||||
# cannot DoS the control-plane DB (PR #876 F3). Applied after every write.
|
||||
USAGE_EVENTS_MAX_ROWS = 10_000
|
||||
USAGE_EVENTS_RETENTION_DAYS = 90
|
||||
|
||||
def record_usage_event(
|
||||
self,
|
||||
*,
|
||||
session_id: str | None = None,
|
||||
remote: str = "dadeschools",
|
||||
org: str = "",
|
||||
repo: str = "",
|
||||
project_id: str | None = None,
|
||||
role: str = "unknown",
|
||||
model: str = "unknown",
|
||||
issue_number: int | None = None,
|
||||
pr_number: int | None = None,
|
||||
stage: str = "unknown",
|
||||
input_tokens: int | None = None,
|
||||
output_tokens: int | None = None,
|
||||
total_tokens: int | None = None,
|
||||
estimated_cost_usd: float | None = None,
|
||||
latency_ms: int | None = None,
|
||||
duration_ms: int | None = None,
|
||||
status: str = "success",
|
||||
metadata: str | dict[str, Any] | None = None,
|
||||
created_at: str | None = None,
|
||||
) -> int:
|
||||
"""Record a model usage, token cost, latency, or stage performance event (#651)."""
|
||||
ts = created_at or _ts()
|
||||
meta_str: str | None = None
|
||||
if metadata is not None:
|
||||
from webui import console_redaction
|
||||
redacted_meta = console_redaction.redact_payload(metadata)
|
||||
if isinstance(redacted_meta, str):
|
||||
meta_str = redacted_meta
|
||||
else:
|
||||
try:
|
||||
meta_str = json.dumps(redacted_meta, default=str)
|
||||
except Exception:
|
||||
meta_str = str(redacted_meta)
|
||||
|
||||
if total_tokens is None and (input_tokens is not None or output_tokens is not None):
|
||||
total_tokens = (input_tokens or 0) + (output_tokens or 0)
|
||||
|
||||
with self._tx(immediate=True) as conn:
|
||||
cursor = conn.execute(
|
||||
"""
|
||||
INSERT INTO usage_events (
|
||||
session_id, remote, org, repo, project_id, role, model,
|
||||
issue_number, pr_number, stage, input_tokens, output_tokens,
|
||||
total_tokens, estimated_cost_usd, latency_ms, duration_ms,
|
||||
status, metadata, created_at
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
session_id,
|
||||
remote,
|
||||
org,
|
||||
repo,
|
||||
project_id,
|
||||
role,
|
||||
model,
|
||||
issue_number,
|
||||
pr_number,
|
||||
stage,
|
||||
input_tokens,
|
||||
output_tokens,
|
||||
total_tokens,
|
||||
estimated_cost_usd,
|
||||
latency_ms,
|
||||
duration_ms,
|
||||
status,
|
||||
meta_str,
|
||||
ts,
|
||||
),
|
||||
)
|
||||
usage_id = cursor.lastrowid
|
||||
self._enforce_usage_events_retention(conn)
|
||||
return usage_id
|
||||
|
||||
def _enforce_usage_events_retention(self, conn: sqlite3.Connection) -> None:
|
||||
"""Drop aged and excess usage_events rows (PR #876 F3)."""
|
||||
# Age-based: ISO-8601 UTC timestamps compare lexicographically.
|
||||
cutoff = (
|
||||
datetime.now(timezone.utc)
|
||||
- timedelta(days=int(self.USAGE_EVENTS_RETENTION_DAYS))
|
||||
).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
conn.execute(
|
||||
"DELETE FROM usage_events WHERE created_at < ?",
|
||||
(cutoff,),
|
||||
)
|
||||
# Count-based: keep the newest USAGE_EVENTS_MAX_ROWS by usage_id.
|
||||
max_rows = int(self.USAGE_EVENTS_MAX_ROWS)
|
||||
if max_rows > 0:
|
||||
conn.execute(
|
||||
"""
|
||||
DELETE FROM usage_events
|
||||
WHERE usage_id NOT IN (
|
||||
SELECT usage_id FROM usage_events
|
||||
ORDER BY usage_id DESC
|
||||
LIMIT ?
|
||||
)
|
||||
""",
|
||||
(max_rows,),
|
||||
)
|
||||
|
||||
def query_usage_events(
|
||||
self,
|
||||
*,
|
||||
remote: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
project_id: str | None = None,
|
||||
role: str | None = None,
|
||||
model: str | None = None,
|
||||
issue_number: int | None = None,
|
||||
pr_number: int | None = None,
|
||||
stage: str | None = None,
|
||||
session_id: str | None = None,
|
||||
limit: int = 500,
|
||||
offset: int = 0,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Query stored usage events matching filters (#651)."""
|
||||
conditions = []
|
||||
params = []
|
||||
if remote:
|
||||
conditions.append("remote = ?")
|
||||
params.append(remote)
|
||||
if org:
|
||||
conditions.append("org = ?")
|
||||
params.append(org)
|
||||
if repo:
|
||||
conditions.append("repo = ?")
|
||||
params.append(repo)
|
||||
if project_id:
|
||||
conditions.append("project_id = ?")
|
||||
params.append(project_id)
|
||||
if role:
|
||||
conditions.append("role = ?")
|
||||
params.append(role)
|
||||
if model:
|
||||
conditions.append("model = ?")
|
||||
params.append(model)
|
||||
if issue_number is not None:
|
||||
conditions.append("issue_number = ?")
|
||||
params.append(issue_number)
|
||||
if pr_number is not None:
|
||||
conditions.append("pr_number = ?")
|
||||
params.append(pr_number)
|
||||
if stage:
|
||||
conditions.append("stage = ?")
|
||||
params.append(stage)
|
||||
if session_id:
|
||||
conditions.append("session_id = ?")
|
||||
params.append(session_id)
|
||||
|
||||
where_clause = f"WHERE {' AND '.join(conditions)}" if conditions else ""
|
||||
sql = f"""
|
||||
SELECT * FROM usage_events
|
||||
{where_clause}
|
||||
ORDER BY usage_id ASC
|
||||
LIMIT ? OFFSET ?
|
||||
"""
|
||||
params.extend([limit, offset])
|
||||
|
||||
with self._tx(immediate=False) as conn:
|
||||
cursor = conn.execute(sql, params)
|
||||
rows = cursor.fetchall()
|
||||
return [dict(row) for row in rows]
|
||||
|
||||
# ── sessions ──────────────────────────────────────────────────────────
|
||||
|
||||
def upsert_session(
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
# Model Usage, Token Cost, Latency, and Workflow Analytics (Phase 4)
|
||||
|
||||
- **Tracking Issue:** [#651](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
|
||||
- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/651)
|
||||
- **Console Surface:** `/analytics`, `/api/v1/analytics`, `/api/v1/analytics/usage`
|
||||
|
||||
## 1. Overview
|
||||
|
||||
The Web Console Analytics module provides durable, aggregate visibility into **model usage, token cost, latency percentiles, and workflow-stage performance** across projects, worker roles, AI models, issues, and PRs.
|
||||
|
||||
### Non-Goals
|
||||
- No mandatory client-side telemetry that leaks prompts or secret keys.
|
||||
- No third-party payment provider or billing integration.
|
||||
- No automatic model routing changes without controller policy (#647).
|
||||
|
||||
---
|
||||
|
||||
## 2. Event Schema (`usage_events`)
|
||||
|
||||
Usage metrics are stored in the control-plane database under table `usage_events`.
|
||||
|
||||
| Column | Type | Description |
|
||||
|---|---|---|
|
||||
| `usage_id` | `INTEGER` | Primary key (autoincrement) |
|
||||
| `session_id` | `TEXT` | Optional active session identifier |
|
||||
| `remote` | `TEXT` | Known Gitea instance (`dadeschools` or `prgs`) |
|
||||
| `org` | `TEXT` | Repository owner / organization |
|
||||
| `repo` | `TEXT` | Repository name |
|
||||
| `project_id` | `TEXT` | Optional project identifier |
|
||||
| `role` | `TEXT` | Active worker role (`author`, `reviewer`, `merger`, `reconciler`, `controller`) |
|
||||
| `model` | `TEXT` | LLM model identifier (e.g. `gemini-3.6-flash`, `claude-3-5-sonnet`) |
|
||||
| `issue_number` | `INTEGER` | Correlated Gitea issue number (optional) |
|
||||
| `pr_number` | `INTEGER` | Correlated Gitea PR number (optional) |
|
||||
| `stage` | `TEXT` | Workflow stage (`preflight`, `implementation`, `review`, `merge`, `reconciliation`) |
|
||||
| `input_tokens` | `INTEGER` | Input token count (optional / nullable) |
|
||||
| `output_tokens` | `INTEGER` | Output token count (optional / nullable) |
|
||||
| `total_tokens` | `INTEGER` | Total token count (optional / nullable) |
|
||||
| `estimated_cost_usd` | `REAL` | Estimated USD cost (optional / nullable) |
|
||||
| `latency_ms` | `INTEGER` | Request latency in milliseconds (optional / nullable) |
|
||||
| `duration_ms` | `INTEGER` | Stage execution duration in milliseconds (optional / nullable) |
|
||||
| `status` | `TEXT` | Outcome status (`success`, `failure`, `timeout`) |
|
||||
| `metadata` | `TEXT` | Redacted metadata or summary string |
|
||||
| `created_at` | `TEXT` | ISO 8601 UTC timestamp |
|
||||
|
||||
---
|
||||
|
||||
## 3. Handling of Missing Data ("Unknown" vs. Zero Fabrication)
|
||||
|
||||
To ensure operational metrics accurately reflect evidence:
|
||||
- **Untracked or missing metrics are displayed as `Unknown`**, never zero-fabricated.
|
||||
- If an event omits `estimated_cost_usd`, `latency_ms`, or token counts, the aggregator marks those fields as missing (`None`) rather than defaulting to `0` or `$0.00`.
|
||||
- Summary tables and KPI cards explicitly indicate when data is unmeasured or partially reported.
|
||||
|
||||
---
|
||||
|
||||
## 4. Redaction & Security Rules
|
||||
|
||||
Per `#633` security policy:
|
||||
- Free-text fields (`metadata`, `prompt_summary`, `session_id`) are run through `console_redaction.redact_text` before persistence and output serialization.
|
||||
- Secret tokens, keychain commands, authorization headers, passwords, and JWTs are stripped automatically.
|
||||
|
||||
---
|
||||
|
||||
## 5. Opt-in Instrumentation Guide
|
||||
|
||||
Applications, MCP servers, and background sessions can report usage metrics through either Python API or HTTP ingestion.
|
||||
|
||||
### Python Ingestion
|
||||
|
||||
```python
|
||||
from webui.analytics_loader import record_usage
|
||||
|
||||
record_usage(
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role="author",
|
||||
model="gemini-3.6-flash",
|
||||
issue_number=651,
|
||||
stage="implementation",
|
||||
input_tokens=1420,
|
||||
output_tokens=380,
|
||||
total_tokens=1800,
|
||||
estimated_cost_usd=0.00045,
|
||||
latency_ms=320,
|
||||
duration_ms=4500,
|
||||
status="success",
|
||||
metadata={"note": "Implementation of analytics module"},
|
||||
)
|
||||
```
|
||||
|
||||
### HTTP Ingestion API (authorized write)
|
||||
|
||||
`POST /api/v1/analytics/usage` is a **gated write**. It runs through
|
||||
`console_authz` action `record_analytics_usage` (operator+, Phase 2 execution).
|
||||
Unauthenticated or phase-inactive requests receive **403** and do not write.
|
||||
Prefer in-process `record_usage` for MCP / session instrumentation.
|
||||
|
||||
```http
|
||||
POST /api/v1/analytics/usage HTTP/1.1
|
||||
Content-Type: application/json
|
||||
# Requires authenticated principal with record_analytics_usage execution enabled
|
||||
|
||||
{
|
||||
"remote": "dadeschools",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"role": "author",
|
||||
"model": "gemini-3.6-flash",
|
||||
"issue_number": 651,
|
||||
"stage": "implementation",
|
||||
"input_tokens": 1420,
|
||||
"output_tokens": 380,
|
||||
"total_tokens": 1800,
|
||||
"estimated_cost_usd": 0.00045,
|
||||
"latency_ms": 320,
|
||||
"duration_ms": 4500,
|
||||
"status": "success",
|
||||
"metadata": "Analytics schema landed"
|
||||
}
|
||||
```
|
||||
|
||||
### Retention
|
||||
|
||||
`usage_events` is retained with hard caps applied on every write:
|
||||
|
||||
| Limit | Default |
|
||||
|---|---|
|
||||
| Max rows | 10,000 (`ControlPlaneDB.USAGE_EVENTS_MAX_ROWS`) |
|
||||
| Max age | 90 days (`ControlPlaneDB.USAGE_EVENTS_RETENTION_DAYS`) |
|
||||
|
||||
Older rows (by `created_at`) and excess oldest rows (by `usage_id`) are deleted
|
||||
after each insert so unbounded growth / DoS-by-volume cannot fill the DB.
|
||||
|
||||
---
|
||||
|
||||
## 6. Querying Analytics API
|
||||
|
||||
```http
|
||||
GET /api/v1/analytics?role=author&stage=implementation HTTP/1.1
|
||||
```
|
||||
|
||||
Returns `AnalyticsSnapshot` JSON containing aggregations (`by_model`, `by_stage`, `by_role`, `by_work_item`, `by_project`) and latency percentiles (`p50`, `p90`, `p95`, `p99`).
|
||||
@@ -91,6 +91,7 @@ already define, and a regression test asserts each mapping matches.
|
||||
| `close_pr` | controller | privileged | `gitea.pr.close` | Yes | No | No | 3 |
|
||||
| `merge_pr` | controller | privileged | `gitea.pr.merge` | Yes | **Yes** | **Yes** | 3 |
|
||||
| `delete_branch` | admin | destructive | `gitea.branch.delete` | Yes | **Yes** | **Yes** | 3 |
|
||||
| `record_analytics_usage` | operator | gated_write | `runtime.record_analytics_usage` | Yes | No | No | 2 |
|
||||
|
||||
**Dual control** means the acting principal may not be the sole authority: a
|
||||
second distinct principal must confirm. **Break-glass** means the action is
|
||||
|
||||
@@ -66,8 +66,6 @@ status, onboarding checklist state, and the fail-closed error payloads (#635).
|
||||
| `/api/prompts` | JSON prompt export with workflow hashes |
|
||||
| `/runtime` | MCP runtime health and stale detection (#430) |
|
||||
| `/api/runtime` | JSON runtime health export |
|
||||
| `/policy` | Workflow policy and guardrail configuration visibility (#646) |
|
||||
| `/api/v1/policy` | Versioned JSON guardrail inventory (redacted, read-only) |
|
||||
| `/audit` | Report audit paste + validator preview (#431) |
|
||||
| `/api/audit` | JSON validator preview (POST `report_text`, optional `task_kind`) |
|
||||
| `/worktrees` | Worktree hygiene dashboard (#432) |
|
||||
@@ -241,19 +239,6 @@ health, workflow/schema SHA-256 hashes, and stale-runtime warnings when the
|
||||
checkout is behind merged safety-gate changes. Restart guidance links to #420;
|
||||
no tokens or MCP restart actions are exposed.
|
||||
|
||||
## Policy & guardrail visibility (#646)
|
||||
|
||||
`/policy` (HTML) and `/api/v1/policy` (JSON) surface a **read-only** projection
|
||||
of the major workflow guardrails — role separation/RBAC, lease lifecycle,
|
||||
author worktree binding, merge confirmation, secret redaction, contamination
|
||||
containment, allocator policy, audit logging, and mutation gating. Each entry
|
||||
carries source pointers to the file/module/doc that owns it, a compact active
|
||||
value derived from the existing safe policy accessors, and — where a documented
|
||||
default is declared — a diff of active vs documented. The whole payload is run
|
||||
through the console redaction pass before it is emitted, so a planted or
|
||||
accidental secret degrades to the placeholder rather than reaching a client.
|
||||
The view never edits policy and exposes no gate-weakening toggle.
|
||||
|
||||
## Application shell — Phase 1 (#638)
|
||||
|
||||
The console shell (`webui/layout.py`) renders a grouped navigation driven by a
|
||||
|
||||
@@ -495,6 +495,15 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.issue.comment",
|
||||
"role": "author",
|
||||
},
|
||||
|
||||
# #651 console analytics ingest — control-plane DB write, not a Gitea API
|
||||
# call. Authority comes from console RBAC (operator+) plus phase gating;
|
||||
# permission string is a non-Gitea runtime capability so no Gitea profile
|
||||
# can satisfy it by accident.
|
||||
"record_analytics_usage": {
|
||||
"permission": "runtime.record_analytics_usage",
|
||||
"role": "author",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -36,7 +36,7 @@ class ControlPlaneDBTest(unittest.TestCase):
|
||||
rows = dict(conn.execute("SELECT key, value FROM schema_meta").fetchall())
|
||||
finally:
|
||||
conn.close()
|
||||
self.assertEqual(rows["schema_version"], "4")
|
||||
self.assertEqual(rows["schema_version"], "5")
|
||||
self.assertIn("DB coordinates", rows["architecture"])
|
||||
self.assertIn("bridge", rows["architecture"].lower())
|
||||
|
||||
|
||||
@@ -0,0 +1,252 @@
|
||||
"""Unit and integration tests for Model Usage & Performance Analytics (#651)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
import control_plane_db
|
||||
from webui.analytics_loader import (
|
||||
ANALYTICS_SCHEMA_VERSION,
|
||||
compute_percentile,
|
||||
load_analytics,
|
||||
record_usage,
|
||||
)
|
||||
from webui.app import create_app
|
||||
from webui import console_redaction
|
||||
|
||||
|
||||
class AnalyticsLoaderTest(unittest.TestCase):
|
||||
|
||||
def setUp(self) -> None:
|
||||
self.temp_dir = tempfile.TemporaryDirectory()
|
||||
self.db_path = os.path.join(self.temp_dir.name, "test_control_plane.sqlite3")
|
||||
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
|
||||
self.db = control_plane_db.ControlPlaneDB(db_path=self.db_path)
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self.temp_dir.cleanup()
|
||||
|
||||
def test_compute_percentile(self) -> None:
|
||||
self.assertIsNone(compute_percentile([], 50.0))
|
||||
self.assertEqual(compute_percentile([100], 50.0), 100.0)
|
||||
|
||||
# 2 elements: [100, 200]
|
||||
self.assertEqual(compute_percentile([100, 200], 50.0), 150.0)
|
||||
|
||||
# 100 elements: 1..100
|
||||
vals = list(range(1, 101))
|
||||
self.assertEqual(compute_percentile(vals, 50.0), 50.5)
|
||||
self.assertAlmostEqual(compute_percentile(vals, 90.0), 90.1)
|
||||
|
||||
def test_record_and_aggregate_usage(self) -> None:
|
||||
# Record event 1 (complete data)
|
||||
u1 = record_usage(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role="author",
|
||||
model="gemini-3.6-flash",
|
||||
issue_number=651,
|
||||
stage="implementation",
|
||||
input_tokens=1000,
|
||||
output_tokens=500,
|
||||
estimated_cost_usd=0.0015,
|
||||
latency_ms=200,
|
||||
duration_ms=3000,
|
||||
metadata={"secret_key": "secret123", "note": "token=secret123"},
|
||||
)
|
||||
self.assertGreater(u1, 0)
|
||||
|
||||
# Record event 2 (missing tokens and cost -> unknown)
|
||||
u2 = record_usage(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role="reviewer",
|
||||
model="claude-3-5-sonnet",
|
||||
pr_number=846,
|
||||
stage="review",
|
||||
latency_ms=500,
|
||||
duration_ms=6000,
|
||||
)
|
||||
self.assertGreater(u2, u1)
|
||||
|
||||
snapshot = load_analytics(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
)
|
||||
|
||||
self.assertTrue(snapshot.ok)
|
||||
self.assertEqual(snapshot.schema_version, ANALYTICS_SCHEMA_VERSION)
|
||||
self.assertEqual(snapshot.total_events, 2)
|
||||
|
||||
# Verify overall summary
|
||||
summary = snapshot.overall_summary
|
||||
self.assertEqual(summary.total_events, 2)
|
||||
self.assertEqual(summary.events_with_tokens, 1)
|
||||
self.assertEqual(summary.total_tokens, 1500)
|
||||
self.assertEqual(summary.events_with_cost, 1)
|
||||
self.assertEqual(summary.estimated_cost_usd, 0.0015)
|
||||
self.assertEqual(summary.events_with_latency, 2)
|
||||
self.assertEqual(summary.latency_p50_ms, 350.0)
|
||||
|
||||
# Verify missing data handling (AC 3: not zero-fabricated)
|
||||
reviewer_model = snapshot.by_model.get("claude-3-5-sonnet")
|
||||
self.assertIsNotNone(reviewer_model)
|
||||
self.assertEqual(reviewer_model.total_events, 1)
|
||||
self.assertEqual(reviewer_model.events_with_tokens, 0)
|
||||
self.assertIsNone(reviewer_model.total_tokens)
|
||||
self.assertEqual(reviewer_model.display_tokens, "Unknown")
|
||||
self.assertEqual(reviewer_model.events_with_cost, 0)
|
||||
self.assertIsNone(reviewer_model.estimated_cost_usd)
|
||||
self.assertEqual(reviewer_model.display_cost, "Unknown")
|
||||
|
||||
# Verify redaction (AC 4)
|
||||
e1 = [e for e in snapshot.events if e.usage_id == u1][0]
|
||||
self.assertIsNotNone(e1.metadata)
|
||||
self.assertNotIn("secret123", e1.metadata)
|
||||
self.assertIn("[REDACTED]", e1.metadata)
|
||||
|
||||
def test_missing_db_fail_soft(self) -> None:
|
||||
invalid_path = "/nonexistent_path_dir/db.sqlite3"
|
||||
snapshot = load_analytics(db_path=invalid_path)
|
||||
self.assertFalse(snapshot.ok)
|
||||
self.assertIn("control_plane_db_unavailable", snapshot.reason)
|
||||
self.assertEqual(snapshot.overall_summary.display_tokens, "Unknown")
|
||||
|
||||
|
||||
class AnalyticsWebUITest(unittest.TestCase):
|
||||
|
||||
def setUp(self) -> None:
|
||||
self.temp_dir = tempfile.TemporaryDirectory()
|
||||
self.db_path = os.path.join(self.temp_dir.name, "test_webui.sqlite3")
|
||||
os.environ["GITEA_CONTROL_PLANE_DB"] = self.db_path
|
||||
self.app = create_app()
|
||||
self.client = TestClient(self.app)
|
||||
|
||||
record_usage(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role="author",
|
||||
model="gemini-3.6-flash",
|
||||
issue_number=651,
|
||||
stage="implementation",
|
||||
input_tokens=2000,
|
||||
output_tokens=1000,
|
||||
estimated_cost_usd=0.003,
|
||||
latency_ms=150,
|
||||
duration_ms=2500,
|
||||
)
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self.temp_dir.cleanup()
|
||||
|
||||
def test_analytics_html_route(self) -> None:
|
||||
response = self.client.get("/analytics")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertIn("Model Usage & Performance Analytics", response.text)
|
||||
self.assertIn("gemini-3.6-flash", response.text)
|
||||
self.assertIn("3,000", response.text)
|
||||
|
||||
def test_analytics_api_route(self) -> None:
|
||||
response = self.client.get("/api/v1/analytics")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
data = response.json()
|
||||
self.assertTrue(data["ok"])
|
||||
self.assertEqual(data["total_events"], 1)
|
||||
self.assertIn("gemini-3.6-flash", data["by_model"])
|
||||
|
||||
def test_analytics_ingest_unauthorized_denied(self) -> None:
|
||||
"""F2: unauthenticated POST must not write the control-plane DB."""
|
||||
payload = {
|
||||
"remote": "dadeschools",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"role": "reviewer",
|
||||
"model": "claude-3-5-sonnet",
|
||||
"pr_number": 846,
|
||||
"stage": "review",
|
||||
"input_tokens": 500,
|
||||
"output_tokens": 100,
|
||||
"latency_ms": 400,
|
||||
"metadata": "Review note token=secret456",
|
||||
}
|
||||
response = self.client.post("/api/v1/analytics/usage", json=payload)
|
||||
self.assertEqual(response.status_code, 403)
|
||||
res_json = response.json()
|
||||
self.assertFalse(res_json.get("ok", True))
|
||||
self.assertEqual(res_json.get("error"), "unauthorized")
|
||||
authorization = res_json.get("authorization") or {}
|
||||
self.assertFalse(authorization.get("allowed"))
|
||||
self.assertFalse(authorization.get("execution_enabled"))
|
||||
|
||||
# No new row written
|
||||
res2 = self.client.get("/api/v1/analytics")
|
||||
self.assertEqual(res2.status_code, 200)
|
||||
self.assertEqual(res2.json()["total_events"], 1)
|
||||
|
||||
def test_html_escapes_script_bearing_model_role_stage(self) -> None:
|
||||
"""F1: stored XSS — dynamic model/role/stage must render escaped."""
|
||||
xss = '<script>alert(1)</script>'
|
||||
record_usage(
|
||||
db_path=self.db_path,
|
||||
remote="dadeschools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
role=xss,
|
||||
model=xss,
|
||||
stage=xss,
|
||||
issue_number=999,
|
||||
status="success",
|
||||
)
|
||||
response = self.client.get("/analytics")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
# Raw tag must not appear; escaped form must.
|
||||
self.assertNotIn("<script>alert(1)</script>", response.text)
|
||||
self.assertIn("<script>alert(1)</script>", response.text)
|
||||
|
||||
def test_load_analytics_coerces_none_scope(self) -> None:
|
||||
"""F4: None remote/org/repo become empty strings, never None."""
|
||||
snapshot = load_analytics(db_path=self.db_path, remote=None, org=None, repo=None)
|
||||
self.assertIsInstance(snapshot.remote, str)
|
||||
self.assertIsInstance(snapshot.org, str)
|
||||
self.assertIsInstance(snapshot.repo, str)
|
||||
self.assertEqual(snapshot.remote, "")
|
||||
self.assertEqual(snapshot.org, "")
|
||||
self.assertEqual(snapshot.repo, "")
|
||||
|
||||
def test_usage_events_retention_max_rows(self) -> None:
|
||||
"""F3: record_usage_event enforces USAGE_EVENTS_MAX_ROWS."""
|
||||
db = control_plane_db.ControlPlaneDB(db_path=self.db_path)
|
||||
original_max = db.USAGE_EVENTS_MAX_ROWS
|
||||
try:
|
||||
db.USAGE_EVENTS_MAX_ROWS = 3
|
||||
for i in range(5):
|
||||
db.record_usage_event(
|
||||
remote="dadeschools",
|
||||
org="org",
|
||||
repo="repo",
|
||||
role="author",
|
||||
model=f"model-{i}",
|
||||
stage="test",
|
||||
)
|
||||
rows = db.query_usage_events(limit=100)
|
||||
self.assertLessEqual(len(rows), 3)
|
||||
# Newest three retained
|
||||
models = {r["model"] for r in rows}
|
||||
self.assertEqual(models, {"model-2", "model-3", "model-4"})
|
||||
finally:
|
||||
db.USAGE_EVENTS_MAX_ROWS = original_max
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -1,222 +0,0 @@
|
||||
"""Tests for the read-only workflow policy/guardrail visibility view (#646).
|
||||
|
||||
Covers issue #646 acceptance criteria:
|
||||
|
||||
1. Console lists major guardrails with source pointers.
|
||||
2. Secrets redacted.
|
||||
3. Tests ensure sample secrets never appear.
|
||||
4. Docs explain read-only nature (asserted here for the page copy; the doc
|
||||
itself is covered by inspection).
|
||||
"""
|
||||
|
||||
import json
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from webui import console_redaction
|
||||
from webui import policy_inventory
|
||||
from webui.app import create_app
|
||||
from webui.policy_inventory import (
|
||||
PolicyEntry,
|
||||
PolicyInventorySnapshot,
|
||||
SourcePointer,
|
||||
load_policy_inventory,
|
||||
snapshot_to_dict,
|
||||
)
|
||||
from webui.policy_views import render_policy_page
|
||||
|
||||
|
||||
def _entry(key, category, *, active=None, error=None):
|
||||
return PolicyEntry(
|
||||
key=key,
|
||||
title=key.replace("_", " ").title(),
|
||||
category=category,
|
||||
summary=f"summary for {key}",
|
||||
sources=(SourcePointer("src", f"{key}.py", "module"),),
|
||||
active=active,
|
||||
documented_default=None,
|
||||
diff=None,
|
||||
error=error,
|
||||
)
|
||||
|
||||
|
||||
def _snapshot(entries):
|
||||
return PolicyInventorySnapshot(
|
||||
schema_version=1,
|
||||
read_only=True,
|
||||
note="read-only projection",
|
||||
entries=tuple(entries),
|
||||
categories=tuple(dict.fromkeys(e.category for e in entries)),
|
||||
build_errors=(),
|
||||
)
|
||||
|
||||
# The guardrail categories issue #646 names as in-scope.
|
||||
_EXPECTED_CATEGORIES = {
|
||||
"role_separation",
|
||||
"lease_rules",
|
||||
"worktree_rules",
|
||||
"merge_confirmation",
|
||||
"redaction",
|
||||
"contamination",
|
||||
"allocator_policy",
|
||||
"audit_logging",
|
||||
"mutation_gating",
|
||||
}
|
||||
|
||||
|
||||
class TestPolicyInventoryModel(unittest.TestCase):
|
||||
def test_major_guardrails_present(self):
|
||||
snapshot = load_policy_inventory()
|
||||
categories = {e.category for e in snapshot.entries}
|
||||
self.assertEqual(_EXPECTED_CATEGORIES, categories)
|
||||
self.assertGreaterEqual(len(snapshot.entries), len(_EXPECTED_CATEGORIES))
|
||||
|
||||
def test_every_guardrail_has_source_pointers(self):
|
||||
# AC1: source attribution (file/module/doc) for every guardrail.
|
||||
snapshot = load_policy_inventory()
|
||||
for entry in snapshot.entries:
|
||||
with self.subTest(entry=entry.key):
|
||||
self.assertTrue(entry.sources, "guardrail must carry source pointers")
|
||||
for source in entry.sources:
|
||||
self.assertTrue(source.path)
|
||||
self.assertIn(source.kind, {"module", "doc", "script", "config"})
|
||||
|
||||
def test_diff_reported_where_documented_default_declared(self):
|
||||
snapshot = load_policy_inventory()
|
||||
checked_any = False
|
||||
for entry in snapshot.entries:
|
||||
if entry.documented_default is None:
|
||||
self.assertIsNone(entry.diff)
|
||||
continue
|
||||
checked_any = True
|
||||
self.assertIsNotNone(entry.diff)
|
||||
self.assertEqual(
|
||||
entry.diff["status"],
|
||||
"matches_documented_default",
|
||||
f"{entry.key} drifted from its documented default: {entry.diff}",
|
||||
)
|
||||
self.assertTrue(checked_any, "at least one guardrail should declare a default")
|
||||
|
||||
def test_live_projections_populate_active(self):
|
||||
snapshot = load_policy_inventory()
|
||||
by_key = {e.key: e for e in snapshot.entries}
|
||||
for key in ("role_separation", "redaction", "audit_logging"):
|
||||
self.assertIsNone(by_key[key].error, f"{key} projection failed")
|
||||
self.assertIsInstance(by_key[key].active, dict)
|
||||
|
||||
def test_build_entry_is_fail_soft_on_projection_error(self):
|
||||
def _boom():
|
||||
raise RuntimeError("projection exploded")
|
||||
|
||||
row = (
|
||||
"redaction",
|
||||
"Secret redaction",
|
||||
"redaction",
|
||||
"summary",
|
||||
(SourcePointer("x", "webui/console_redaction.py", "module"),),
|
||||
_boom,
|
||||
{"redact_before_persist": True},
|
||||
)
|
||||
entry = policy_inventory._build_entry(row)
|
||||
self.assertIsNone(entry.active)
|
||||
self.assertIsNotNone(entry.error)
|
||||
self.assertEqual(entry.diff["status"], "active_unavailable")
|
||||
|
||||
|
||||
class TestPolicyRedaction(unittest.TestCase):
|
||||
def test_real_snapshot_has_no_secret_shapes(self):
|
||||
# AC3: the real emitted payload never carries a known secret shape.
|
||||
payload = snapshot_to_dict(load_policy_inventory())
|
||||
self.assertEqual(console_redaction.scan_for_secrets(payload), [])
|
||||
|
||||
def test_planted_keychain_secret_is_redacted(self):
|
||||
# AC2/AC3: a secret planted in an active projection is masked before emit.
|
||||
snapshot = _snapshot([
|
||||
_entry(
|
||||
"redaction",
|
||||
"redaction",
|
||||
active={"leaked": "keychain:prgs-author-super-secret", "roles": ["author"]},
|
||||
)
|
||||
])
|
||||
payload = snapshot_to_dict(snapshot)
|
||||
blob = json.dumps(payload)
|
||||
self.assertNotIn("keychain:prgs-author-super-secret", blob)
|
||||
self.assertEqual(console_redaction.scan_for_secrets(payload), [])
|
||||
|
||||
def test_planted_credential_assignment_is_redacted(self):
|
||||
snapshot = _snapshot([
|
||||
_entry(
|
||||
"audit_logging",
|
||||
"audit_logging",
|
||||
active={"leaked": "token=abcd1234efgh5678", "append_only": True},
|
||||
)
|
||||
])
|
||||
payload = snapshot_to_dict(snapshot)
|
||||
blob = json.dumps(payload)
|
||||
self.assertNotIn("abcd1234efgh5678", blob)
|
||||
self.assertEqual(console_redaction.scan_for_secrets(payload), [])
|
||||
|
||||
|
||||
class TestPolicyRoutes(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.client = TestClient(create_app())
|
||||
|
||||
def test_policy_html_lists_guardrails_with_sources(self):
|
||||
response = self.client.get("/policy")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
text = response.text
|
||||
self.assertIn("Workflow policy", text)
|
||||
self.assertIn("Role separation and RBAC", text)
|
||||
self.assertIn("Source pointers", text)
|
||||
self.assertIn("task_capability_map.py", text)
|
||||
self.assertIn("docs/safety-model.md", text)
|
||||
|
||||
def test_policy_html_states_read_only(self):
|
||||
# AC4: the page explains its read-only nature.
|
||||
text = self.client.get("/policy").text
|
||||
self.assertIn("read-only", text.lower())
|
||||
self.assertNotIn("<form", text.lower())
|
||||
|
||||
def test_policy_html_has_no_secret_shapes(self):
|
||||
text = self.client.get("/policy").text
|
||||
self.assertEqual(console_redaction.scan_for_secrets(text), [])
|
||||
|
||||
def test_api_v1_policy_returns_inventory(self):
|
||||
response = self.client.get("/api/v1/policy")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
data = response.json()
|
||||
self.assertEqual(data["schema_version"], policy_inventory.SCHEMA_VERSION)
|
||||
self.assertTrue(data["read_only"])
|
||||
self.assertEqual(data["entry_count"], len(data["entries"]))
|
||||
self.assertEqual(set(data["categories"]), _EXPECTED_CATEGORIES)
|
||||
|
||||
def test_policy_is_read_only_no_post(self):
|
||||
# AC4 / non-goal: no mutation endpoint.
|
||||
response = self.client.post("/policy")
|
||||
self.assertIn(response.status_code, (404, 405))
|
||||
|
||||
def test_nav_links_policy(self):
|
||||
text = self.client.get("/").text
|
||||
self.assertIn('href="/policy"', text)
|
||||
|
||||
|
||||
class TestPolicyViewFailSoft(unittest.TestCase):
|
||||
def test_page_renders_when_a_projection_errors(self):
|
||||
snapshot = _snapshot([
|
||||
_entry("role_separation", "role_separation", error="active projection unavailable: boom"),
|
||||
_entry("redaction", "redaction", active={"redact_before_persist": True}),
|
||||
])
|
||||
page = render_policy_page(snapshot)
|
||||
# The errored guardrail surfaces its error; other guardrails still render.
|
||||
self.assertIn("Active value unavailable", page)
|
||||
self.assertIn("Redaction", page)
|
||||
self.assertIn("Workflow policy", page)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,433 @@
|
||||
"""Model usage, token cost, latency, and workflow-performance analytics (#651, Phase 4).
|
||||
|
||||
Ingests session instrumentation metrics, aggregates usage/cost/latency percentiles
|
||||
by project, role, model, issue/PR, and stage, enforcing secret redaction and
|
||||
explicitly rendering missing metrics as "Unknown" without zero-fabrication.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Any, Sequence
|
||||
|
||||
import control_plane_db
|
||||
from webui import console_redaction
|
||||
|
||||
ANALYTICS_SCHEMA_VERSION = 1
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class UsageEvent:
|
||||
usage_id: int
|
||||
session_id: str | None
|
||||
remote: str
|
||||
org: str
|
||||
repo: str
|
||||
project_id: str | None
|
||||
role: str
|
||||
model: str
|
||||
issue_number: int | None
|
||||
pr_number: int | None
|
||||
stage: str
|
||||
input_tokens: int | None
|
||||
output_tokens: int | None
|
||||
total_tokens: int | None
|
||||
estimated_cost_usd: float | None
|
||||
latency_ms: int | None
|
||||
duration_ms: int | None
|
||||
status: str
|
||||
metadata: str | None
|
||||
created_at: str
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
d = asdict(self)
|
||||
if d["metadata"]:
|
||||
d["metadata"] = console_redaction.redact_text(d["metadata"])
|
||||
return d
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class GroupMetrics:
|
||||
name: str
|
||||
total_events: int
|
||||
events_with_tokens: int
|
||||
input_tokens: int | None
|
||||
output_tokens: int | None
|
||||
total_tokens: int | None
|
||||
events_with_cost: int
|
||||
estimated_cost_usd: float | None
|
||||
events_with_latency: int
|
||||
latency_p50_ms: float | None
|
||||
latency_p90_ms: float | None
|
||||
latency_p95_ms: float | None
|
||||
latency_p99_ms: float | None
|
||||
latency_avg_ms: float | None
|
||||
events_with_duration: int
|
||||
duration_avg_ms: float | None
|
||||
display_tokens: str
|
||||
display_cost: str
|
||||
display_latency_p50: str
|
||||
display_latency_p90: str
|
||||
display_duration_avg: str
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AnalyticsSnapshot:
|
||||
ok: bool
|
||||
reason: str
|
||||
schema_version: int
|
||||
remote: str
|
||||
org: str
|
||||
repo: str
|
||||
total_events: int
|
||||
overall_summary: GroupMetrics
|
||||
by_project: dict[str, GroupMetrics]
|
||||
by_role: dict[str, GroupMetrics]
|
||||
by_model: dict[str, GroupMetrics]
|
||||
by_work_item: dict[str, GroupMetrics]
|
||||
by_stage: dict[str, GroupMetrics]
|
||||
events: tuple[UsageEvent, ...]
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"ok": self.ok,
|
||||
"reason": self.reason,
|
||||
"schema_version": self.schema_version,
|
||||
"remote": self.remote,
|
||||
"org": self.org,
|
||||
"repo": self.repo,
|
||||
"total_events": self.total_events,
|
||||
"overall_summary": self.overall_summary.to_dict(),
|
||||
"by_project": {k: v.to_dict() for k, v in self.by_project.items()},
|
||||
"by_role": {k: v.to_dict() for k, v in self.by_role.items()},
|
||||
"by_model": {k: v.to_dict() for k, v in self.by_model.items()},
|
||||
"by_work_item": {k: v.to_dict() for k, v in self.by_work_item.items()},
|
||||
"by_stage": {k: v.to_dict() for k, v in self.by_stage.items()},
|
||||
"events": [e.to_dict() for e in self.events],
|
||||
}
|
||||
|
||||
|
||||
def compute_percentile(values: Sequence[float | int], percentile: float) -> float | None:
|
||||
if not values:
|
||||
return None
|
||||
sorted_vals = sorted(values)
|
||||
n = len(sorted_vals)
|
||||
if n == 1:
|
||||
return float(sorted_vals[0])
|
||||
k = (n - 1) * (percentile / 100.0)
|
||||
f = math.floor(k)
|
||||
c = math.ceil(k)
|
||||
if f == c:
|
||||
return float(sorted_vals[int(f)])
|
||||
d0 = sorted_vals[int(f)] * (c - k)
|
||||
d1 = sorted_vals[int(c)] * (k - f)
|
||||
return float(d0 + d1)
|
||||
|
||||
|
||||
def aggregate_events(group_name: str, events: Sequence[UsageEvent]) -> GroupMetrics:
|
||||
total_events = len(events)
|
||||
if total_events == 0:
|
||||
return GroupMetrics(
|
||||
name=group_name,
|
||||
total_events=0,
|
||||
events_with_tokens=0,
|
||||
input_tokens=None,
|
||||
output_tokens=None,
|
||||
total_tokens=None,
|
||||
events_with_cost=0,
|
||||
estimated_cost_usd=None,
|
||||
events_with_latency=0,
|
||||
latency_p50_ms=None,
|
||||
latency_p90_ms=None,
|
||||
latency_p95_ms=None,
|
||||
latency_p99_ms=None,
|
||||
latency_avg_ms=None,
|
||||
events_with_duration=0,
|
||||
duration_avg_ms=None,
|
||||
display_tokens="Unknown",
|
||||
display_cost="Unknown",
|
||||
display_latency_p50="Unknown",
|
||||
display_latency_p90="Unknown",
|
||||
display_duration_avg="Unknown",
|
||||
)
|
||||
|
||||
token_events = [
|
||||
e for e in events
|
||||
if e.total_tokens is not None or e.input_tokens is not None or e.output_tokens is not None
|
||||
]
|
||||
events_with_tokens = len(token_events)
|
||||
if events_with_tokens > 0:
|
||||
input_tokens = sum(e.input_tokens or 0 for e in token_events)
|
||||
output_tokens = sum(e.output_tokens or 0 for e in token_events)
|
||||
total_tokens = sum(
|
||||
e.total_tokens if e.total_tokens is not None else ((e.input_tokens or 0) + (e.output_tokens or 0))
|
||||
for e in token_events
|
||||
)
|
||||
display_tokens = f"{total_tokens:,}"
|
||||
else:
|
||||
input_tokens = None
|
||||
output_tokens = None
|
||||
total_tokens = None
|
||||
display_tokens = "Unknown"
|
||||
|
||||
cost_events = [e for e in events if e.estimated_cost_usd is not None]
|
||||
events_with_cost = len(cost_events)
|
||||
if events_with_cost > 0:
|
||||
estimated_cost_usd = round(sum(e.estimated_cost_usd for e in cost_events), 6)
|
||||
display_cost = f"${estimated_cost_usd:.4f}"
|
||||
else:
|
||||
estimated_cost_usd = None
|
||||
display_cost = "Unknown"
|
||||
|
||||
latency_vals = [e.latency_ms for e in events if e.latency_ms is not None]
|
||||
events_with_latency = len(latency_vals)
|
||||
if events_with_latency > 0:
|
||||
latency_p50_ms = compute_percentile(latency_vals, 50.0)
|
||||
latency_p90_ms = compute_percentile(latency_vals, 90.0)
|
||||
latency_p95_ms = compute_percentile(latency_vals, 95.0)
|
||||
latency_p99_ms = compute_percentile(latency_vals, 99.0)
|
||||
latency_avg_ms = round(sum(latency_vals) / events_with_latency, 2)
|
||||
display_latency_p50 = f"{round(latency_p50_ms, 1)} ms" if latency_p50_ms is not None else "Unknown"
|
||||
display_latency_p90 = f"{round(latency_p90_ms, 1)} ms" if latency_p90_ms is not None else "Unknown"
|
||||
else:
|
||||
latency_p50_ms = None
|
||||
latency_p90_ms = None
|
||||
latency_p95_ms = None
|
||||
latency_p99_ms = None
|
||||
latency_avg_ms = None
|
||||
display_latency_p50 = "Unknown"
|
||||
display_latency_p90 = "Unknown"
|
||||
|
||||
duration_vals = [e.duration_ms for e in events if e.duration_ms is not None]
|
||||
events_with_duration = len(duration_vals)
|
||||
if events_with_duration > 0:
|
||||
duration_avg_ms = round(sum(duration_vals) / events_with_duration, 2)
|
||||
display_duration_avg = f"{round(duration_avg_ms / 1000.0, 2)} s" if duration_avg_ms >= 1000 else f"{round(duration_avg_ms, 1)} ms"
|
||||
else:
|
||||
duration_avg_ms = None
|
||||
display_duration_avg = "Unknown"
|
||||
|
||||
return GroupMetrics(
|
||||
name=group_name,
|
||||
total_events=total_events,
|
||||
events_with_tokens=events_with_tokens,
|
||||
input_tokens=input_tokens,
|
||||
output_tokens=output_tokens,
|
||||
total_tokens=total_tokens,
|
||||
events_with_cost=events_with_cost,
|
||||
estimated_cost_usd=estimated_cost_usd,
|
||||
events_with_latency=events_with_latency,
|
||||
latency_p50_ms=latency_p50_ms,
|
||||
latency_p90_ms=latency_p90_ms,
|
||||
latency_p95_ms=latency_p95_ms,
|
||||
latency_p99_ms=latency_p99_ms,
|
||||
latency_avg_ms=latency_avg_ms,
|
||||
events_with_duration=events_with_duration,
|
||||
duration_avg_ms=duration_avg_ms,
|
||||
display_tokens=display_tokens,
|
||||
display_cost=display_cost,
|
||||
display_latency_p50=display_latency_p50,
|
||||
display_latency_p90=display_latency_p90,
|
||||
display_duration_avg=display_duration_avg,
|
||||
)
|
||||
|
||||
|
||||
def record_usage(
|
||||
*,
|
||||
db_path: str | None = None,
|
||||
session_id: str | None = None,
|
||||
remote: str = "dadeschools",
|
||||
org: str = "",
|
||||
repo: str = "",
|
||||
project_id: str | None = None,
|
||||
role: str = "unknown",
|
||||
model: str = "unknown",
|
||||
issue_number: int | None = None,
|
||||
pr_number: int | None = None,
|
||||
stage: str = "unknown",
|
||||
input_tokens: int | None = None,
|
||||
output_tokens: int | None = None,
|
||||
total_tokens: int | None = None,
|
||||
estimated_cost_usd: float | None = None,
|
||||
latency_ms: int | None = None,
|
||||
duration_ms: int | None = None,
|
||||
status: str = "success",
|
||||
metadata: str | dict[str, Any] | None = None,
|
||||
created_at: str | None = None,
|
||||
) -> int:
|
||||
"""Ingest/record a single usage event with optional metrics."""
|
||||
db = control_plane_db.ControlPlaneDB(db_path=db_path)
|
||||
return db.record_usage_event(
|
||||
session_id=session_id,
|
||||
remote=remote,
|
||||
org=org,
|
||||
repo=repo,
|
||||
project_id=project_id,
|
||||
role=role,
|
||||
model=model,
|
||||
issue_number=issue_number,
|
||||
pr_number=pr_number,
|
||||
stage=stage,
|
||||
input_tokens=input_tokens,
|
||||
output_tokens=output_tokens,
|
||||
total_tokens=total_tokens,
|
||||
estimated_cost_usd=estimated_cost_usd,
|
||||
latency_ms=latency_ms,
|
||||
duration_ms=duration_ms,
|
||||
status=status,
|
||||
metadata=metadata,
|
||||
created_at=created_at,
|
||||
)
|
||||
|
||||
|
||||
def load_analytics(
|
||||
*,
|
||||
db_path: str | None = None,
|
||||
remote: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
project_id: str | None = None,
|
||||
role: str | None = None,
|
||||
model: str | None = None,
|
||||
stage: str | None = None,
|
||||
issue_number: int | None = None,
|
||||
pr_number: int | None = None,
|
||||
limit: int = 500,
|
||||
) -> AnalyticsSnapshot:
|
||||
"""Load analytics snapshot aggregated by project, role, model, issue/PR, and stage."""
|
||||
remote_filter = (remote or "").strip() or None
|
||||
org_filter = (org or "").strip() or None
|
||||
repo_filter = (repo or "").strip() or None
|
||||
role_filter = (role or "").strip() or None
|
||||
model_filter = (model or "").strip() or None
|
||||
stage_filter = (stage or "").strip() or None
|
||||
|
||||
# F4: coerce optional scope filters to str so AnalyticsSnapshot never holds None.
|
||||
scope_remote = (remote or "").strip()
|
||||
scope_org = (org or "").strip()
|
||||
scope_repo = (repo or "").strip()
|
||||
|
||||
try:
|
||||
db = control_plane_db.ControlPlaneDB(db_path=db_path)
|
||||
rows = db.query_usage_events(
|
||||
remote=remote_filter,
|
||||
org=org_filter,
|
||||
repo=repo_filter,
|
||||
project_id=project_id,
|
||||
role=role_filter,
|
||||
model=model_filter,
|
||||
stage=stage_filter,
|
||||
issue_number=issue_number,
|
||||
pr_number=pr_number,
|
||||
limit=limit,
|
||||
)
|
||||
except Exception as exc:
|
||||
empty_summary = aggregate_events("Overall", [])
|
||||
return AnalyticsSnapshot(
|
||||
ok=False,
|
||||
reason=f"control_plane_db_unavailable: {exc}",
|
||||
schema_version=ANALYTICS_SCHEMA_VERSION,
|
||||
remote=scope_remote,
|
||||
org=scope_org,
|
||||
repo=scope_repo,
|
||||
total_events=0,
|
||||
overall_summary=empty_summary,
|
||||
by_project={},
|
||||
by_role={},
|
||||
by_model={},
|
||||
by_work_item={},
|
||||
by_stage={},
|
||||
events=(),
|
||||
)
|
||||
|
||||
parsed_events: list[UsageEvent] = []
|
||||
for r in rows:
|
||||
meta = console_redaction.redact_text(r.get("metadata")) if r.get("metadata") else None
|
||||
parsed_events.append(
|
||||
UsageEvent(
|
||||
usage_id=r["usage_id"],
|
||||
session_id=r.get("session_id"),
|
||||
remote=r.get("remote") or scope_remote,
|
||||
org=r.get("org") or scope_org,
|
||||
repo=r.get("repo") or scope_repo,
|
||||
project_id=r.get("project_id"),
|
||||
role=r.get("role") or "unknown",
|
||||
model=r.get("model") or "unknown",
|
||||
issue_number=r.get("issue_number"),
|
||||
pr_number=r.get("pr_number"),
|
||||
stage=r.get("stage") or "unknown",
|
||||
input_tokens=r.get("input_tokens"),
|
||||
output_tokens=r.get("output_tokens"),
|
||||
total_tokens=r.get("total_tokens"),
|
||||
estimated_cost_usd=r.get("estimated_cost_usd"),
|
||||
latency_ms=r.get("latency_ms"),
|
||||
duration_ms=r.get("duration_ms"),
|
||||
status=r.get("status") or "success",
|
||||
metadata=meta,
|
||||
created_at=r.get("created_at") or "",
|
||||
)
|
||||
)
|
||||
|
||||
overall_summary = aggregate_events("Overall", parsed_events)
|
||||
|
||||
# Group by project
|
||||
groups_by_project: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
key = e.project_id or (f"{e.org}/{e.repo}" if e.org and e.repo else "default")
|
||||
groups_by_project.setdefault(key, []).append(e)
|
||||
by_project = {k: aggregate_events(k, v) for k, v in groups_by_project.items()}
|
||||
|
||||
# Group by role
|
||||
groups_by_role: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
groups_by_role.setdefault(e.role, []).append(e)
|
||||
by_role = {k: aggregate_events(k, v) for k, v in groups_by_role.items()}
|
||||
|
||||
# Group by model
|
||||
groups_by_model: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
groups_by_model.setdefault(e.model, []).append(e)
|
||||
by_model = {k: aggregate_events(k, v) for k, v in groups_by_model.items()}
|
||||
|
||||
# Group by work item
|
||||
groups_by_work_item: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
if e.issue_number:
|
||||
key = f"issue #{e.issue_number}"
|
||||
elif e.pr_number:
|
||||
key = f"pr #{e.pr_number}"
|
||||
else:
|
||||
key = "unlinked"
|
||||
groups_by_work_item.setdefault(key, []).append(e)
|
||||
by_work_item = {k: aggregate_events(k, v) for k, v in groups_by_work_item.items()}
|
||||
|
||||
# Group by stage
|
||||
groups_by_stage: dict[str, list[UsageEvent]] = {}
|
||||
for e in parsed_events:
|
||||
groups_by_stage.setdefault(e.stage, []).append(e)
|
||||
by_stage = {k: aggregate_events(k, v) for k, v in groups_by_stage.items()}
|
||||
|
||||
return AnalyticsSnapshot(
|
||||
ok=True,
|
||||
reason="ok",
|
||||
schema_version=ANALYTICS_SCHEMA_VERSION,
|
||||
remote=scope_remote,
|
||||
org=scope_org,
|
||||
repo=scope_repo,
|
||||
total_events=len(parsed_events),
|
||||
overall_summary=overall_summary,
|
||||
by_project=by_project,
|
||||
by_role=by_role,
|
||||
by_model=by_model,
|
||||
by_work_item=by_work_item,
|
||||
by_stage=by_stage,
|
||||
events=tuple(parsed_events),
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: AnalyticsSnapshot) -> dict[str, Any]:
|
||||
return snapshot.to_dict()
|
||||
@@ -0,0 +1,248 @@
|
||||
"""HTML views for the Model Usage & Performance Analytics console (#651)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
|
||||
from webui.analytics_loader import AnalyticsSnapshot, GroupMetrics, UsageEvent
|
||||
from webui.layout import render_page
|
||||
|
||||
|
||||
def _escape(text: object) -> str:
|
||||
"""HTML-escape dynamic analytics fields (mirrors audit_views / project_views)."""
|
||||
return html.escape(str(text), quote=True)
|
||||
|
||||
|
||||
def _render_badge(text: str, badge_type: str = "muted") -> str:
|
||||
return f'<span class="badge badge-{_escape(badge_type)}">{_escape(text)}</span>'
|
||||
|
||||
|
||||
def _render_group_table(title: str, groups: dict[str, GroupMetrics], key_header: str = "Group") -> str:
|
||||
if not groups:
|
||||
return (
|
||||
f"<h3>{_escape(title)}</h3>"
|
||||
'<div class="card"><p class="muted">No telemetry events recorded for this dimension.</p></div>'
|
||||
)
|
||||
|
||||
rows = []
|
||||
for key, g in sorted(groups.items(), key=lambda x: x[1].total_events, reverse=True):
|
||||
cost_cell = (
|
||||
f'<span class="accent">{_escape(g.display_cost)}</span>'
|
||||
if g.events_with_cost > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
tokens_cell = (
|
||||
_escape(g.display_tokens)
|
||||
if g.events_with_tokens > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
lat_p50 = (
|
||||
_escape(g.display_latency_p50)
|
||||
if g.events_with_latency > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
lat_p90 = (
|
||||
_escape(g.display_latency_p90)
|
||||
if g.events_with_latency > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
dur_avg = (
|
||||
_escape(g.display_duration_avg)
|
||||
if g.events_with_duration > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td><strong>{_escape(key)}</strong></td>"
|
||||
f"<td>{g.total_events}</td>"
|
||||
f"<td>{tokens_cell}</td>"
|
||||
f"<td>{cost_cell}</td>"
|
||||
f"<td>{lat_p50}</td>"
|
||||
f"<td>{lat_p90}</td>"
|
||||
f"<td>{dur_avg}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
|
||||
rows_html = "".join(rows)
|
||||
return f"""
|
||||
<h3>{_escape(title)}</h3>
|
||||
<div class="card" style="overflow-x: auto;">
|
||||
<table class="data-table">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>{_escape(key_header)}</th>
|
||||
<th>Events</th>
|
||||
<th>Total Tokens</th>
|
||||
<th>Est. Cost</th>
|
||||
<th>Latency (p50)</th>
|
||||
<th>Latency (p90)</th>
|
||||
<th>Avg Stage Duration</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{rows_html}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
"""
|
||||
|
||||
|
||||
def _render_events_table(events: tuple[UsageEvent, ...]) -> str:
|
||||
if not events:
|
||||
return (
|
||||
"<h3>Recent Usage & Instrumentation Events</h3>"
|
||||
'<div class="card"><p class="muted">No individual telemetry events recorded yet. Opt-in instrumentation via session logging or authorized POST /api/v1/analytics/usage.</p></div>'
|
||||
)
|
||||
|
||||
rows = []
|
||||
for e in list(events)[-50:]: # Display latest 50
|
||||
if e.issue_number is not None:
|
||||
work_item = f"issue #{e.issue_number}"
|
||||
elif e.pr_number is not None:
|
||||
work_item = f"pr #{e.pr_number}"
|
||||
else:
|
||||
work_item = "unlinked"
|
||||
tokens = (
|
||||
_escape(f"{e.total_tokens:,}")
|
||||
if e.total_tokens is not None
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
cost = (
|
||||
_escape(f"${e.estimated_cost_usd:.4f}")
|
||||
if e.estimated_cost_usd is not None
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
latency = (
|
||||
_escape(f"{e.latency_ms} ms")
|
||||
if e.latency_ms is not None
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
duration = (
|
||||
_escape(f"{e.duration_ms} ms")
|
||||
if e.duration_ms is not None
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
status_badge = _render_badge(
|
||||
e.status, "success" if e.status == "success" else "danger"
|
||||
)
|
||||
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td>#{e.usage_id}</td>"
|
||||
f"<td><small>{_escape(e.created_at)}</small></td>"
|
||||
f"<td><span class=\"badge\">{_escape(e.role)}</span></td>"
|
||||
f"<td><strong>{_escape(e.model)}</strong></td>"
|
||||
f"<td>{_escape(e.stage)}</td>"
|
||||
f"<td>{_escape(work_item)}</td>"
|
||||
f"<td>{tokens}</td>"
|
||||
f"<td>{cost}</td>"
|
||||
f"<td>{latency}</td>"
|
||||
f"<td>{duration}</td>"
|
||||
f"<td>{status_badge}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
|
||||
rows_html = "".join(rows)
|
||||
return f"""
|
||||
<h3>Recent Telemetry Events</h3>
|
||||
<div class="card" style="overflow-x: auto;">
|
||||
<table class="data-table">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>ID</th>
|
||||
<th>Timestamp</th>
|
||||
<th>Role</th>
|
||||
<th>Model</th>
|
||||
<th>Stage</th>
|
||||
<th>Work Item</th>
|
||||
<th>Tokens</th>
|
||||
<th>Cost</th>
|
||||
<th>Latency</th>
|
||||
<th>Duration</th>
|
||||
<th>Status</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{rows_html}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
"""
|
||||
|
||||
|
||||
def render_analytics_page(snapshot: AnalyticsSnapshot) -> str:
|
||||
"""Render the main Model Usage & Performance Analytics console page."""
|
||||
summary = snapshot.overall_summary
|
||||
|
||||
kpi_tokens = (
|
||||
_escape(summary.display_tokens)
|
||||
if summary.events_with_tokens > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
kpi_cost = (
|
||||
_escape(summary.display_cost)
|
||||
if summary.events_with_cost > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
kpi_lat_p50 = (
|
||||
_escape(summary.display_latency_p50)
|
||||
if summary.events_with_latency > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
kpi_dur_avg = (
|
||||
_escape(summary.display_duration_avg)
|
||||
if summary.events_with_duration > 0
|
||||
else _render_badge("Unknown")
|
||||
)
|
||||
|
||||
status_notice = ""
|
||||
if not snapshot.ok:
|
||||
status_notice = (
|
||||
f'<div class="card warning-card"><strong>Degraded Data Source:</strong> '
|
||||
f'{_escape(snapshot.reason)}</div>'
|
||||
)
|
||||
|
||||
body_html = f"""
|
||||
<h2>Model Usage & Performance Analytics (Phase 4)</h2>
|
||||
<p class="muted">
|
||||
Durable console analytics for model usage, token cost, latency percentiles, and workflow-stage performance correlated to issues, PRs, and worker roles.
|
||||
</p>
|
||||
|
||||
{status_notice}
|
||||
|
||||
<div class="notice-card" style="background: rgba(91, 159, 212, 0.1); border: 1px solid var(--border); padding: 0.75rem 1rem; border-radius: 6px; margin-bottom: 1.5rem;">
|
||||
<small><strong>Note on telemetry fidelity:</strong> Missing data or untracked metrics are explicitly labeled as <em>Unknown</em>. No token costs or latency metrics are zero-fabricated.</small>
|
||||
</div>
|
||||
|
||||
<div class="card-grid" style="display: grid; grid-template-columns: repeat(auto-fit, minmax(180px, 1fr)); gap: 1rem; margin-bottom: 1.5rem;">
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Total Events</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{summary.total_events}</h3>
|
||||
</div>
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Total Tokens</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{kpi_tokens}</h3>
|
||||
</div>
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Est. Token Cost</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{kpi_cost}</h3>
|
||||
</div>
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Latency (p50)</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{kpi_lat_p50}</h3>
|
||||
</div>
|
||||
<div class="card">
|
||||
<span class="muted" style="font-size: 0.85rem;">Avg Stage Duration</span>
|
||||
<h3 style="margin: 0.25rem 0 0 0;">{kpi_dur_avg}</h3>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{_render_group_table("Usage & Cost by Model", snapshot.by_model, "Model")}
|
||||
{_render_group_table("Performance by Workflow Stage", snapshot.by_stage, "Stage")}
|
||||
{_render_group_table("Usage & Cost by Role", snapshot.by_role, "Role")}
|
||||
{_render_group_table("Work Item Analytics", snapshot.by_work_item, "Work Item")}
|
||||
{_render_events_table(snapshot.events)}
|
||||
"""
|
||||
|
||||
return render_page(title="Model Usage & Performance Analytics", body_html=body_html)
|
||||
+118
-15
@@ -46,9 +46,13 @@ from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as wo
|
||||
from webui.worktree_views import render_worktrees_page
|
||||
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
|
||||
from webui.runtime_views import render_runtime_page
|
||||
from webui.policy_inventory import load_policy_inventory, snapshot_to_dict as policy_snapshot_to_dict
|
||||
from webui.policy_views import render_policy_page
|
||||
from webui.timeline import load_timeline, snapshot_to_dict as timeline_snapshot_to_dict
|
||||
from webui.analytics_loader import (
|
||||
load_analytics,
|
||||
record_usage,
|
||||
snapshot_to_dict as analytics_snapshot_to_dict,
|
||||
)
|
||||
from webui.analytics_views import render_analytics_page
|
||||
from webui.system_health import (
|
||||
API_PATH as SYSTEM_HEALTH_API_PATH,
|
||||
load_system_health,
|
||||
@@ -304,17 +308,6 @@ async def api_runtime(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(runtime_snapshot_to_dict(load_runtime_snapshot()))
|
||||
|
||||
|
||||
async def policy(_request: Request) -> HTMLResponse:
|
||||
snapshot = load_policy_inventory()
|
||||
return HTMLResponse(
|
||||
render_page(title="Policy", body_html=render_policy_page(snapshot))
|
||||
)
|
||||
|
||||
|
||||
async def api_v1_policy(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(policy_snapshot_to_dict(load_policy_inventory()))
|
||||
|
||||
|
||||
async def _parse_audit_form(request: Request) -> tuple[str, str | None]:
|
||||
if request.method == "GET":
|
||||
return "", None
|
||||
@@ -580,6 +573,114 @@ async def api_v1_timeline(request: Request) -> JSONResponse:
|
||||
return JSONResponse(timeline_snapshot_to_dict(snapshot), status_code=status_code)
|
||||
|
||||
|
||||
async def analytics(request: Request) -> HTMLResponse:
|
||||
"""Read-only model usage, token cost, latency, and performance analytics HTML view (#651)."""
|
||||
snapshot = load_analytics(
|
||||
remote=request.query_params.get("remote"),
|
||||
org=request.query_params.get("org"),
|
||||
repo=request.query_params.get("repo"),
|
||||
role=request.query_params.get("role"),
|
||||
model=request.query_params.get("model"),
|
||||
stage=request.query_params.get("stage"),
|
||||
issue_number=_query_int(request, "issue"),
|
||||
pr_number=_query_int(request, "pr"),
|
||||
limit=_query_int(request, "limit") or 200,
|
||||
)
|
||||
return HTMLResponse(render_analytics_page(snapshot))
|
||||
|
||||
|
||||
async def api_v1_analytics(request: Request) -> JSONResponse:
|
||||
"""Read-only model usage, token cost, latency, and performance analytics API (#651)."""
|
||||
snapshot = load_analytics(
|
||||
remote=request.query_params.get("remote"),
|
||||
org=request.query_params.get("org"),
|
||||
repo=request.query_params.get("repo"),
|
||||
role=request.query_params.get("role"),
|
||||
model=request.query_params.get("model"),
|
||||
stage=request.query_params.get("stage"),
|
||||
issue_number=_query_int(request, "issue"),
|
||||
pr_number=_query_int(request, "pr"),
|
||||
limit=_query_int(request, "limit") or 500,
|
||||
)
|
||||
status_code = 200 if snapshot.ok else 500
|
||||
return JSONResponse(analytics_snapshot_to_dict(snapshot), status_code=status_code)
|
||||
|
||||
|
||||
async def api_v1_analytics_ingest(request: Request) -> JSONResponse:
|
||||
"""Optional session instrumentation ingestion endpoint (#651).
|
||||
|
||||
Fail-closed write: every request is authorized through console_authz
|
||||
(``record_analytics_usage``) before any control-plane DB mutation. Phase 1
|
||||
keeps ``execution_enabled=False`` and denies unauthenticated callers, so
|
||||
this route cannot be used as an unauthenticated write or XSS injection
|
||||
vector (PR #876 F2).
|
||||
"""
|
||||
try:
|
||||
body = await request.json()
|
||||
except Exception:
|
||||
body = {}
|
||||
if not isinstance(body, dict):
|
||||
body = {}
|
||||
|
||||
principal = resolve_principal(headers=dict(request.headers))
|
||||
decision = authorize(
|
||||
"record_analytics_usage", principal, for_execution=True
|
||||
)
|
||||
allowed = bool(decision.allowed and decision.execution_enabled)
|
||||
console_audit.record_event(
|
||||
action_id="record_analytics_usage",
|
||||
result=(
|
||||
console_audit.RESULT_ALLOWED
|
||||
if allowed
|
||||
else console_audit.RESULT_DENIED
|
||||
),
|
||||
decision=decision,
|
||||
principal=principal,
|
||||
target=_audit_target("record_analytics_usage", body),
|
||||
request_id=_request_id(),
|
||||
detail=decision.detail,
|
||||
)
|
||||
authorization = decision.to_dict()
|
||||
if not allowed:
|
||||
return JSONResponse(
|
||||
{
|
||||
"ok": False,
|
||||
"error": "unauthorized",
|
||||
"detail": (
|
||||
"POST /api/v1/analytics/usage requires an authenticated "
|
||||
"principal with record_analytics_usage execution enabled"
|
||||
),
|
||||
"authorization": authorization,
|
||||
},
|
||||
status_code=403,
|
||||
)
|
||||
|
||||
usage_id = record_usage(
|
||||
session_id=body.get("session_id"),
|
||||
remote=body.get("remote", "dadeschools"),
|
||||
org=body.get("org", ""),
|
||||
repo=body.get("repo", ""),
|
||||
project_id=body.get("project_id"),
|
||||
role=body.get("role", "unknown"),
|
||||
model=body.get("model", "unknown"),
|
||||
issue_number=body.get("issue_number") or body.get("issue"),
|
||||
pr_number=body.get("pr_number") or body.get("pr"),
|
||||
stage=body.get("stage", "unknown"),
|
||||
input_tokens=body.get("input_tokens"),
|
||||
output_tokens=body.get("output_tokens"),
|
||||
total_tokens=body.get("total_tokens"),
|
||||
estimated_cost_usd=body.get("estimated_cost_usd"),
|
||||
latency_ms=body.get("latency_ms"),
|
||||
duration_ms=body.get("duration_ms"),
|
||||
status=body.get("status", "success"),
|
||||
metadata=body.get("metadata"),
|
||||
)
|
||||
return JSONResponse(
|
||||
{"ok": True, "usage_id": usage_id, "authorization": authorization},
|
||||
status_code=201,
|
||||
)
|
||||
|
||||
|
||||
async def method_not_allowed(request: Request, _exc: Exception) -> Response:
|
||||
path = request.url.path
|
||||
if path in _AUDIT_MUTATION_PATHS and request.method == "POST":
|
||||
@@ -620,9 +721,11 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/api/prompts", api_prompts, methods=["GET"]),
|
||||
Route("/runtime", runtime, methods=["GET"]),
|
||||
Route("/api/runtime", api_runtime, methods=["GET"]),
|
||||
Route("/policy", policy, methods=["GET"]),
|
||||
Route("/api/v1/policy", api_v1_policy, methods=["GET"]),
|
||||
Route("/api/v1/timeline", api_v1_timeline, methods=["GET"]),
|
||||
Route("/analytics", analytics, methods=["GET"]),
|
||||
Route("/api/analytics", api_v1_analytics, methods=["GET"]),
|
||||
Route("/api/v1/analytics", api_v1_analytics, methods=["GET"]),
|
||||
Route("/api/v1/analytics/usage", api_v1_analytics_ingest, methods=["POST"]),
|
||||
Route("/audit", audit, methods=["GET", "POST"]),
|
||||
Route("/api/audit", api_audit, methods=["GET", "POST"]),
|
||||
Route("/worktrees", worktrees, methods=["GET"]),
|
||||
|
||||
@@ -236,6 +236,19 @@ _ACTION_SPECS: tuple[ConsoleAction, ...] = (
|
||||
phase=3,
|
||||
summary="Remove a remote feature branch.",
|
||||
),
|
||||
# #651 analytics ingest: local control-plane write, not a Gitea mutation.
|
||||
# Phase 2 gated write so Phase 1 (ACTIVE_PHASE=1) fails closed on execution.
|
||||
ConsoleAction(
|
||||
action_id="record_analytics_usage",
|
||||
task_key="record_analytics_usage",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Ingest a model-usage / latency analytics event into the control-plane DB.",
|
||||
),
|
||||
)
|
||||
|
||||
ACTIONS: dict[str, ConsoleAction] = {a.action_id: a for a in _ACTION_SPECS}
|
||||
|
||||
@@ -65,6 +65,7 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
)),
|
||||
NavGroup("Insights", (
|
||||
NavItem("/insights", "Insights", "stub"),
|
||||
NavItem("/analytics", "Analytics"),
|
||||
NavItem("/audit", "Audit"),
|
||||
)),
|
||||
)
|
||||
|
||||
@@ -1,387 +0,0 @@
|
||||
"""Read-only workflow policy and guardrail inventory for the web UI (#646).
|
||||
|
||||
Policy and guardrails live in code, profiles, docs, and skills. An operator
|
||||
cannot *see* the active workflow policy configuration from the console without
|
||||
reading the repository tree. This module projects the major guardrails into a
|
||||
redacted, machine-readable inventory with source attribution (file / module /
|
||||
doc), so the console can render them as HTML tables with source pointers.
|
||||
|
||||
Design constraints (Phase 3, #646):
|
||||
|
||||
- **Read-only projection.** Nothing here edits policy or exposes a toggle that
|
||||
could weaken a gate. It reports what is already enforced elsewhere.
|
||||
- **Source attribution without secrets.** Every guardrail carries pointers to
|
||||
the file/module/doc that owns it. Live values are compact summaries derived
|
||||
from the safe policy accessors that already exist (``rbac_matrix``,
|
||||
``redaction_policy``, ``audit_policy``); raw regex, tokens, and endpoints are
|
||||
never embedded.
|
||||
- **Redact before emit.** ``snapshot_to_dict`` runs the whole payload through
|
||||
``console_redaction.redact_payload`` so a planted or accidental secret in any
|
||||
projected value degrades to the placeholder rather than reaching a client.
|
||||
- **Fail soft.** A projection that raises is recorded as a per-entry error and
|
||||
never takes the page down; a guardrail is still listed with its sources.
|
||||
- **Diff vs documented defaults where feasible.** When a guardrail declares a
|
||||
documented invariant, the active projection is compared against it and the
|
||||
result is reported; otherwise the diff is explicitly ``None`` with a reason.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable
|
||||
|
||||
from webui import console_audit
|
||||
from webui import console_authz
|
||||
from webui import console_redaction
|
||||
|
||||
SCHEMA_VERSION = 1
|
||||
|
||||
READ_ONLY_NOTE = (
|
||||
"Read-only projection of guardrails enforced in code, profiles, docs, and "
|
||||
"skills. This view never edits policy and exposes no gate-weakening toggle."
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SourcePointer:
|
||||
"""Where a guardrail is defined. Attribution only — never a secret."""
|
||||
|
||||
label: str
|
||||
path: str
|
||||
kind: str # "module" | "doc" | "script" | "config"
|
||||
anchor: str | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"label": self.label,
|
||||
"path": self.path,
|
||||
"kind": self.kind,
|
||||
"anchor": self.anchor,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PolicyEntry:
|
||||
key: str
|
||||
title: str
|
||||
category: str
|
||||
summary: str
|
||||
sources: tuple[SourcePointer, ...]
|
||||
active: dict[str, Any] | None
|
||||
documented_default: dict[str, Any] | None
|
||||
diff: dict[str, Any] | None
|
||||
error: str | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"key": self.key,
|
||||
"title": self.title,
|
||||
"category": self.category,
|
||||
"summary": self.summary,
|
||||
"sources": [s.to_dict() for s in self.sources],
|
||||
"active": self.active,
|
||||
"documented_default": self.documented_default,
|
||||
"diff": self.diff,
|
||||
"error": self.error,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PolicyInventorySnapshot:
|
||||
schema_version: int
|
||||
read_only: bool
|
||||
note: str
|
||||
entries: tuple[PolicyEntry, ...]
|
||||
categories: tuple[str, ...]
|
||||
build_errors: tuple[str, ...]
|
||||
|
||||
|
||||
def _diff_active_vs_default(
|
||||
active: dict[str, Any] | None,
|
||||
documented_default: dict[str, Any] | None,
|
||||
) -> dict[str, Any] | None:
|
||||
"""Compare only the keys the documented default declares.
|
||||
|
||||
Returns ``None`` when no documented default is declared (diff not feasible)
|
||||
or when the active projection is unavailable. Otherwise reports, per
|
||||
declared key, whether the active value matches the documented invariant.
|
||||
"""
|
||||
if not documented_default:
|
||||
return None
|
||||
if not active:
|
||||
return {"status": "active_unavailable", "checked": {}}
|
||||
checked: dict[str, Any] = {}
|
||||
matches = True
|
||||
for key, expected in documented_default.items():
|
||||
observed = active.get(key)
|
||||
ok = observed == expected
|
||||
matches = matches and ok
|
||||
checked[key] = {"expected": expected, "observed": observed, "matches": ok}
|
||||
return {
|
||||
"status": "matches_documented_default" if matches else "drift_detected",
|
||||
"checked": checked,
|
||||
}
|
||||
|
||||
|
||||
# ── Live projections (compact, safe, fail-soft) ──────────────────────────────
|
||||
# Each returns a small dict of already-safe machine values. They are module
|
||||
# level so tests can substitute one to prove the redaction pass runs.
|
||||
|
||||
|
||||
def _project_role_separation() -> dict[str, Any]:
|
||||
matrix = console_authz.rbac_matrix()
|
||||
return {
|
||||
"model_version": matrix.get("model_version"),
|
||||
"active_phase": matrix.get("active_phase"),
|
||||
"roles": [r.get("role") for r in matrix.get("roles", [])],
|
||||
"privileged_action_count": len(matrix.get("privileged_actions", [])),
|
||||
"default_decision": matrix.get("default_decision"),
|
||||
"execution_enabled": matrix.get("execution_enabled"),
|
||||
}
|
||||
|
||||
|
||||
def _project_redaction() -> dict[str, Any]:
|
||||
policy = console_redaction.redaction_policy()
|
||||
return {
|
||||
"policy_version": policy.get("policy_version"),
|
||||
"placeholder": policy.get("placeholder"),
|
||||
"applies_to": policy.get("applies_to"),
|
||||
"console_detector_count": len(policy.get("console_rules", [])),
|
||||
"redact_before_persist": policy.get("redact_before_persist"),
|
||||
"failure_mode": policy.get("failure_mode"),
|
||||
}
|
||||
|
||||
|
||||
def _project_audit() -> dict[str, Any]:
|
||||
policy = console_audit.audit_policy()
|
||||
return {
|
||||
"schema_version": policy.get("schema_version"),
|
||||
"required_field_count": len(policy.get("required_fields", [])),
|
||||
"results": policy.get("results"),
|
||||
"retention_defaults_days": policy.get("retention_defaults_days"),
|
||||
"append_only": policy.get("append_only"),
|
||||
"redact_before_persist": policy.get("redact_before_persist"),
|
||||
"enabled": policy.get("enabled"),
|
||||
}
|
||||
|
||||
|
||||
def _static(value: dict[str, Any]) -> Callable[[], dict[str, Any]]:
|
||||
return lambda: dict(value)
|
||||
|
||||
|
||||
# ── Guardrail catalog ────────────────────────────────────────────────────────
|
||||
# One row per major guardrail. ``project`` yields the active value (may raise;
|
||||
# caught per entry). ``documented_default`` drives the feasible diff.
|
||||
|
||||
_CatalogRow = tuple[
|
||||
str,
|
||||
str,
|
||||
str,
|
||||
str,
|
||||
tuple[SourcePointer, ...],
|
||||
Callable[[], dict[str, Any]] | None,
|
||||
dict[str, Any] | None,
|
||||
]
|
||||
|
||||
_CATALOG: tuple[_CatalogRow, ...] = (
|
||||
(
|
||||
"role_separation",
|
||||
"Role separation and RBAC",
|
||||
"role_separation",
|
||||
"Author, reviewer, merger, and reconciler capabilities are disjoint and "
|
||||
"role-exclusive; self-review and self-merge are always blocked. The "
|
||||
"console RBAC model defaults to deny.",
|
||||
(
|
||||
SourcePointer("task capability map", "task_capability_map.py", "module"),
|
||||
SourcePointer("role/namespace gate", "role_namespace_gate.py", "module"),
|
||||
SourcePointer("console RBAC", "webui/console_authz.py", "module"),
|
||||
),
|
||||
_project_role_separation,
|
||||
{"default_decision": "deny", "execution_enabled": False},
|
||||
),
|
||||
(
|
||||
"lease_rules",
|
||||
"Issue and PR lease lifecycle",
|
||||
"lease_rules",
|
||||
"Durable work is claimed through issue locks and control-plane leases "
|
||||
"with freshness, expiry, and dead-session recovery; abandoned or stale "
|
||||
"claims are reclaimed only through the sanctioned recovery path.",
|
||||
(
|
||||
SourcePointer("issue lock store", "issue_lock_store.py", "module"),
|
||||
SourcePointer("branch cleanup guard", "branch_cleanup_guard.py", "module"),
|
||||
SourcePointer("safety model §5", "docs/safety-model.md", "doc", "5-mutation-gating"),
|
||||
),
|
||||
None,
|
||||
None,
|
||||
),
|
||||
(
|
||||
"worktree_rules",
|
||||
"Author worktree binding",
|
||||
"worktree_rules",
|
||||
"Author mutations require a validated worktree under branches/ derived "
|
||||
"from the active issue lock; silent fallback to the stable control "
|
||||
"checkout or master is forbidden (#618).",
|
||||
(
|
||||
SourcePointer("author worktree gate", "author_mutation_worktree.py", "module"),
|
||||
SourcePointer("worktree bootstrap", "scripts/worktree-start", "script"),
|
||||
SourcePointer("workflow scope guard", "workflow_scope_guard.py", "module"),
|
||||
),
|
||||
None,
|
||||
None,
|
||||
),
|
||||
(
|
||||
"merge_confirmation",
|
||||
"Explicit merge confirmation",
|
||||
"merge_confirmation",
|
||||
"A merge fails closed unless the caller passes the exact confirmation "
|
||||
"phrase for that PR; reviewing never implies merging.",
|
||||
(
|
||||
SourcePointer("merge path", "merge_pr.py", "module"),
|
||||
SourcePointer("merge tool gate", "gitea_mcp_server.py", "module"),
|
||||
),
|
||||
_static({"required_confirmation_format": "MERGE PR <n>", "auto_merge": False}),
|
||||
{"auto_merge": False},
|
||||
),
|
||||
(
|
||||
"redaction",
|
||||
"Secret redaction",
|
||||
"redaction",
|
||||
"Every console surface runs the shared gitea_audit pass then console "
|
||||
"patterns before any payload, HTML, log line, or audit record leaves "
|
||||
"the server; unredactable values fail closed to the placeholder.",
|
||||
(
|
||||
SourcePointer("console redaction", "webui/console_redaction.py", "module"),
|
||||
SourcePointer("shared redaction", "gitea_audit.py", "module"),
|
||||
SourcePointer("safety model §3", "docs/safety-model.md", "doc", "3-secret-redaction"),
|
||||
),
|
||||
_project_redaction,
|
||||
{"redact_before_persist": True},
|
||||
),
|
||||
(
|
||||
"contamination",
|
||||
"Contamination containment",
|
||||
"contamination",
|
||||
"A session contaminated by a direct stable-branch push or a manual MCP "
|
||||
"daemon kill is blocked from review, merge, close, and completion "
|
||||
"mutations until cleared (reconciler-exempt).",
|
||||
(
|
||||
SourcePointer("contamination gates", "gitea_mcp_server.py", "module"),
|
||||
SourcePointer("stable-branch audit", "workflow_scope_guard.py", "module"),
|
||||
),
|
||||
None,
|
||||
None,
|
||||
),
|
||||
(
|
||||
"allocator_policy",
|
||||
"Work allocation policy",
|
||||
"allocator_policy",
|
||||
"Workers do not self-select exclusive work; the controller-owned "
|
||||
"allocator ranks the complete queue by priority then PRs-before-issues "
|
||||
"then ascending number, honoring dependency edges and foreign claims.",
|
||||
(
|
||||
SourcePointer("allocator", "gitea_mcp_server.py", "module"),
|
||||
SourcePointer("safety model §5", "docs/safety-model.md", "doc", "5-mutation-gating"),
|
||||
),
|
||||
_static(
|
||||
{
|
||||
"self_select_exclusive_work": False,
|
||||
"ranking": "priority desc, PRs before issues, number asc",
|
||||
"respects_dependency_edges": True,
|
||||
"respects_foreign_claims": True,
|
||||
}
|
||||
),
|
||||
{"self_select_exclusive_work": False},
|
||||
),
|
||||
(
|
||||
"audit_logging",
|
||||
"Audit logging",
|
||||
"audit_logging",
|
||||
"Console intent and authorization outcomes are recorded to an "
|
||||
"append-only, redact-before-persist audit log; MCP mutations are "
|
||||
"recorded by gitea_audit and correlated by request id.",
|
||||
(
|
||||
SourcePointer("console audit", "webui/console_audit.py", "module"),
|
||||
SourcePointer("MCP audit", "gitea_audit.py", "module"),
|
||||
SourcePointer("safety model §1", "docs/safety-model.md", "doc", "1-audit-logging-and-confirmation"),
|
||||
),
|
||||
_project_audit,
|
||||
{"append_only": True, "redact_before_persist": True},
|
||||
),
|
||||
(
|
||||
"mutation_gating",
|
||||
"Mutation gating and master parity",
|
||||
"mutation_gating",
|
||||
"Mutations fail closed while the running server is stale relative to "
|
||||
"master, and every mutation is preceded by identity and capability "
|
||||
"resolution in a fixed pre-flight order.",
|
||||
(
|
||||
SourcePointer("mutation gate", "gitea_mcp_server.py", "module"),
|
||||
SourcePointer("safety model §5", "docs/safety-model.md", "doc", "5-mutation-gating"),
|
||||
),
|
||||
_static(
|
||||
{
|
||||
"stale_runtime_blocks_mutations": True,
|
||||
"preflight_order": "whoami -> resolve_task_capability -> mutation",
|
||||
}
|
||||
),
|
||||
{"stale_runtime_blocks_mutations": True},
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _build_entry(row: _CatalogRow) -> PolicyEntry:
|
||||
key, title, category, summary, sources, project, documented_default = row
|
||||
active: dict[str, Any] | None = None
|
||||
error: str | None = None
|
||||
if project is not None:
|
||||
try:
|
||||
active = project()
|
||||
except Exception as exc: # noqa: BLE001 — fail soft; never take the page down
|
||||
active = None
|
||||
error = f"active projection unavailable: {exc}"
|
||||
diff = _diff_active_vs_default(active, documented_default)
|
||||
return PolicyEntry(
|
||||
key=key,
|
||||
title=title,
|
||||
category=category,
|
||||
summary=summary,
|
||||
sources=sources,
|
||||
active=active,
|
||||
documented_default=documented_default,
|
||||
diff=diff,
|
||||
error=error,
|
||||
)
|
||||
|
||||
|
||||
def load_policy_inventory() -> PolicyInventorySnapshot:
|
||||
"""Build the read-only guardrail inventory. Never raises for one bad entry."""
|
||||
entries: list[PolicyEntry] = []
|
||||
build_errors: list[str] = []
|
||||
for row in _CATALOG:
|
||||
try:
|
||||
entries.append(_build_entry(row))
|
||||
except Exception as exc: # noqa: BLE001 — one row must not break the rest
|
||||
build_errors.append(f"{row[0]}: {exc}")
|
||||
categories = tuple(dict.fromkeys(e.category for e in entries))
|
||||
return PolicyInventorySnapshot(
|
||||
schema_version=SCHEMA_VERSION,
|
||||
read_only=True,
|
||||
note=READ_ONLY_NOTE,
|
||||
entries=tuple(entries),
|
||||
categories=categories,
|
||||
build_errors=tuple(build_errors),
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: PolicyInventorySnapshot) -> dict[str, Any]:
|
||||
"""Serialize the snapshot, redacting the entire payload before it is emitted."""
|
||||
payload = {
|
||||
"schema_version": snapshot.schema_version,
|
||||
"read_only": snapshot.read_only,
|
||||
"note": snapshot.note,
|
||||
"categories": list(snapshot.categories),
|
||||
"entry_count": len(snapshot.entries),
|
||||
"entries": [entry.to_dict() for entry in snapshot.entries],
|
||||
"build_errors": list(snapshot.build_errors),
|
||||
}
|
||||
return console_redaction.redact_payload(payload)
|
||||
@@ -1,104 +0,0 @@
|
||||
"""HTML views for the workflow policy and guardrail inventory (#646)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import json
|
||||
|
||||
from webui.policy_inventory import PolicyEntry, PolicyInventorySnapshot
|
||||
|
||||
|
||||
def _source_pointer(source) -> str:
|
||||
path = source.path
|
||||
if source.anchor:
|
||||
path = f"{path}#{source.anchor}"
|
||||
return (
|
||||
f"<li>{html.escape(source.label)} — "
|
||||
f"<code>{html.escape(path)}</code> "
|
||||
f"<span class='muted'>({html.escape(source.kind)})</span></li>"
|
||||
)
|
||||
|
||||
|
||||
def _active_block(entry: PolicyEntry) -> str:
|
||||
if entry.error:
|
||||
return (
|
||||
"<p class='muted'><strong>Active value unavailable:</strong> "
|
||||
f"{html.escape(entry.error)}</p>"
|
||||
)
|
||||
if not entry.active:
|
||||
return "<p class='muted'>No live projection for this guardrail.</p>"
|
||||
pretty = json.dumps(entry.active, indent=2, sort_keys=True, default=str)
|
||||
return f"<pre class='prompt-text'>{html.escape(pretty)}</pre>"
|
||||
|
||||
|
||||
def _diff_block(entry: PolicyEntry) -> str:
|
||||
if entry.diff is None:
|
||||
if entry.documented_default is None:
|
||||
return "<p class='muted'>Diff vs documented default: not feasible (no declared default).</p>"
|
||||
return "<p class='muted'>Diff vs documented default: unavailable.</p>"
|
||||
status = entry.diff.get("status", "unknown")
|
||||
badge = "badge-claimed" if status == "matches_documented_default" else "badge-blocked"
|
||||
rows = []
|
||||
for key, cell in (entry.diff.get("checked") or {}).items():
|
||||
marker = "✓" if cell.get("matches") else "✗"
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td><code>{html.escape(str(key))}</code></td>"
|
||||
f"<td><code>{html.escape(str(cell.get('expected')))}</code></td>"
|
||||
f"<td><code>{html.escape(str(cell.get('observed')))}</code></td>"
|
||||
f"<td>{marker}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
table = ""
|
||||
if rows:
|
||||
table = (
|
||||
"<table class='detail'><thead><tr>"
|
||||
"<th>Key</th><th>Documented</th><th>Active</th><th>Match</th>"
|
||||
"</tr></thead><tbody>"
|
||||
f"{''.join(rows)}</tbody></table>"
|
||||
)
|
||||
return (
|
||||
f"<p class='meta'>Diff vs documented default: "
|
||||
f"<span class='badge {badge}'>{html.escape(status)}</span></p>"
|
||||
f"{table}"
|
||||
)
|
||||
|
||||
|
||||
def _entry_card(entry: PolicyEntry) -> str:
|
||||
sources = "".join(_source_pointer(s) for s in entry.sources)
|
||||
return (
|
||||
"<div class='prompt-card'>"
|
||||
f"<h3>{html.escape(entry.title)} "
|
||||
f"<span class='badge'>{html.escape(entry.category)}</span></h3>"
|
||||
f"<p>{html.escape(entry.summary)}</p>"
|
||||
"<p class='meta'><strong>Source pointers</strong></p>"
|
||||
f"<ul>{sources}</ul>"
|
||||
"<p class='meta'><strong>Active configuration</strong></p>"
|
||||
f"{_active_block(entry)}"
|
||||
f"{_diff_block(entry)}"
|
||||
"</div>"
|
||||
)
|
||||
|
||||
|
||||
def render_policy_page(snapshot: PolicyInventorySnapshot) -> str:
|
||||
categories = ", ".join(html.escape(c) for c in snapshot.categories) or "none"
|
||||
cards = "".join(_entry_card(e) for e in snapshot.entries)
|
||||
build_errors = ""
|
||||
if snapshot.build_errors:
|
||||
items = "".join(
|
||||
f"<li>{html.escape(err)}</li>" for err in snapshot.build_errors
|
||||
)
|
||||
build_errors = (
|
||||
"<div class='stub'><p><strong>Some guardrails could not be built:"
|
||||
f"</strong></p><ul>{items}</ul></div>"
|
||||
)
|
||||
return (
|
||||
"<h2>Workflow policy & guardrails</h2>"
|
||||
f"<p class='muted'>{html.escape(snapshot.note)}</p>"
|
||||
f"<p class='meta'>Schema v{snapshot.schema_version} · "
|
||||
f"{len(snapshot.entries)} guardrails · categories: {categories}</p>"
|
||||
f"{build_errors}"
|
||||
f"{cards}"
|
||||
"<p class='muted'>This page is read-only. It reports enforced policy "
|
||||
"and never edits or weakens a gate. Secret values are redacted.</p>"
|
||||
)
|
||||
Reference in New Issue
Block a user