Merge branch 'master' of https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools into feat/issue-636-inventory-api
# Conflicts: # docs/webui-local-dev.md # webui/app.py
This commit is contained in:
+268
-11
@@ -2,6 +2,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import uuid
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from starlette.applications import Starlette
|
||||
@@ -11,6 +12,7 @@ from starlette.routing import Route
|
||||
|
||||
from webui.deployment_boundary import deployment_snapshot
|
||||
from webui.layout import render_page
|
||||
from webui.nav import NAV_GROUPS, STUB_PAGES
|
||||
from webui.project_registry import (
|
||||
ProjectRegistry,
|
||||
RegistryError,
|
||||
@@ -31,6 +33,9 @@ from final_report_validator import FINAL_REPORT_TASK_KINDS
|
||||
|
||||
from webui.gated_actions import attempt_action, load_action_registry, preview_action
|
||||
from webui.gated_action_views import render_actions_page
|
||||
from webui import console_audit
|
||||
from webui.console_authz import authorize, rbac_matrix, resolve_principal
|
||||
from webui.console_redaction import redaction_policy
|
||||
from webui.audit_validator import audit_report, audit_to_dict
|
||||
from webui.audit_views import render_audit_page
|
||||
from webui.lease_loader import load_lease_snapshot, snapshot_to_dict as lease_snapshot_to_dict
|
||||
@@ -46,6 +51,13 @@ from webui.inventory import (
|
||||
load_inventory_snapshot,
|
||||
snapshot_to_dict as inventory_snapshot_to_dict,
|
||||
)
|
||||
from webui.timeline import load_timeline, snapshot_to_dict as timeline_snapshot_to_dict
|
||||
from webui.system_health import (
|
||||
API_PATH as SYSTEM_HEALTH_API_PATH,
|
||||
load_system_health,
|
||||
process_uptime,
|
||||
snapshot_to_dict as system_health_to_dict,
|
||||
)
|
||||
|
||||
_READ_ONLY_METHODS = frozenset({"GET", "HEAD", "OPTIONS"})
|
||||
_AUDIT_MUTATION_PATHS = frozenset({"/audit", "/api/audit"})
|
||||
@@ -60,35 +72,100 @@ def _stub_page(title: str, description: str) -> HTMLResponse:
|
||||
return HTMLResponse(render_page(title=title, body_html=body))
|
||||
|
||||
|
||||
_LEGACY_PAGES = (
|
||||
("/queue", "Queue", "live PR and issue dashboard (#429)"),
|
||||
("/projects", "Projects", "registry and onboarding (#427)"),
|
||||
("/prompts", "Prompts", "canonical workflow prompt library (#428)"),
|
||||
("/runtime", "Runtime", "MCP health and stale-runtime detection (#430)"),
|
||||
("/audit", "Audit", "final-report paste and validator preview (#431)"),
|
||||
("/worktrees", "Worktrees", "branch hygiene dashboard (#432)"),
|
||||
("/leases", "Leases", "collision and lease visibility (#433)"),
|
||||
("/actions", "Actions", "gated write-action framework (#434)"),
|
||||
)
|
||||
|
||||
|
||||
def _render_home_nav_groups() -> str:
|
||||
groups = []
|
||||
for group in NAV_GROUPS:
|
||||
items = "".join(
|
||||
f'<li><a href="{item.href}">{item.label}</a>'
|
||||
+ ("" if item.status == "live" else " <span class=\"muted\">(stub)</span>")
|
||||
+ "</li>"
|
||||
for item in group.items
|
||||
)
|
||||
groups.append(f"<h3>{group.label}</h3><ul>{items}</ul>")
|
||||
return "".join(groups)
|
||||
|
||||
|
||||
async def home(_request: Request) -> HTMLResponse:
|
||||
legacy = "".join(
|
||||
f"<li><strong>{label}</strong> — {desc} "
|
||||
f'(<a href="{href}">{href}</a>)</li>'
|
||||
for href, label, desc in _LEGACY_PAGES
|
||||
)
|
||||
body = (
|
||||
"<h2>Operator console</h2>"
|
||||
"<p>Local entry point for MCP Control Plane operational views.</p>"
|
||||
"<ul>"
|
||||
"<li><strong>Queue</strong> — live PR and issue dashboard (#429)</li>"
|
||||
"<li><strong>Projects</strong> — registry and onboarding (#427)</li>"
|
||||
"<li><strong>Prompts</strong> — canonical workflow prompt library (#428)</li>"
|
||||
"<li><strong>Runtime</strong> — MCP health and stale-runtime detection (#430)</li>"
|
||||
"<li><strong>Audit</strong> — final-report paste and validator preview (#431)</li>"
|
||||
"<li><strong>Worktrees</strong> — branch hygiene dashboard (#432)</li>"
|
||||
"<li><strong>Leases</strong> — collision and lease visibility (#433)</li>"
|
||||
"<li><strong>Actions</strong> — gated write-action framework (#434)</li>"
|
||||
"</ul>"
|
||||
"<p>Read-only home for the MCP Control Plane Phase 1 operator console. "
|
||||
"Gitea, MCP capability gates, and canonical workflows remain the source "
|
||||
"of truth; this console never mutates them.</p>"
|
||||
"<h2>Phase 1 surfaces</h2>"
|
||||
+ _render_home_nav_groups()
|
||||
+ "<h2>MVP legacy pages</h2>"
|
||||
"<ul>" + legacy + "</ul>"
|
||||
)
|
||||
return HTMLResponse(render_page(title="Home", body_html=body))
|
||||
|
||||
|
||||
async def phase_stub(request: Request) -> HTMLResponse:
|
||||
"""Graceful read-only placeholder for a not-yet-implemented Phase 1 surface."""
|
||||
title, description = STUB_PAGES[request.url.path]
|
||||
body = (
|
||||
f"<h2>{title}</h2>"
|
||||
f'<div class="stub"><p>{description}</p>'
|
||||
"<p>Phase 1 shell placeholder — no write actions. Tracked under "
|
||||
"epic #631.</p></div>"
|
||||
)
|
||||
return HTMLResponse(render_page(title=title, body_html=body))
|
||||
|
||||
|
||||
async def health(_request: Request) -> JSONResponse:
|
||||
"""Liveness only — deliberately cheap, runs no dependency probe (#634).
|
||||
|
||||
Every MVP key is retained so existing pollers keep working; the additions
|
||||
are a pointer to the structured API and the in-memory process uptime.
|
||||
Readiness lives at that API because answering it costs real probes.
|
||||
"""
|
||||
bind_host = _request.app.state.webui_bind_host
|
||||
started_at, uptime_seconds = process_uptime()
|
||||
return JSONResponse({
|
||||
"status": "ok",
|
||||
"service": "mcp-control-plane-webui",
|
||||
"mode": "read-only-mvp",
|
||||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||
"deployment": deployment_snapshot(bind_host=bind_host),
|
||||
"started_at": started_at,
|
||||
"uptime_seconds": uptime_seconds,
|
||||
"system_health_api": SYSTEM_HEALTH_API_PATH,
|
||||
})
|
||||
|
||||
|
||||
def _truthy_flag(value: str | None) -> bool:
|
||||
return (value or "").strip().lower() in {"1", "true", "yes", "on"}
|
||||
|
||||
|
||||
async def api_system_health(request: Request) -> JSONResponse:
|
||||
"""Structured read-only system health (#634).
|
||||
|
||||
`?deep=1` opts into the expensive network probe. The response status code
|
||||
reflects readiness so automated checks can branch on it without parsing the
|
||||
body: 200 when ready, 503 when a required dependency failed or never ran.
|
||||
"""
|
||||
deep = _truthy_flag(request.query_params.get("deep"))
|
||||
snapshot = load_system_health(deep=deep)
|
||||
payload = system_health_to_dict(snapshot)
|
||||
return JSONResponse(payload, status_code=200 if snapshot.ready else 503)
|
||||
|
||||
|
||||
async def queue(_request: Request) -> HTMLResponse:
|
||||
snapshot = load_queue_snapshot()
|
||||
return HTMLResponse(render_page(title="Queue", body_html=render_queue_page(snapshot)))
|
||||
@@ -281,6 +358,49 @@ async def api_actions(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(load_action_registry().to_dict())
|
||||
|
||||
|
||||
def _request_id() -> str:
|
||||
return f"req-{uuid.uuid4().hex}"
|
||||
|
||||
|
||||
def _audit_target(action_id: str, params: dict[str, object]) -> dict[str, object]:
|
||||
"""Describe the action target for the audit record (never secrets)."""
|
||||
if "pr_number" in params:
|
||||
return {"kind": "pr", "ref": f"#{params['pr_number']}"}
|
||||
if "issue_number" in params:
|
||||
return {"kind": "issue", "ref": f"#{params['issue_number']}"}
|
||||
if "branch_name" in params:
|
||||
return {"kind": "branch", "ref": str(params["branch_name"])}
|
||||
return {"kind": "unspecified", "ref": action_id}
|
||||
|
||||
|
||||
def _authorize_request(
|
||||
request: Request,
|
||||
action_id: str,
|
||||
params: dict[str, object],
|
||||
*,
|
||||
for_execution: bool,
|
||||
result: str,
|
||||
) -> dict[str, object]:
|
||||
"""Resolve principal, decide, and audit. Returns the decision payload.
|
||||
|
||||
Phase 1 records the decision rather than enforcing it as the terminal
|
||||
outcome: ``webui.gated_actions`` already fails closed for every action, so
|
||||
this layer cannot loosen anything. Phase 2 enforces on this same decision.
|
||||
"""
|
||||
principal = resolve_principal(headers=dict(request.headers))
|
||||
decision = authorize(action_id, principal, for_execution=for_execution)
|
||||
console_audit.record_event(
|
||||
action_id=action_id,
|
||||
result=result,
|
||||
decision=decision,
|
||||
principal=principal,
|
||||
target=_audit_target(action_id, params),
|
||||
request_id=_request_id(),
|
||||
detail=decision.detail,
|
||||
)
|
||||
return decision.to_dict()
|
||||
|
||||
|
||||
async def api_action_preview(request: Request) -> JSONResponse:
|
||||
action_id = request.path_params["action_id"]
|
||||
params = dict(request.query_params)
|
||||
@@ -290,6 +410,13 @@ async def api_action_preview(request: Request) -> JSONResponse:
|
||||
result = preview_action(action_id, **params)
|
||||
if "error" in result:
|
||||
return JSONResponse(result, status_code=404)
|
||||
result["authorization"] = _authorize_request(
|
||||
request,
|
||||
action_id,
|
||||
params,
|
||||
for_execution=False,
|
||||
result=console_audit.RESULT_PREVIEWED,
|
||||
)
|
||||
return JSONResponse(result)
|
||||
|
||||
|
||||
@@ -303,6 +430,18 @@ async def api_action_attempt(request: Request) -> JSONResponse:
|
||||
if not isinstance(body, dict):
|
||||
body = {}
|
||||
result = attempt_action(action_id, **body)
|
||||
authorization = _authorize_request(
|
||||
request,
|
||||
action_id,
|
||||
body,
|
||||
for_execution=True,
|
||||
result=(
|
||||
console_audit.RESULT_DENIED
|
||||
if not result.get("success")
|
||||
else console_audit.RESULT_ALLOWED
|
||||
),
|
||||
)
|
||||
result["authorization"] = authorization
|
||||
status = 403 if not result.get("success") else 200
|
||||
return JSONResponse(result, status_code=status)
|
||||
|
||||
@@ -331,6 +470,113 @@ async def api_inventory_section(request: Request) -> JSONResponse:
|
||||
return JSONResponse(payload)
|
||||
|
||||
|
||||
async def api_console_security_model(_request: Request) -> JSONResponse:
|
||||
"""Read-only publication of the #633 authorization/redaction/audit model."""
|
||||
return JSONResponse({
|
||||
"rbac": rbac_matrix(),
|
||||
"redaction": redaction_policy(),
|
||||
"audit": console_audit.audit_policy(),
|
||||
})
|
||||
|
||||
|
||||
def _query_int(request: Request, key: str) -> int | None:
|
||||
"""Parse an optional integer query parameter; None when absent/invalid."""
|
||||
raw = request.query_params.get(key)
|
||||
if raw is None or not str(raw).strip():
|
||||
return None
|
||||
try:
|
||||
return int(str(raw).strip())
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _derive_remote(host: str) -> str:
|
||||
"""Map a Gitea host to its known short remote name (control-plane scope key)."""
|
||||
text = (host or "").lower()
|
||||
if "prgs" in text:
|
||||
return "prgs"
|
||||
if "dadeschools" in text:
|
||||
return "dadeschools"
|
||||
return text.split(".")[0] if text else ""
|
||||
|
||||
|
||||
def _timeline_comment_source(host: str, org: str, repo: str):
|
||||
"""Build a fail-soft CTH-comment fetcher for one repo, or None when offline.
|
||||
|
||||
Returns a callable ``(kind, number) -> list[comment]``. Credentials or
|
||||
network failures raise inside the callable so ``load_timeline`` degrades the
|
||||
handoff source rather than the whole timeline. Offline test mode yields no
|
||||
live source so the handoff section reports ``not run``.
|
||||
"""
|
||||
import os
|
||||
|
||||
from gitea_auth import api_fetch_page, get_auth_header, repo_api_url
|
||||
|
||||
offline = (os.environ.get("WEBUI_TEST_OFFLINE") or "").strip().lower() in {"1", "true", "yes"}
|
||||
if offline:
|
||||
return None
|
||||
auth = get_auth_header(host)
|
||||
if not auth:
|
||||
return None
|
||||
|
||||
def _fetch(kind: str, number: int) -> list:
|
||||
segment = "pulls" if kind == "pr" else "issues"
|
||||
url = f"{repo_api_url(host, org, repo)}/{segment}/{int(number)}/comments"
|
||||
comments: list = []
|
||||
page = 1
|
||||
while page <= 20:
|
||||
raw, meta = api_fetch_page(url, auth, page=page, limit=50)
|
||||
comments.extend(raw)
|
||||
if bool(meta["is_final_page"]):
|
||||
break
|
||||
page += 1
|
||||
return comments
|
||||
|
||||
return _fetch
|
||||
|
||||
|
||||
async def api_v1_timeline(request: Request) -> JSONResponse:
|
||||
"""Read-only workflow-event timeline (#637). Filter by issue/PR/session."""
|
||||
from webui.queue_loader import _host_from_url # host normalisation helper
|
||||
|
||||
registry, error = _load_project_registry()
|
||||
if error is not None:
|
||||
return JSONResponse(error.to_dict(), status_code=500)
|
||||
project = registry.projects[0] if registry.projects else None
|
||||
|
||||
org = request.query_params.get("org") or (project.gitea_owner if project else "")
|
||||
repo = request.query_params.get("repo") or (project.repo_name if project else "")
|
||||
host = _host_from_url(project.remote_host) if project else ""
|
||||
remote = request.query_params.get("remote") or _derive_remote(host)
|
||||
|
||||
if not (remote and org and repo):
|
||||
return JSONResponse(
|
||||
{
|
||||
"error": "timeline_scope_unresolved",
|
||||
"detail": "no project in registry and no remote/org/repo query params provided",
|
||||
},
|
||||
status_code=400,
|
||||
)
|
||||
|
||||
comment_source = _timeline_comment_source(host, org, repo) if (host and org and repo) else None
|
||||
|
||||
snapshot = load_timeline(
|
||||
remote=remote,
|
||||
org=org,
|
||||
repo=repo,
|
||||
issue=_query_int(request, "issue"),
|
||||
pr=_query_int(request, "pr"),
|
||||
session=(request.query_params.get("session") or None),
|
||||
limit=_query_int(request, "limit"),
|
||||
offset=_query_int(request, "offset"),
|
||||
comment_source=comment_source,
|
||||
)
|
||||
# A filter no surviving source can carry is refused, not answered empty:
|
||||
# a 200 with zero events would tell the operator no such activity exists.
|
||||
status_code = 200 if snapshot.ok else 422
|
||||
return JSONResponse(timeline_snapshot_to_dict(snapshot), status_code=status_code)
|
||||
|
||||
|
||||
async def method_not_allowed(request: Request, _exc: Exception) -> Response:
|
||||
path = request.url.path
|
||||
if path in _AUDIT_MUTATION_PATHS and request.method == "POST":
|
||||
@@ -353,6 +599,7 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
routes=[
|
||||
Route("/", home, methods=["GET"]),
|
||||
Route("/health", health, methods=["GET"]),
|
||||
Route(SYSTEM_HEALTH_API_PATH, api_system_health, methods=["GET"]),
|
||||
Route("/queue", queue, methods=["GET"]),
|
||||
Route("/api/queue", api_queue, methods=["GET"]),
|
||||
Route("/projects", projects, methods=["GET"]),
|
||||
@@ -369,6 +616,7 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/api/prompts", api_prompts, methods=["GET"]),
|
||||
Route("/runtime", runtime, methods=["GET"]),
|
||||
Route("/api/runtime", api_runtime, methods=["GET"]),
|
||||
Route("/api/v1/timeline", api_v1_timeline, methods=["GET"]),
|
||||
Route("/audit", audit, methods=["GET", "POST"]),
|
||||
Route("/api/audit", api_audit, methods=["GET", "POST"]),
|
||||
Route("/worktrees", worktrees, methods=["GET"]),
|
||||
@@ -393,6 +641,15 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
api_inventory_section,
|
||||
methods=["GET"],
|
||||
),
|
||||
Route(
|
||||
"/api/console/security-model",
|
||||
api_console_security_model,
|
||||
methods=["GET"],
|
||||
),
|
||||
*[
|
||||
Route(path, phase_stub, methods=["GET"])
|
||||
for path in STUB_PAGES
|
||||
],
|
||||
],
|
||||
exception_handlers={405: method_not_allowed},
|
||||
)
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
"""Console audit event schema, retention, and append-only sink (#633).
|
||||
|
||||
``gitea_audit`` records MCP-side *mutations*: which profile and Gitea user
|
||||
performed which tool call. It carries no console actor, no identity source, no
|
||||
correlation identifier, and no retention class, so it cannot answer the
|
||||
question #633 exists to answer — *who sat at the console, what did they
|
||||
attempt, and was it authorized?* An authorization denial is not a mutation and
|
||||
would never appear there at all.
|
||||
|
||||
This module adds the console-side record. It does not replace ``gitea_audit``:
|
||||
when a Phase 2 action eventually reaches MCP, both fire, correlated by
|
||||
``correlation.request_id``.
|
||||
|
||||
Design constraints:
|
||||
|
||||
- **Redact before persist.** Every record passes through
|
||||
``webui.console_redaction.redact_payload`` before serialization, so an
|
||||
unredacted field is never durable.
|
||||
- **Append-only.** Records are appended as JSON lines. Nothing here updates or
|
||||
deletes; retention is metadata on each record, enforced by an operator-run
|
||||
policy, never by silent rewriting.
|
||||
- **Never raises.** Auditing must not break the request it describes. A failed
|
||||
write returns ``False``.
|
||||
- **Off by default.** With ``WEBUI_CONSOLE_AUDIT_LOG`` unset, events are still
|
||||
*built* (so callers and tests see the schema) but nothing is written.
|
||||
|
||||
A record looks like this (synthetic values):
|
||||
|
||||
{"schema_version": 1, "event_id": "evt-0001",
|
||||
"timestamp": "2026-07-22T10:16:42+00:00",
|
||||
"actor": {"subject": "[email protected]", "role": "operator",
|
||||
"identity_source": "access_proxy", "authenticated": true},
|
||||
"action": "merge_pr", "action_class": "privileged",
|
||||
"target": {"kind": "pr", "ref": "#123"},
|
||||
"result": "denied", "reason_code": "insufficient_role",
|
||||
"correlation": {"request_id": "req-abc", "session_id": null,
|
||||
"mcp_task": "merge_pr", "mcp_permission": "gitea.pr.merge"},
|
||||
"retention": {"class": "privileged", "days": 365,
|
||||
"expires_at": "2027-07-22T10:16:42+00:00"},
|
||||
"redacted": true}
|
||||
|
||||
Timestamps are timezone-aware ISO-8601 in UTC.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import datetime
|
||||
import json
|
||||
import os
|
||||
import uuid
|
||||
from typing import Any
|
||||
|
||||
from webui import console_authz
|
||||
from webui.console_redaction import redact_payload, scan_for_secrets
|
||||
|
||||
SCHEMA_VERSION = 1
|
||||
|
||||
AUDIT_LOG_ENV = "WEBUI_CONSOLE_AUDIT_LOG"
|
||||
|
||||
# Result vocabulary. ``denied`` is the one ``gitea_audit`` has no equivalent
|
||||
# for: an authorization refusal never reaches the MCP layer.
|
||||
RESULT_ALLOWED = "allowed"
|
||||
RESULT_DENIED = "denied"
|
||||
RESULT_PREVIEWED = "previewed"
|
||||
RESULT_FAILED = "failed"
|
||||
RESULT_SUCCEEDED = "succeeded"
|
||||
|
||||
RESULTS = frozenset(
|
||||
{
|
||||
RESULT_ALLOWED,
|
||||
RESULT_DENIED,
|
||||
RESULT_PREVIEWED,
|
||||
RESULT_FAILED,
|
||||
RESULT_SUCCEEDED,
|
||||
}
|
||||
)
|
||||
|
||||
# Retention classes and default lifetimes in days. Privileged and break-glass
|
||||
# records outlive routine ones because they are what an incident review needs.
|
||||
RETENTION_STANDARD = "standard"
|
||||
RETENTION_PRIVILEGED = "privileged"
|
||||
RETENTION_BREAK_GLASS = "break_glass"
|
||||
|
||||
RETENTION_DAYS: dict[str, int] = {
|
||||
RETENTION_STANDARD: 90,
|
||||
RETENTION_PRIVILEGED: 365,
|
||||
RETENTION_BREAK_GLASS: 730,
|
||||
}
|
||||
|
||||
# Fields every record must carry. Asserted by the test suite so a future edit
|
||||
# cannot quietly drop one.
|
||||
REQUIRED_FIELDS: tuple[str, ...] = (
|
||||
"schema_version",
|
||||
"event_id",
|
||||
"timestamp",
|
||||
"actor",
|
||||
"action",
|
||||
"action_class",
|
||||
"target",
|
||||
"result",
|
||||
"reason_code",
|
||||
"correlation",
|
||||
"retention",
|
||||
"redacted",
|
||||
)
|
||||
|
||||
REQUIRED_ACTOR_FIELDS: tuple[str, ...] = (
|
||||
"subject",
|
||||
"role",
|
||||
"identity_source",
|
||||
"authenticated",
|
||||
)
|
||||
|
||||
REQUIRED_CORRELATION_FIELDS: tuple[str, ...] = (
|
||||
"request_id",
|
||||
"session_id",
|
||||
"mcp_task",
|
||||
"mcp_permission",
|
||||
)
|
||||
|
||||
|
||||
def audit_log_path() -> str | None:
|
||||
"""Configured sink path, or ``None`` when console auditing is off."""
|
||||
return (os.environ.get(AUDIT_LOG_ENV) or "").strip() or None
|
||||
|
||||
|
||||
def audit_enabled() -> bool:
|
||||
return audit_log_path() is not None
|
||||
|
||||
|
||||
def retention_class_for(action: console_authz.ConsoleAction | None) -> str:
|
||||
"""Classify retention from the action, defaulting to the longest-lived.
|
||||
|
||||
An unknown action is treated as privileged rather than standard: for a
|
||||
safety control the conservative direction is to keep the record longer.
|
||||
"""
|
||||
if action is None:
|
||||
return RETENTION_PRIVILEGED
|
||||
if action.break_glass:
|
||||
return RETENTION_BREAK_GLASS
|
||||
if action.privileged:
|
||||
return RETENTION_PRIVILEGED
|
||||
return RETENTION_STANDARD
|
||||
|
||||
|
||||
def _retention_block(
|
||||
retention_class: str, now: datetime.datetime
|
||||
) -> dict[str, Any]:
|
||||
days = RETENTION_DAYS.get(
|
||||
retention_class, RETENTION_DAYS[RETENTION_PRIVILEGED]
|
||||
)
|
||||
return {
|
||||
"class": retention_class,
|
||||
"days": days,
|
||||
"expires_at": (now + datetime.timedelta(days=days)).isoformat(),
|
||||
}
|
||||
|
||||
|
||||
def build_event(
|
||||
*,
|
||||
action_id: str,
|
||||
result: str,
|
||||
decision: console_authz.AuthorizationDecision | None = None,
|
||||
principal: console_authz.Principal | None = None,
|
||||
target: dict[str, Any] | None = None,
|
||||
reason_code: str | None = None,
|
||||
request_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
detail: str | None = None,
|
||||
metadata: dict[str, Any] | None = None,
|
||||
now: datetime.datetime | None = None,
|
||||
event_id: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build one redacted, JSON-able console audit record.
|
||||
|
||||
Redaction runs here rather than at write time so an in-memory record handed
|
||||
to a template or an API response is already clean.
|
||||
"""
|
||||
ts = now or datetime.datetime.now(datetime.timezone.utc)
|
||||
action = console_authz.get_action(action_id)
|
||||
who = principal or (
|
||||
decision.principal if decision else console_authz.ANONYMOUS
|
||||
)
|
||||
resolved_result = result if result in RESULTS else RESULT_FAILED
|
||||
resolved_reason = reason_code or (
|
||||
decision.reason_code if decision else "unspecified"
|
||||
)
|
||||
retention_class = retention_class_for(action)
|
||||
|
||||
event: dict[str, Any] = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"event_id": event_id or f"evt-{uuid.uuid4().hex}",
|
||||
"timestamp": ts.isoformat(),
|
||||
"actor": who.to_dict(),
|
||||
"action": action_id,
|
||||
"action_class": action.action_class if action else "unknown",
|
||||
"target": dict(target or {}),
|
||||
"result": resolved_result,
|
||||
"reason_code": resolved_reason,
|
||||
"correlation": {
|
||||
"request_id": request_id,
|
||||
"session_id": session_id,
|
||||
"mcp_task": action.task_key if action else None,
|
||||
"mcp_permission": action.mcp_permission if action else None,
|
||||
},
|
||||
"retention": _retention_block(retention_class, ts),
|
||||
"redacted": True,
|
||||
"detail": detail,
|
||||
"metadata": dict(metadata or {}),
|
||||
}
|
||||
if decision is not None:
|
||||
# Deliberately *not* named "authorization": ``gitea_audit`` treats that
|
||||
# substring as a secret key hint (it matches the HTTP Authorization
|
||||
# header) and would replace this whole block with the placeholder.
|
||||
event["decision"] = {
|
||||
"allowed": decision.allowed,
|
||||
"required_role": decision.required_role,
|
||||
"requires_confirmation": decision.requires_confirmation,
|
||||
"dual_control": decision.dual_control,
|
||||
"break_glass": decision.break_glass,
|
||||
"execution_enabled": decision.execution_enabled,
|
||||
}
|
||||
|
||||
redacted = redact_payload(event)
|
||||
if not isinstance(redacted, dict): # pragma: no cover - defensive
|
||||
return {"schema_version": SCHEMA_VERSION, "redacted": True}
|
||||
return redacted
|
||||
|
||||
|
||||
def write_event(event: dict[str, Any], path: str | None = None) -> bool:
|
||||
"""Append *event* as one JSON line. Never raises.
|
||||
|
||||
Returns ``True`` when a line was written, ``False`` when auditing is off or
|
||||
the write failed. A record that still trips a secret detector is dropped
|
||||
rather than persisted.
|
||||
"""
|
||||
sink = path or audit_log_path()
|
||||
if not sink:
|
||||
return False
|
||||
try:
|
||||
if scan_for_secrets(event):
|
||||
return False
|
||||
line = json.dumps(event, default=str, sort_keys=True)
|
||||
with open(sink, "a", encoding="utf-8") as handle:
|
||||
handle.write(line + "\n")
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def record_event(**kwargs: Any) -> dict[str, Any]:
|
||||
"""Build and persist one record; return the record either way.
|
||||
|
||||
Callers get the record back so it can be surfaced in a response or a test
|
||||
regardless of whether a sink is configured.
|
||||
"""
|
||||
event = build_event(**kwargs)
|
||||
written = write_event(event)
|
||||
return {"event": event, "written": written}
|
||||
|
||||
|
||||
def audit_policy() -> dict[str, Any]:
|
||||
"""Machine-readable audit schema and retention defaults (never secrets)."""
|
||||
return {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"required_fields": list(REQUIRED_FIELDS),
|
||||
"required_actor_fields": list(REQUIRED_ACTOR_FIELDS),
|
||||
"required_correlation_fields": list(REQUIRED_CORRELATION_FIELDS),
|
||||
"results": sorted(RESULTS),
|
||||
"retention_defaults_days": dict(RETENTION_DAYS),
|
||||
"sink_env": AUDIT_LOG_ENV,
|
||||
"enabled": audit_enabled(),
|
||||
"append_only": True,
|
||||
"redact_before_persist": True,
|
||||
"timestamp_format": "ISO-8601, timezone-aware, UTC",
|
||||
"relationship_to_mcp_audit": (
|
||||
"webui.console_audit records console intent and authorization "
|
||||
"outcomes; gitea_audit records MCP mutations. A Phase 2 action "
|
||||
"emits both, correlated by correlation.request_id."
|
||||
),
|
||||
}
|
||||
@@ -0,0 +1,537 @@
|
||||
"""Console authorization and RBAC model (#633, Phase 1).
|
||||
|
||||
The read-only MVP (#426–#436) ships with no authentication: protection comes
|
||||
from network placement alone (#435). That is adequate while every route is a
|
||||
GET, and inadequate the moment Phase 2 wires a gated write. This module is the
|
||||
authorization model those writes must go through, landed *before* any of them
|
||||
exists so no write can be added without an authority to check against.
|
||||
|
||||
Phase 1 scope is the model itself: identity resolution, the role matrix, the
|
||||
privileged-action list, and a fail-closed :func:`authorize`. It deliberately
|
||||
does **not** enable any write. ``webui.gated_actions`` stays globally disabled,
|
||||
so an allow decision here is necessary but never sufficient.
|
||||
|
||||
Two invariants hold for every caller:
|
||||
|
||||
- **Default deny.** An unrecognised action, an unknown role, or an absent
|
||||
principal denies. There is no implicit allow branch and no "unless" clause.
|
||||
- **Authorization is not execution.** :func:`authorize` returns a decision
|
||||
record. It never calls MCP, never mutates, and never consults credentials.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from typing import Any
|
||||
|
||||
from task_capability_map import required_permission, required_role
|
||||
|
||||
# --- Roles ------------------------------------------------------------------
|
||||
# Ordered least to most authority. Higher ranks inherit every lower rank's
|
||||
# permitted actions; the matrix below is expressed as a minimum required rank.
|
||||
VIEWER = "viewer"
|
||||
OPERATOR = "operator"
|
||||
CONTROLLER = "controller"
|
||||
ADMIN = "admin"
|
||||
|
||||
ROLE_ORDER: tuple[str, ...] = (VIEWER, OPERATOR, CONTROLLER, ADMIN)
|
||||
_ROLE_RANK: dict[str, int] = {role: idx for idx, role in enumerate(ROLE_ORDER)}
|
||||
|
||||
ROLE_DESCRIPTIONS: dict[str, str] = {
|
||||
VIEWER: "Read every console view. No write, ever, in any phase.",
|
||||
OPERATOR: "Viewer, plus author-class work: claim, comment, open a PR.",
|
||||
CONTROLLER: "Operator, plus reviewer/merger-class decisions on a PR.",
|
||||
ADMIN: "Controller, plus destructive and policy-editing actions.",
|
||||
}
|
||||
|
||||
# --- Identity sources -------------------------------------------------------
|
||||
IDENTITY_NONE = "none"
|
||||
IDENTITY_LOCAL_DEV = "local_dev"
|
||||
IDENTITY_ACCESS_PROXY = "access_proxy"
|
||||
|
||||
IDENTITY_SOURCES: dict[str, dict[str, Any]] = {
|
||||
IDENTITY_NONE: {
|
||||
"description": (
|
||||
"No authentication configured. Every request is anonymous and "
|
||||
"capped at viewer. This is the MVP default and the only mode "
|
||||
"whose safety rests entirely on network placement (#435)."
|
||||
),
|
||||
"authenticated": False,
|
||||
"safe_for_shared_host": False,
|
||||
"phase_available": 1,
|
||||
},
|
||||
IDENTITY_LOCAL_DEV: {
|
||||
"description": (
|
||||
"Developer-supplied principal read from the environment. INSECURE: "
|
||||
"the subject and role are asserted, never verified. Loopback only."
|
||||
),
|
||||
"authenticated": True,
|
||||
"safe_for_shared_host": False,
|
||||
"phase_available": 1,
|
||||
},
|
||||
IDENTITY_ACCESS_PROXY: {
|
||||
"description": (
|
||||
"Subject asserted by a trusted access proxy (Cloudflare Access, "
|
||||
"WARP, or an org VPN portal) via a verified request header. The "
|
||||
"proxy performs authentication; the console performs authorization."
|
||||
),
|
||||
"authenticated": True,
|
||||
"safe_for_shared_host": True,
|
||||
"phase_available": 2,
|
||||
},
|
||||
}
|
||||
|
||||
# Environment configuration. All are read server-side and never rendered.
|
||||
AUTH_MODE_ENV = "WEBUI_AUTH_MODE"
|
||||
DEV_SUBJECT_ENV = "WEBUI_DEV_SUBJECT"
|
||||
DEV_ROLE_ENV = "WEBUI_DEV_ROLE"
|
||||
ROLE_MAP_ENV = "WEBUI_ROLE_MAP"
|
||||
REQUIRE_PROBE_AUTH_ENV = "WEBUI_REQUIRE_PROBE_AUTH"
|
||||
ACCESS_SUBJECT_HEADER = "cf-access-authenticated-user-email"
|
||||
|
||||
# --- Action classes ---------------------------------------------------------
|
||||
CLASS_READ = "read"
|
||||
CLASS_WRITE = "gated_write"
|
||||
CLASS_PRIVILEGED = "privileged"
|
||||
CLASS_DESTRUCTIVE = "destructive"
|
||||
|
||||
# --- Privileged action list -------------------------------------------------
|
||||
# ``task_key`` ties each console action back to ``task_capability_map``, so the
|
||||
# console cannot invent an authority the MCP layer does not already define.
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ConsoleAction:
|
||||
"""One console action and the authority required to invoke it."""
|
||||
|
||||
action_id: str
|
||||
task_key: str
|
||||
action_class: str
|
||||
minimum_role: str
|
||||
requires_confirmation: bool
|
||||
dual_control: bool
|
||||
break_glass: bool
|
||||
phase: int
|
||||
summary: str
|
||||
|
||||
@property
|
||||
def mcp_permission(self) -> str:
|
||||
return required_permission(self.task_key)
|
||||
|
||||
@property
|
||||
def mcp_role(self) -> str:
|
||||
return required_role(self.task_key)
|
||||
|
||||
@property
|
||||
def privileged(self) -> bool:
|
||||
return self.action_class in {CLASS_PRIVILEGED, CLASS_DESTRUCTIVE}
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
data = asdict(self)
|
||||
data["mcp_permission"] = self.mcp_permission
|
||||
data["mcp_role"] = self.mcp_role
|
||||
data["privileged"] = self.privileged
|
||||
return data
|
||||
|
||||
|
||||
_ACTION_SPECS: tuple[ConsoleAction, ...] = (
|
||||
ConsoleAction(
|
||||
action_id="claim_issue",
|
||||
task_key="claim_issue",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Apply status:in-progress to an issue.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="comment_issue",
|
||||
task_key="comment_issue",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Post an issue comment.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="create_issue",
|
||||
task_key="create_issue",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Open a new tracking issue.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="comment_pr",
|
||||
task_key="comment_pr",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Post a PR thread comment.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="create_pr",
|
||||
task_key="create_pr",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Open a PR from a locked feature branch.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="review_pr",
|
||||
task_key="review_pr",
|
||||
action_class=CLASS_PRIVILEGED,
|
||||
minimum_role=CONTROLLER,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=3,
|
||||
summary="Submit an approve / request-changes verdict.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="close_pr",
|
||||
task_key="close_pr",
|
||||
action_class=CLASS_PRIVILEGED,
|
||||
minimum_role=CONTROLLER,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=3,
|
||||
summary="Close a pull request without merging.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="merge_pr",
|
||||
task_key="merge_pr",
|
||||
action_class=CLASS_PRIVILEGED,
|
||||
minimum_role=CONTROLLER,
|
||||
requires_confirmation=True,
|
||||
dual_control=True,
|
||||
break_glass=True,
|
||||
phase=3,
|
||||
summary="Merge an approved pull request.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="delete_branch",
|
||||
task_key="delete_branch",
|
||||
action_class=CLASS_DESTRUCTIVE,
|
||||
minimum_role=ADMIN,
|
||||
requires_confirmation=True,
|
||||
dual_control=True,
|
||||
break_glass=True,
|
||||
phase=3,
|
||||
summary="Remove a remote feature branch.",
|
||||
),
|
||||
)
|
||||
|
||||
ACTIONS: dict[str, ConsoleAction] = {a.action_id: a for a in _ACTION_SPECS}
|
||||
|
||||
|
||||
def privileged_actions() -> tuple[ConsoleAction, ...]:
|
||||
"""Actions requiring dual control, break-glass, or controller+ authority."""
|
||||
return tuple(a for a in _ACTION_SPECS if a.privileged)
|
||||
|
||||
|
||||
def get_action(action_id: str) -> ConsoleAction | None:
|
||||
return ACTIONS.get(action_id)
|
||||
|
||||
|
||||
# --- Principals -------------------------------------------------------------
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Principal:
|
||||
"""Who is making a request, and how strongly that is known."""
|
||||
|
||||
subject: str
|
||||
role: str
|
||||
identity_source: str
|
||||
authenticated: bool
|
||||
warnings: tuple[str, ...] = field(default_factory=tuple)
|
||||
|
||||
@property
|
||||
def rank(self) -> int:
|
||||
return _ROLE_RANK.get(self.role, -1)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"subject": self.subject,
|
||||
"role": self.role,
|
||||
"identity_source": self.identity_source,
|
||||
"authenticated": self.authenticated,
|
||||
"warnings": list(self.warnings),
|
||||
}
|
||||
|
||||
|
||||
ANONYMOUS = Principal(
|
||||
subject="anonymous",
|
||||
role=VIEWER,
|
||||
identity_source=IDENTITY_NONE,
|
||||
authenticated=False,
|
||||
warnings=("No authentication configured; capped at viewer.",),
|
||||
)
|
||||
|
||||
|
||||
def auth_mode(env: dict[str, str] | None = None) -> str:
|
||||
"""Resolve the configured identity source, defaulting to ``none``."""
|
||||
source = env if env is not None else os.environ
|
||||
raw = (source.get(AUTH_MODE_ENV) or "").strip().lower().replace("-", "_")
|
||||
if raw in IDENTITY_SOURCES:
|
||||
return raw
|
||||
return IDENTITY_NONE
|
||||
|
||||
|
||||
def _role_map(env: dict[str, str]) -> dict[str, str]:
|
||||
"""Parse ``WEBUI_ROLE_MAP`` (JSON subject→role). Invalid config yields {}."""
|
||||
raw = (env.get(ROLE_MAP_ENV) or "").strip()
|
||||
if not raw:
|
||||
return {}
|
||||
try:
|
||||
parsed = json.loads(raw)
|
||||
except Exception:
|
||||
return {}
|
||||
if not isinstance(parsed, dict):
|
||||
return {}
|
||||
return {
|
||||
str(k): str(v).strip().lower()
|
||||
for k, v in parsed.items()
|
||||
if str(v).strip().lower() in _ROLE_RANK
|
||||
}
|
||||
|
||||
|
||||
def resolve_principal(
|
||||
headers: dict[str, str] | None = None,
|
||||
env: dict[str, str] | None = None,
|
||||
) -> Principal:
|
||||
"""Resolve the requesting principal. Unknown or unconfigured → anonymous.
|
||||
|
||||
Never raises and never trusts a client-supplied role: the role always comes
|
||||
from server-side configuration keyed by the resolved subject.
|
||||
"""
|
||||
source_env = dict(env) if env is not None else dict(os.environ)
|
||||
lowered = {str(k).lower(): str(v) for k, v in (headers or {}).items()}
|
||||
mode = auth_mode(source_env)
|
||||
|
||||
if mode == IDENTITY_LOCAL_DEV:
|
||||
subject = (source_env.get(DEV_SUBJECT_ENV) or "").strip()
|
||||
if not subject:
|
||||
return ANONYMOUS
|
||||
role = (source_env.get(DEV_ROLE_ENV) or VIEWER).strip().lower()
|
||||
if role not in _ROLE_RANK:
|
||||
role = VIEWER
|
||||
return Principal(
|
||||
subject=subject,
|
||||
role=role,
|
||||
identity_source=IDENTITY_LOCAL_DEV,
|
||||
authenticated=True,
|
||||
warnings=(
|
||||
"local-dev identity is asserted, not verified; never use "
|
||||
"outside loopback.",
|
||||
),
|
||||
)
|
||||
|
||||
if mode == IDENTITY_ACCESS_PROXY:
|
||||
subject = (lowered.get(ACCESS_SUBJECT_HEADER) or "").strip()
|
||||
if not subject:
|
||||
# Proxy mode with no proxy header means the request did not
|
||||
# traverse the proxy. Fail closed rather than trust it.
|
||||
return ANONYMOUS
|
||||
role = _role_map(source_env).get(subject, VIEWER)
|
||||
return Principal(
|
||||
subject=subject,
|
||||
role=role,
|
||||
identity_source=IDENTITY_ACCESS_PROXY,
|
||||
authenticated=True,
|
||||
)
|
||||
|
||||
return ANONYMOUS
|
||||
|
||||
|
||||
def probe_auth_required(env: dict[str, str] | None = None) -> bool:
|
||||
"""Whether non-public probes must be authenticated. Default False.
|
||||
|
||||
#633 requires the console to *fail closed on missing auth for non-public
|
||||
health probes if configured*. The default stays off so the MVP ``/health``
|
||||
contract is unchanged; an operator opts in explicitly.
|
||||
"""
|
||||
source = env if env is not None else os.environ
|
||||
return (source.get(REQUIRE_PROBE_AUTH_ENV) or "").strip().lower() in {
|
||||
"1",
|
||||
"true",
|
||||
"yes",
|
||||
}
|
||||
|
||||
|
||||
# --- Authorization ----------------------------------------------------------
|
||||
|
||||
DENY_UNKNOWN_ACTION = "unknown_action"
|
||||
DENY_UNAUTHENTICATED = "unauthenticated"
|
||||
DENY_INSUFFICIENT_ROLE = "insufficient_role"
|
||||
DENY_UNKNOWN_ROLE = "unknown_role"
|
||||
DENY_PHASE_NOT_ACTIVE = "phase_not_active"
|
||||
ALLOW_PREVIEW = "allowed_preview_only"
|
||||
|
||||
# Phase 1 is the only active console phase. Phase 2 opens gated writes and is
|
||||
# gated on this model landing; nothing here enables it.
|
||||
ACTIVE_PHASE = 1
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AuthorizationDecision:
|
||||
"""Result of an authorization check. Never an execution grant."""
|
||||
|
||||
allowed: bool
|
||||
reason_code: str
|
||||
detail: str
|
||||
action_id: str
|
||||
principal: Principal
|
||||
required_role: str | None = None
|
||||
action_class: str | None = None
|
||||
requires_confirmation: bool = False
|
||||
dual_control: bool = False
|
||||
break_glass: bool = False
|
||||
execution_enabled: bool = False
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"allowed": self.allowed,
|
||||
"reason_code": self.reason_code,
|
||||
"detail": self.detail,
|
||||
"action_id": self.action_id,
|
||||
"principal": self.principal.to_dict(),
|
||||
"required_role": self.required_role,
|
||||
"action_class": self.action_class,
|
||||
"requires_confirmation": self.requires_confirmation,
|
||||
"dual_control": self.dual_control,
|
||||
"break_glass": self.break_glass,
|
||||
"execution_enabled": self.execution_enabled,
|
||||
"active_phase": ACTIVE_PHASE,
|
||||
}
|
||||
|
||||
|
||||
def authorize(
|
||||
action_id: str,
|
||||
principal: Principal | None = None,
|
||||
*,
|
||||
for_execution: bool = False,
|
||||
) -> AuthorizationDecision:
|
||||
"""Decide whether *principal* may invoke *action_id*. Deny by default.
|
||||
|
||||
``for_execution`` distinguishes a read-only preview from a real invocation.
|
||||
Even an allowed decision reports ``execution_enabled=False`` while the
|
||||
console is in Phase 1, so no caller can read an allow as permission to
|
||||
mutate.
|
||||
"""
|
||||
who = principal if principal is not None else ANONYMOUS
|
||||
action = get_action(action_id)
|
||||
|
||||
if action is None:
|
||||
return AuthorizationDecision(
|
||||
allowed=False,
|
||||
reason_code=DENY_UNKNOWN_ACTION,
|
||||
detail=f"No console action registered as {action_id!r}.",
|
||||
action_id=action_id,
|
||||
principal=who,
|
||||
)
|
||||
|
||||
base: dict[str, Any] = {
|
||||
"action_id": action_id,
|
||||
"principal": who,
|
||||
"required_role": action.minimum_role,
|
||||
"action_class": action.action_class,
|
||||
"requires_confirmation": action.requires_confirmation,
|
||||
"dual_control": action.dual_control,
|
||||
"break_glass": action.break_glass,
|
||||
"execution_enabled": False,
|
||||
}
|
||||
|
||||
if not who.authenticated:
|
||||
return AuthorizationDecision(
|
||||
allowed=False,
|
||||
reason_code=DENY_UNAUTHENTICATED,
|
||||
detail=(
|
||||
"Write actions require an authenticated principal; this "
|
||||
"request is anonymous."
|
||||
),
|
||||
**base,
|
||||
)
|
||||
|
||||
if who.rank < 0:
|
||||
return AuthorizationDecision(
|
||||
allowed=False,
|
||||
reason_code=DENY_UNKNOWN_ROLE,
|
||||
detail=f"Role {who.role!r} is not in the console role matrix.",
|
||||
**base,
|
||||
)
|
||||
|
||||
if who.rank < _ROLE_RANK[action.minimum_role]:
|
||||
return AuthorizationDecision(
|
||||
allowed=False,
|
||||
reason_code=DENY_INSUFFICIENT_ROLE,
|
||||
detail=(
|
||||
f"Action {action_id!r} requires {action.minimum_role!r}; "
|
||||
f"principal holds {who.role!r}."
|
||||
),
|
||||
**base,
|
||||
)
|
||||
|
||||
if for_execution and action.phase > ACTIVE_PHASE:
|
||||
return AuthorizationDecision(
|
||||
allowed=False,
|
||||
reason_code=DENY_PHASE_NOT_ACTIVE,
|
||||
detail=(
|
||||
f"Action {action_id!r} belongs to phase {action.phase}; the "
|
||||
f"console is in phase {ACTIVE_PHASE}. Execution is not wired."
|
||||
),
|
||||
**base,
|
||||
)
|
||||
|
||||
return AuthorizationDecision(
|
||||
allowed=True,
|
||||
reason_code=ALLOW_PREVIEW,
|
||||
detail=(
|
||||
"Principal holds the required role. Preview only — execution "
|
||||
"remains disabled until the Phase 2 action framework ships."
|
||||
),
|
||||
**base,
|
||||
)
|
||||
|
||||
|
||||
def rbac_matrix() -> dict[str, Any]:
|
||||
"""Machine-readable RBAC matrix and privileged-action list."""
|
||||
return {
|
||||
"model_version": 1,
|
||||
"active_phase": ACTIVE_PHASE,
|
||||
"roles": [
|
||||
{
|
||||
"role": role,
|
||||
"rank": _ROLE_RANK[role],
|
||||
"description": ROLE_DESCRIPTIONS[role],
|
||||
"permitted_actions": sorted(
|
||||
a.action_id
|
||||
for a in _ACTION_SPECS
|
||||
if _ROLE_RANK[role] >= _ROLE_RANK[a.minimum_role]
|
||||
),
|
||||
}
|
||||
for role in ROLE_ORDER
|
||||
],
|
||||
"identity_sources": IDENTITY_SOURCES,
|
||||
"actions": [a.to_dict() for a in _ACTION_SPECS],
|
||||
"privileged_actions": [a.action_id for a in privileged_actions()],
|
||||
"default_decision": "deny",
|
||||
"execution_enabled": False,
|
||||
}
|
||||
@@ -0,0 +1,169 @@
|
||||
"""Secret redaction policy for every console surface (#633).
|
||||
|
||||
The MVP already redacts MCP-side mutation records through ``gitea_audit``.
|
||||
This module is the console-facing policy: one redaction pass applied to API
|
||||
payloads, rendered HTML, log lines, and audit records *before* they leave the
|
||||
server or reach persistent storage.
|
||||
|
||||
Design constraints:
|
||||
|
||||
- **Reuse, never fork.** ``gitea_audit.redact`` remains the authority for
|
||||
secret-looking dict keys, ``Authorization`` material, and raw URLs. This
|
||||
module runs that pass first and then applies console-specific patterns for
|
||||
keychain references, key/value assignments, private-key blocks, and JWTs.
|
||||
- **Never raises.** Redaction is a safety control; a malformed payload must
|
||||
degrade to a redacted placeholder rather than propagate an exception.
|
||||
- **Redact before persist.** ``webui.console_audit`` calls this module before
|
||||
writing, so an unredacted record is never durable.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
import gitea_audit
|
||||
|
||||
REDACTED = gitea_audit.REDACTED
|
||||
|
||||
# Console-specific patterns applied after the shared ``gitea_audit`` pass.
|
||||
# Each keeps the identifying key so an operator can still tell *what* was
|
||||
# removed, and replaces only the secret run itself.
|
||||
_KEYCHAIN_REF = re.compile(r"(?i)\bkeychain:[\w.\-/@]+")
|
||||
_KEYCHAIN_CMD = re.compile(
|
||||
r"(?i)\bsecurity\s+find-(?:generic|internet)-password\b[^\n]*"
|
||||
)
|
||||
_ASSIGNMENT = re.compile(
|
||||
r"(?i)\b(token|password|passwd|secret|api[_-]?key|access[_-]?key|"
|
||||
r"client[_-]?secret|private[_-]?key)\b(\s*[:=]\s*)"
|
||||
r"(\"[^\"]*\"|'[^']*'|\S+)"
|
||||
)
|
||||
_ENV_ASSIGNMENT = re.compile(
|
||||
r"(?i)\b(GITEA_(?:TOKEN|PASS|PASSWORD)[A-Z0-9_]*)(\s*=\s*)"
|
||||
r"(\"[^\"]*\"|'[^']*'|\S+)"
|
||||
)
|
||||
_PRIVATE_KEY_BLOCK = re.compile(
|
||||
r"-----BEGIN [A-Z ]*PRIVATE KEY-----.*?-----END [A-Z ]*PRIVATE KEY-----",
|
||||
re.S,
|
||||
)
|
||||
_JWT = re.compile(
|
||||
r"\beyJ[A-Za-z0-9_\-]{8,}\.[A-Za-z0-9_\-]{8,}\.[A-Za-z0-9_\-]{8,}\b"
|
||||
)
|
||||
|
||||
# Shapes that mean a payload still carries a secret. ``scan_for_secrets`` uses
|
||||
# these to assert a surface is clean.
|
||||
_DETECTORS: tuple[tuple[str, re.Pattern[str]], ...] = (
|
||||
("keychain_reference", _KEYCHAIN_REF),
|
||||
("keychain_command", _KEYCHAIN_CMD),
|
||||
("credential_assignment", _ASSIGNMENT),
|
||||
("credential_env_assignment", _ENV_ASSIGNMENT),
|
||||
("private_key_block", _PRIVATE_KEY_BLOCK),
|
||||
("json_web_token", _JWT),
|
||||
("bearer_credential", re.compile(r"(?i)\b(?:bearer|basic)\s+\S{8,}")),
|
||||
)
|
||||
|
||||
|
||||
def _mask_assignment(match: re.Match[str]) -> str:
|
||||
"""Keep the key and separator, replace the value."""
|
||||
return f"{match.group(1)}{match.group(2)}{REDACTED}"
|
||||
|
||||
|
||||
def redact_text(text: Any) -> Any:
|
||||
"""Redact secret material from a single string.
|
||||
|
||||
Non-strings are returned unchanged so this is safe to map over mixed
|
||||
payloads. Runs the shared ``gitea_audit`` pass first, then the
|
||||
console-specific patterns.
|
||||
"""
|
||||
if not isinstance(text, str) or not text:
|
||||
return text
|
||||
try:
|
||||
out = gitea_audit.redact(text)
|
||||
if not isinstance(out, str): # defensive; redact() returns str for str
|
||||
return REDACTED
|
||||
out = _PRIVATE_KEY_BLOCK.sub(f"{REDACTED}_PRIVATE_KEY", out)
|
||||
out = _ENV_ASSIGNMENT.sub(_mask_assignment, out)
|
||||
out = _ASSIGNMENT.sub(_mask_assignment, out)
|
||||
out = _KEYCHAIN_CMD.sub(f"{REDACTED}_KEYCHAIN_COMMAND", out)
|
||||
out = _KEYCHAIN_REF.sub(f"{REDACTED}_KEYCHAIN_REF", out)
|
||||
out = _JWT.sub(f"{REDACTED}_JWT", out)
|
||||
return out
|
||||
except Exception:
|
||||
# Fail closed: an unredactable string is dropped rather than emitted raw.
|
||||
return REDACTED
|
||||
|
||||
|
||||
def redact_payload(value: Any) -> Any:
|
||||
"""Recursively redact a JSON-able payload for any console surface.
|
||||
|
||||
Secret-looking dict keys are replaced wholesale by the shared
|
||||
``gitea_audit`` policy; every remaining string is run through
|
||||
:func:`redact_text`.
|
||||
"""
|
||||
try:
|
||||
shared = gitea_audit.redact(value)
|
||||
except Exception:
|
||||
return REDACTED
|
||||
return _walk(shared)
|
||||
|
||||
|
||||
def _walk(value: Any) -> Any:
|
||||
if isinstance(value, dict):
|
||||
return {k: _walk(v) for k, v in value.items()}
|
||||
if isinstance(value, (list, tuple)):
|
||||
return [_walk(v) for v in value]
|
||||
if isinstance(value, str):
|
||||
return redact_text(value)
|
||||
return value
|
||||
|
||||
|
||||
def scan_for_secrets(value: Any) -> list[str]:
|
||||
"""Return detector names that still match *value* after serialization.
|
||||
|
||||
Used to assert an outbound payload or rendered page is clean. An empty
|
||||
list means no known secret shape was found. Already-redacted hits are not
|
||||
findings.
|
||||
"""
|
||||
if isinstance(value, str):
|
||||
text = value
|
||||
else:
|
||||
try:
|
||||
text = json.dumps(value, default=str)
|
||||
except Exception:
|
||||
text = str(value)
|
||||
findings: list[str] = []
|
||||
for name, pattern in _DETECTORS:
|
||||
for match in pattern.finditer(text):
|
||||
if REDACTED in match.group(0):
|
||||
continue
|
||||
findings.append(name)
|
||||
break
|
||||
return findings
|
||||
|
||||
|
||||
def redaction_policy() -> dict[str, Any]:
|
||||
"""Machine-readable statement of the redaction rules (never secrets)."""
|
||||
return {
|
||||
"policy_version": 1,
|
||||
"applies_to": [
|
||||
"json_api_responses",
|
||||
"rendered_html",
|
||||
"server_logs",
|
||||
"audit_records",
|
||||
],
|
||||
"ordering": "shared gitea_audit pass, then console patterns",
|
||||
"redact_before_persist": True,
|
||||
"shared_rules": {
|
||||
"source": "gitea_audit.redact",
|
||||
"secret_key_hints": list(gitea_audit._SECRET_KEY_HINTS),
|
||||
"secret_value_prefixes": list(gitea_audit._SECRET_VALUE_PREFIXES),
|
||||
"urls": "credentials, secret query parameters, and real hosts redacted",
|
||||
},
|
||||
"console_rules": [
|
||||
{"name": name, "pattern": pattern.pattern}
|
||||
for name, pattern in _DETECTORS
|
||||
],
|
||||
"placeholder": REDACTED,
|
||||
"failure_mode": "fail closed — unredactable values become the placeholder",
|
||||
}
|
||||
+95
-17
@@ -2,28 +2,66 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
NAV_ITEMS = (
|
||||
("/", "Home"),
|
||||
("/queue", "Queue"),
|
||||
("/projects", "Projects"),
|
||||
("/prompts", "Prompts"),
|
||||
("/runtime", "Runtime"),
|
||||
("/audit", "Audit"),
|
||||
("/worktrees", "Worktrees"),
|
||||
("/leases", "Leases"),
|
||||
("/actions", "Actions"),
|
||||
)
|
||||
import os
|
||||
|
||||
from webui.nav import NAV_GROUPS
|
||||
|
||||
MVP_NOTICE = (
|
||||
"Read-only MVP — Gitea, MCP tools, and canonical workflows remain the "
|
||||
"source of truth. No mutation endpoints."
|
||||
)
|
||||
|
||||
# Canonical docs entry point surfaced from the shell header (#638).
|
||||
DOCS_URL = (
|
||||
"https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/src/branch/"
|
||||
"master/docs/webui-local-dev.md"
|
||||
)
|
||||
|
||||
_LOCAL_HOSTS = frozenset({"", "127.0.0.1", "localhost", "::1"})
|
||||
|
||||
|
||||
def environment_label() -> str:
|
||||
"""Classify the serving environment as ``local`` or ``remote`` (#638).
|
||||
|
||||
Derived from the same ``WEBUI_HOST`` default the app binds to; loopback
|
||||
hosts are ``local``, anything else is ``remote``. Read-only signal only.
|
||||
"""
|
||||
host = (os.environ.get("WEBUI_HOST", "127.0.0.1") or "").strip().lower()
|
||||
return "local" if host in _LOCAL_HOSTS else "remote"
|
||||
|
||||
|
||||
def _render_nav() -> str:
|
||||
groups_html = []
|
||||
for group in NAV_GROUPS:
|
||||
links = "".join(
|
||||
f'<a href="{item.href}"'
|
||||
+ (' class="nav-stub"' if item.status == "stub" else "")
|
||||
+ f'>{item.label}</a>'
|
||||
for item in group.items
|
||||
)
|
||||
groups_html.append(
|
||||
'<div class="nav-group">'
|
||||
f'<span class="nav-group-label">{group.label}</span>'
|
||||
f'<span class="nav-group-links">{links}</span>'
|
||||
"</div>"
|
||||
)
|
||||
return "".join(groups_html)
|
||||
|
||||
|
||||
def _render_badges() -> str:
|
||||
env = environment_label()
|
||||
return (
|
||||
'<div class="header-badges">'
|
||||
f'<span class="badge env-badge env-{env}">env: {env}</span>'
|
||||
'<span class="badge mode-badge">mode: read-only</span>'
|
||||
f'<a class="badge docs-link" href="{DOCS_URL}">Docs</a>'
|
||||
"</div>"
|
||||
)
|
||||
|
||||
|
||||
def render_page(*, title: str, body_html: str, extra_head: str = "") -> str:
|
||||
nav_links = "".join(
|
||||
f'<a href="{href}">{label}</a>' for href, label in NAV_ITEMS
|
||||
)
|
||||
nav_links = _render_nav()
|
||||
header_badges = _render_badges()
|
||||
return f"""<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
@@ -53,21 +91,58 @@ def render_page(*, title: str, body_html: str, extra_head: str = "") -> str:
|
||||
padding: 0.75rem 1.25rem;
|
||||
}}
|
||||
header h1 {{
|
||||
margin: 0 0 0.5rem;
|
||||
margin: 0;
|
||||
font-size: 1.1rem;
|
||||
font-weight: 600;
|
||||
}}
|
||||
.header-top {{
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
gap: 0.5rem 1rem;
|
||||
margin-bottom: 0.6rem;
|
||||
}}
|
||||
.header-badges {{ display: inline-flex; flex-wrap: wrap; gap: 0.4rem; }}
|
||||
.env-badge.env-local {{ color: #8fd19e; border-color: #3d6b4a; }}
|
||||
.env-badge.env-remote {{ color: #e0c27a; border-color: #6b5730; }}
|
||||
.mode-badge {{ color: #9ec8f0; border-color: #3d5f7a; }}
|
||||
a.docs-link {{
|
||||
color: var(--accent);
|
||||
border-color: var(--accent);
|
||||
text-decoration: none;
|
||||
text-transform: none;
|
||||
}}
|
||||
a.docs-link:hover {{ filter: brightness(1.12); }}
|
||||
nav {{
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 0.75rem 1rem;
|
||||
gap: 0.5rem 1.25rem;
|
||||
}}
|
||||
.nav-group {{
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 0.15rem;
|
||||
}}
|
||||
.nav-group-label {{
|
||||
font-size: 0.68rem;
|
||||
text-transform: uppercase;
|
||||
letter-spacing: 0.04em;
|
||||
color: var(--muted);
|
||||
}}
|
||||
.nav-group-links {{ display: inline-flex; flex-wrap: wrap; gap: 0.6rem; }}
|
||||
nav a {{
|
||||
color: var(--accent);
|
||||
text-decoration: none;
|
||||
font-size: 0.9rem;
|
||||
}}
|
||||
nav a:hover {{ text-decoration: underline; }}
|
||||
nav a.nav-stub {{ color: var(--muted); }}
|
||||
nav a.nav-stub::after {{
|
||||
content: " ·stub";
|
||||
font-size: 0.7rem;
|
||||
color: var(--muted);
|
||||
}}
|
||||
main {{
|
||||
max-width: 52rem;
|
||||
margin: 0 auto;
|
||||
@@ -166,7 +241,10 @@ def render_page(*, title: str, body_html: str, extra_head: str = "") -> str:
|
||||
</head>
|
||||
<body>
|
||||
<header>
|
||||
<h1>MCP Control Plane</h1>
|
||||
<div class="header-top">
|
||||
<h1>MCP Control Plane</h1>
|
||||
{header_badges}
|
||||
</div>
|
||||
<nav>{nav_links}</nav>
|
||||
</header>
|
||||
<main>
|
||||
|
||||
+111
@@ -0,0 +1,111 @@
|
||||
"""Navigation IA for the Phase 1 operator console shell (#638).
|
||||
|
||||
Single source of truth for the console navigation so ``webui/layout.py`` and
|
||||
the ``webui/app.py`` route table stay aligned with epic #631. Read-only: every
|
||||
destination is a GET view or a Phase 1 placeholder. No mutation links.
|
||||
|
||||
Nav groups follow the #631 Phase 1 information architecture: Health, Traffic,
|
||||
Runtime/Sessions, Projects, Inventory, Timeline, Policy (placeholder), and
|
||||
Insights (placeholder). Later-phase surfaces are declared as ``stub`` items and
|
||||
backed by ``STUB_PAGES`` so their nav links resolve to a graceful placeholder
|
||||
instead of a 404.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NavItem:
|
||||
"""A single navigation destination.
|
||||
|
||||
``status`` is ``"live"`` for implemented views and ``"stub"`` for Phase 1
|
||||
placeholders whose backing view lands in a later child issue.
|
||||
"""
|
||||
|
||||
href: str
|
||||
label: str
|
||||
status: str = "live"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NavGroup:
|
||||
label: str
|
||||
items: tuple[NavItem, ...]
|
||||
|
||||
|
||||
NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
NavGroup("Health", (
|
||||
NavItem("/health", "Liveness"),
|
||||
)),
|
||||
NavGroup("Traffic", (
|
||||
NavItem("/queue", "Queue"),
|
||||
NavItem("/leases", "Leases"),
|
||||
NavItem("/actions", "Actions"),
|
||||
)),
|
||||
NavGroup("Runtime/Sessions", (
|
||||
NavItem("/runtime", "Runtime health"),
|
||||
NavItem("/sessions", "Sessions", "stub"),
|
||||
)),
|
||||
NavGroup("Projects", (
|
||||
NavItem("/projects", "Projects"),
|
||||
)),
|
||||
NavGroup("Inventory", (
|
||||
NavItem("/inventory", "Inventory", "stub"),
|
||||
NavItem("/worktrees", "Worktrees"),
|
||||
)),
|
||||
NavGroup("Timeline", (
|
||||
NavItem("/timeline", "Timeline", "stub"),
|
||||
)),
|
||||
NavGroup("Policy", (
|
||||
NavItem("/policy", "Policy", "stub"),
|
||||
NavItem("/prompts", "Prompts"),
|
||||
)),
|
||||
NavGroup("Insights", (
|
||||
NavItem("/insights", "Insights", "stub"),
|
||||
NavItem("/audit", "Audit"),
|
||||
)),
|
||||
)
|
||||
|
||||
|
||||
# Phase 1 placeholder destinations whose backing views land in later child
|
||||
# issues of epic #631. Each maps a path to (title, description). Routes are
|
||||
# registered so nav links resolve to a graceful, read-only stub page.
|
||||
STUB_PAGES: dict[str, tuple[str, str]] = {
|
||||
"/sessions": (
|
||||
"Sessions",
|
||||
"Active session, capability, and role inventory. Backed by the unified "
|
||||
"inventory API (#636) once it lands.",
|
||||
),
|
||||
"/inventory": (
|
||||
"Inventory",
|
||||
"Unified sessions, leases, locks, namespaces, and worktree inventory. "
|
||||
"Backed by the Phase 1 inventory API (#636).",
|
||||
),
|
||||
"/timeline": (
|
||||
"Timeline",
|
||||
"Workflow event timeline across issues and PRs. A later Phase 1 surface.",
|
||||
),
|
||||
"/policy": (
|
||||
"Policy",
|
||||
"Capability and role policy surface. Placeholder until a later phase.",
|
||||
),
|
||||
"/insights": (
|
||||
"Insights",
|
||||
"Aggregate operational insights and trends. Placeholder until a later "
|
||||
"phase.",
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def iter_nav_items():
|
||||
"""Yield every ``NavItem`` across all groups in declared order."""
|
||||
for group in NAV_GROUPS:
|
||||
for item in group.items:
|
||||
yield item
|
||||
|
||||
|
||||
def nav_hrefs() -> tuple[str, ...]:
|
||||
"""Return every navigation href in declared order."""
|
||||
return tuple(item.href for item in iter_nav_items())
|
||||
@@ -0,0 +1,682 @@
|
||||
"""Read-only system-health model for the operator console API (#634).
|
||||
|
||||
`/health` answers liveness only. Operators automating readiness checks need a
|
||||
structured view of *why* the control plane is or is not usable: which
|
||||
dependencies answered, how long they took, what version of the code is running,
|
||||
and whether the runtime is stale relative to its remote.
|
||||
|
||||
Three rules shape this module.
|
||||
|
||||
* **Read-only.** Every probe opens its subject read-only. The control-plane
|
||||
database is opened through a ``mode=ro`` URI so a health check can never
|
||||
create or migrate a schema, and no probe writes, restarts, or reloads
|
||||
anything — restart controls are Phase 2, and #630 forbids process-kill
|
||||
recovery outright.
|
||||
* **Fail-soft.** A dependency that is unreachable is a *status*, not an
|
||||
exception. Probes catch their own failures and report them as a degraded or
|
||||
down entry carrying a reason.
|
||||
* **Never claim more than was proven.** Readiness is derived only from probes
|
||||
that actually ran, ``mutation_safe`` stays false unless the parity commits are
|
||||
known and equal, and an MCP namespace is reported unproven because a web
|
||||
process cannot exercise the IDE-managed client path (#543).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import sqlite3
|
||||
import subprocess
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
from urllib.parse import urlsplit, urlunsplit
|
||||
|
||||
import control_plane_db
|
||||
import mcp_namespace_health
|
||||
from gitea_auth import api_request, get_auth_header, gitea_url
|
||||
|
||||
from webui.project_registry import load_registry
|
||||
|
||||
SERVICE_NAME = "mcp-control-plane-webui"
|
||||
API_PATH = "/api/v1/system/health"
|
||||
|
||||
STATUS_OK = "ok"
|
||||
STATUS_DEGRADED = "degraded"
|
||||
STATUS_DOWN = "down"
|
||||
STATUS_SKIPPED = "skipped"
|
||||
STATUS_UNPROVEN = "unproven"
|
||||
|
||||
# Statuses that count as a healthy answer from a probe.
|
||||
_HEALTHY_STATUSES = frozenset({STATUS_OK})
|
||||
# Statuses meaning "this probe did not run", as opposed to "it ran and failed".
|
||||
_NOT_RUN_STATUSES = frozenset({STATUS_SKIPPED})
|
||||
|
||||
_DEEP_PROBE_TTL_ENV = "WEBUI_HEALTH_PROBE_TTL_SECONDS"
|
||||
_DEFAULT_DEEP_PROBE_TTL = 15.0
|
||||
_GITEA_PROBE_TIMEOUT_SECONDS = 5.0
|
||||
|
||||
_OFFLINE_ENV = "WEBUI_TEST_OFFLINE"
|
||||
|
||||
# Credential-shaped material that must never reach the browser, mirroring the
|
||||
# forbidden client patterns in webui/deployment_boundary.py.
|
||||
_SECRET_RE = re.compile(
|
||||
r"(?i)\b(token|password|passwd|secret|authorization|bearer)\b\s*[:=]?\s*\S+"
|
||||
)
|
||||
_LONG_OPAQUE_RE = re.compile(r"\b[A-Za-z0-9_\-]{32,}\b")
|
||||
|
||||
# Captured once at import so uptime measures this process, not the request.
|
||||
_STARTED_AT = datetime.now(timezone.utc)
|
||||
_STARTED_MONOTONIC = time.monotonic()
|
||||
|
||||
# TTL cache for the expensive (network) probe only.
|
||||
_deep_cache: dict[str, tuple[float, "DependencyProbe"]] = {}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DependencyProbe:
|
||||
"""One dependency check, fail-soft, with its own latency."""
|
||||
|
||||
name: str
|
||||
kind: str
|
||||
status: str
|
||||
detail: str
|
||||
required: bool
|
||||
latency_ms: float | None = None
|
||||
metadata: dict[str, Any] | None = None
|
||||
|
||||
@property
|
||||
def healthy(self) -> bool:
|
||||
return self.status in _HEALTHY_STATUSES
|
||||
|
||||
@property
|
||||
def ran(self) -> bool:
|
||||
return self.status not in _NOT_RUN_STATUSES
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class VersionInfo:
|
||||
git_sha: str | None
|
||||
git_describe: str | None
|
||||
control_plane_schema_version: int | None
|
||||
python_version: str
|
||||
known: bool
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StaleRuntime:
|
||||
"""Parity between the running code, the checkout, and the remote.
|
||||
|
||||
``mutation_safe`` is deliberately conservative: unknown is not safe.
|
||||
"""
|
||||
|
||||
daemon_head: str | None
|
||||
checkout_head: str | None
|
||||
remote_head: str | None
|
||||
stale: bool
|
||||
determinable: bool
|
||||
mutation_safe: bool
|
||||
reasons: tuple[str, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SystemHealthSnapshot:
|
||||
status: str
|
||||
ready: bool
|
||||
readiness_complete: bool
|
||||
readiness_reasons: tuple[str, ...]
|
||||
service: str
|
||||
mode: str
|
||||
version: VersionInfo
|
||||
started_at: str
|
||||
uptime_seconds: float
|
||||
timestamp: str
|
||||
deep_probes_requested: bool
|
||||
dependencies: tuple[DependencyProbe, ...]
|
||||
mcp_namespaces: tuple[dict[str, Any], ...]
|
||||
stale_runtime: StaleRuntime
|
||||
probe_errors: tuple[str, ...] = ()
|
||||
|
||||
|
||||
def process_uptime() -> tuple[str, float]:
|
||||
"""Process start timestamp and uptime — in-memory, safe for `/health`."""
|
||||
return _STARTED_AT.isoformat(), round(time.monotonic() - _STARTED_MONOTONIC, 3)
|
||||
|
||||
|
||||
def _offline() -> bool:
|
||||
return (os.environ.get(_OFFLINE_ENV) or "").strip().lower() in {"1", "true", "yes"}
|
||||
|
||||
|
||||
def _repo_root() -> Path:
|
||||
override = (os.environ.get("WEBUI_REPO_ROOT") or "").strip()
|
||||
if override:
|
||||
return Path(override).resolve()
|
||||
return Path(__file__).resolve().parent.parent
|
||||
|
||||
|
||||
def _deep_probe_ttl() -> float:
|
||||
raw = (os.environ.get(_DEEP_PROBE_TTL_ENV) or "").strip()
|
||||
if not raw:
|
||||
return _DEFAULT_DEEP_PROBE_TTL
|
||||
try:
|
||||
value = float(raw)
|
||||
except ValueError:
|
||||
return _DEFAULT_DEEP_PROBE_TTL
|
||||
return value if value >= 0 else _DEFAULT_DEEP_PROBE_TTL
|
||||
|
||||
|
||||
def redact(text: str) -> str:
|
||||
"""Strip credential-shaped material from operator-visible probe text.
|
||||
|
||||
Probe details carry exception strings, and an exception raised by an HTTP
|
||||
client can quote the request that failed. Redaction happens here, at the
|
||||
boundary where those strings become part of a browser-bound payload.
|
||||
"""
|
||||
if not text:
|
||||
return ""
|
||||
cleaned = _redact_urls(text)
|
||||
cleaned = _SECRET_RE.sub(lambda m: f"{m.group(1)}=[redacted]", cleaned)
|
||||
return _LONG_OPAQUE_RE.sub("[redacted]", cleaned)
|
||||
|
||||
|
||||
def _redact_urls(text: str) -> str:
|
||||
return re.sub(r"https?://\S+", lambda m: redact_url(m.group(0)), text)
|
||||
|
||||
|
||||
def redact_url(url: str) -> str:
|
||||
"""Reduce a URL to scheme://host/path — no userinfo, no query, no fragment."""
|
||||
try:
|
||||
parts = urlsplit(url)
|
||||
except ValueError:
|
||||
return "[redacted-url]"
|
||||
if not parts.scheme or not parts.hostname:
|
||||
return "[redacted-url]"
|
||||
netloc = parts.hostname
|
||||
if parts.port:
|
||||
netloc = f"{netloc}:{parts.port}"
|
||||
return urlunsplit((parts.scheme, netloc, parts.path, "", ""))
|
||||
|
||||
|
||||
def _git(repo: Path, *args: str) -> str | None:
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
["git", "-C", str(repo), *args],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
timeout=10,
|
||||
)
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
return None
|
||||
if completed.returncode != 0:
|
||||
return None
|
||||
return (completed.stdout or "").strip() or None
|
||||
|
||||
|
||||
def _load_version(repo: Path, *, schema_version: int | None) -> VersionInfo:
|
||||
import platform
|
||||
|
||||
git_sha = None if _offline() else _git(repo, "rev-parse", "HEAD")
|
||||
describe = None if _offline() else _git(repo, "describe", "--tags", "--always")
|
||||
return VersionInfo(
|
||||
git_sha=git_sha,
|
||||
git_describe=describe,
|
||||
control_plane_schema_version=schema_version,
|
||||
python_version=platform.python_version(),
|
||||
known=bool(git_sha),
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Dependency probes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _elapsed_ms(started: float) -> float:
|
||||
return round((time.monotonic() - started) * 1000, 3)
|
||||
|
||||
|
||||
def probe_control_plane_db(db_path: str | None = None) -> DependencyProbe:
|
||||
"""Read-only reachability check for the control-plane SQLite substrate.
|
||||
|
||||
Opened through a ``mode=ro`` URI on purpose: ``ControlPlaneDB.__init__``
|
||||
creates directories and runs schema migrations, which a health check must
|
||||
never do.
|
||||
"""
|
||||
path = (db_path or control_plane_db.default_db_path()).strip()
|
||||
started = time.monotonic()
|
||||
metadata: dict[str, Any] = {"path": path}
|
||||
|
||||
def _result(status: str, detail: str) -> DependencyProbe:
|
||||
return DependencyProbe(
|
||||
name="control_plane_db",
|
||||
kind="sqlite",
|
||||
status=status,
|
||||
detail=detail,
|
||||
required=True,
|
||||
latency_ms=_elapsed_ms(started),
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
if not path or not os.path.exists(path):
|
||||
return _result(STATUS_DOWN, "control-plane database file does not exist yet")
|
||||
try:
|
||||
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=5)
|
||||
try:
|
||||
row = conn.execute(
|
||||
"SELECT value FROM schema_meta WHERE key = 'schema_version'"
|
||||
).fetchone()
|
||||
leases = conn.execute(
|
||||
"SELECT COUNT(*) FROM leases WHERE status = 'active'"
|
||||
).fetchone()
|
||||
finally:
|
||||
conn.close()
|
||||
except sqlite3.Error as exc:
|
||||
return _result(STATUS_DOWN, redact(f"control-plane database unreadable: {exc}"))
|
||||
|
||||
schema_version = int(row[0]) if row and str(row[0]).isdigit() else None
|
||||
metadata["schema_version"] = schema_version
|
||||
metadata["active_leases"] = int(leases[0]) if leases else None
|
||||
if schema_version is None:
|
||||
return _result(
|
||||
STATUS_DEGRADED, "control-plane database has no recorded schema version"
|
||||
)
|
||||
if schema_version != control_plane_db.SCHEMA_VERSION:
|
||||
return _result(
|
||||
STATUS_DEGRADED,
|
||||
f"control-plane schema version {schema_version} does not match the "
|
||||
f"version this code expects ({control_plane_db.SCHEMA_VERSION})",
|
||||
)
|
||||
return _result(STATUS_OK, f"schema v{schema_version} readable")
|
||||
|
||||
|
||||
def probe_repository(repo: Path) -> DependencyProbe:
|
||||
"""Local checkout reachability — required, cheap, no network."""
|
||||
started = time.monotonic()
|
||||
metadata: dict[str, Any] = {"repo_root": str(repo)}
|
||||
if _offline():
|
||||
return DependencyProbe(
|
||||
name="repository",
|
||||
kind="git",
|
||||
status=STATUS_SKIPPED,
|
||||
detail=f"{_OFFLINE_ENV} is set; git probe skipped",
|
||||
required=True,
|
||||
latency_ms=_elapsed_ms(started),
|
||||
metadata=metadata,
|
||||
)
|
||||
head = _git(repo, "rev-parse", "HEAD")
|
||||
if not head:
|
||||
return DependencyProbe(
|
||||
name="repository",
|
||||
kind="git",
|
||||
status=STATUS_DOWN,
|
||||
detail=f"HEAD could not be read at {repo}",
|
||||
required=True,
|
||||
latency_ms=_elapsed_ms(started),
|
||||
metadata=metadata,
|
||||
)
|
||||
branch = _git(repo, "rev-parse", "--abbrev-ref", "HEAD")
|
||||
metadata["head"] = head
|
||||
metadata["branch"] = branch
|
||||
return DependencyProbe(
|
||||
name="repository",
|
||||
kind="git",
|
||||
status=STATUS_OK,
|
||||
detail=f"checkout readable at {branch or 'detached HEAD'}",
|
||||
required=True,
|
||||
latency_ms=_elapsed_ms(started),
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
|
||||
def probe_gitea(host: str) -> DependencyProbe:
|
||||
"""Live Gitea reachability. Expensive (network), so opt-in via ``deep``.
|
||||
|
||||
Optional by design: the console stays useful for local inventory when the
|
||||
remote is unreachable, so a failure here degrades status without claiming
|
||||
the process itself is unready.
|
||||
"""
|
||||
started = time.monotonic()
|
||||
metadata: dict[str, Any] = {"host": host}
|
||||
|
||||
def _failure(status: str, detail: str) -> DependencyProbe:
|
||||
return DependencyProbe(
|
||||
name="gitea",
|
||||
kind="http",
|
||||
status=status,
|
||||
detail=detail,
|
||||
required=False,
|
||||
latency_ms=_elapsed_ms(started),
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
if not host:
|
||||
return _failure(STATUS_DEGRADED, "no Gitea host is configured in the registry")
|
||||
try:
|
||||
auth = get_auth_header(host)
|
||||
except Exception as exc: # noqa: BLE001 — credential guards are a status here
|
||||
return _failure(STATUS_DEGRADED, redact(f"credential lookup refused: {exc}"))
|
||||
if not auth:
|
||||
return _failure(STATUS_DEGRADED, f"no credentials available for {host}")
|
||||
|
||||
url = gitea_url(host, "/api/v1/version")
|
||||
metadata["endpoint"] = redact_url(url)
|
||||
try:
|
||||
data = api_request("GET", url, auth, timeout=_GITEA_PROBE_TIMEOUT_SECONDS)
|
||||
except Exception as exc: # noqa: BLE001 — a down dependency is a status
|
||||
return _failure(STATUS_DOWN, redact(f"Gitea probe failed: {exc}"))
|
||||
|
||||
if isinstance(data, dict) and data.get("version"):
|
||||
metadata["gitea_version"] = str(data["version"])
|
||||
return DependencyProbe(
|
||||
name="gitea",
|
||||
kind="http",
|
||||
status=STATUS_OK,
|
||||
detail=f"{host} reachable",
|
||||
required=False,
|
||||
latency_ms=_elapsed_ms(started),
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
|
||||
def _skipped_gitea(host: str) -> DependencyProbe:
|
||||
return DependencyProbe(
|
||||
name="gitea",
|
||||
kind="http",
|
||||
status=STATUS_SKIPPED,
|
||||
detail="network probe not requested; call with ?deep=1 to run it",
|
||||
required=False,
|
||||
latency_ms=None,
|
||||
metadata={"host": host},
|
||||
)
|
||||
|
||||
|
||||
def namespace_summaries() -> tuple[dict[str, Any], ...]:
|
||||
"""Declared MCP namespaces, each honestly reported as unproven.
|
||||
|
||||
The web process runs outside the IDE-managed MCP client, so it cannot
|
||||
invoke a namespace tool. Per #543 only a ``client_namespace`` probe proves
|
||||
that path, and inventing a healthy verdict here is exactly the false claim
|
||||
the mutation gates exist to prevent.
|
||||
"""
|
||||
rows: list[dict[str, Any]] = []
|
||||
for namespace, required_tool in sorted(
|
||||
mcp_namespace_health.REQUIRED_NAMESPACE_TOOLS.items()
|
||||
):
|
||||
classification = mcp_namespace_health.classify_namespace_probe(
|
||||
namespace,
|
||||
required_tool=required_tool,
|
||||
probe_result=None,
|
||||
probe_source=mcp_namespace_health.PROBE_SOURCE_UNKNOWN,
|
||||
)
|
||||
rows.append(
|
||||
{
|
||||
"namespace": namespace,
|
||||
"required_tool": required_tool,
|
||||
"status": STATUS_UNPROVEN,
|
||||
"ide_namespace_proven": bool(classification.get("ide_namespace_proven")),
|
||||
"reason": (
|
||||
"the web console cannot invoke the IDE-managed MCP client; "
|
||||
"namespace health must be proven with a client_namespace "
|
||||
"probe (#543)"
|
||||
),
|
||||
"error_type": classification.get("error_type"),
|
||||
}
|
||||
)
|
||||
return tuple(rows)
|
||||
|
||||
|
||||
def assess_stale_runtime(
|
||||
repo: Path,
|
||||
*,
|
||||
daemon_head: str | None = None,
|
||||
git_reader: Callable[..., str | None] | None = None,
|
||||
) -> StaleRuntime:
|
||||
"""Three-way parity view: running code, local checkout, remote-tracking ref.
|
||||
|
||||
``mutation_safe`` requires all three to be known and equal. Anything less —
|
||||
including "the remote ref was never fetched" — is reported as not safe with
|
||||
a reason, so an operator never reads an unproven green.
|
||||
"""
|
||||
reader = git_reader or (lambda *args: _git(repo, *args))
|
||||
reasons: list[str] = []
|
||||
# The offline switch suppresses real subprocess calls; an explicitly
|
||||
# injected reader is already a substitute for them and is always used.
|
||||
offline = _offline() and git_reader is None
|
||||
|
||||
checkout_head = None if offline else reader("rev-parse", "HEAD")
|
||||
remote_head = None if offline else reader("rev-parse", "@{upstream}")
|
||||
if offline:
|
||||
reasons.append(f"{_OFFLINE_ENV} is set; parity commits were not read")
|
||||
else:
|
||||
if checkout_head is None:
|
||||
reasons.append("local checkout HEAD could not be read")
|
||||
if remote_head is None:
|
||||
reasons.append(
|
||||
"no remote-tracking commit is known for the current branch; "
|
||||
"remote staleness is indeterminate (no fetch is performed here)"
|
||||
)
|
||||
|
||||
effective_daemon = daemon_head if daemon_head is not None else checkout_head
|
||||
if daemon_head is None:
|
||||
reasons.append(
|
||||
"the running MCP daemon's startup commit is not observable from the "
|
||||
"web process; the checkout commit is reported in its place"
|
||||
)
|
||||
|
||||
determinable = bool(checkout_head and remote_head and effective_daemon)
|
||||
stale = bool(
|
||||
determinable and len({checkout_head, remote_head, effective_daemon}) > 1
|
||||
)
|
||||
if stale:
|
||||
reasons.append(
|
||||
"runtime, checkout, and remote commits disagree; restart the MCP "
|
||||
"server after updating the checkout before trusting capability gates"
|
||||
)
|
||||
|
||||
return StaleRuntime(
|
||||
daemon_head=effective_daemon,
|
||||
checkout_head=checkout_head,
|
||||
remote_head=remote_head,
|
||||
stale=stale,
|
||||
determinable=determinable,
|
||||
mutation_safe=bool(determinable and not stale),
|
||||
reasons=tuple(reasons),
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Snapshot assembly
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _default_host() -> str:
|
||||
registry = load_registry()
|
||||
if not registry.projects:
|
||||
return ""
|
||||
raw = registry.projects[0].remote_host
|
||||
parts = urlsplit(raw.strip())
|
||||
return parts.netloc or raw.strip().rstrip("/")
|
||||
|
||||
|
||||
def _aggregate(
|
||||
probes: tuple[DependencyProbe, ...],
|
||||
) -> tuple[str, bool, bool, tuple[str, ...]]:
|
||||
"""Fold probe results into overall status and readiness.
|
||||
|
||||
Required probes drive readiness; optional probes can only degrade status.
|
||||
A probe that did not run leaves readiness incomplete rather than passing.
|
||||
"""
|
||||
reasons: list[str] = []
|
||||
required = [probe for probe in probes if probe.required]
|
||||
unrun_required = [probe for probe in required if not probe.ran]
|
||||
failed_required = [probe for probe in required if probe.ran and not probe.healthy]
|
||||
failed_optional = [
|
||||
probe
|
||||
for probe in probes
|
||||
if not probe.required and probe.ran and not probe.healthy
|
||||
]
|
||||
|
||||
for probe in unrun_required:
|
||||
reasons.append(
|
||||
f"required dependency '{probe.name}' was not probed: {probe.detail}"
|
||||
)
|
||||
for probe in failed_required:
|
||||
reasons.append(
|
||||
f"required dependency '{probe.name}' is {probe.status}: {probe.detail}"
|
||||
)
|
||||
for probe in failed_optional:
|
||||
reasons.append(
|
||||
f"optional dependency '{probe.name}' is {probe.status}: {probe.detail}"
|
||||
)
|
||||
|
||||
readiness_complete = not unrun_required
|
||||
ready = readiness_complete and not failed_required
|
||||
|
||||
if any(probe.status == STATUS_DOWN for probe in failed_required):
|
||||
status = STATUS_DOWN
|
||||
elif failed_required or failed_optional or unrun_required:
|
||||
status = STATUS_DEGRADED
|
||||
else:
|
||||
status = STATUS_OK
|
||||
return status, ready, readiness_complete, tuple(reasons)
|
||||
|
||||
|
||||
def load_system_health(
|
||||
*,
|
||||
deep: bool = False,
|
||||
host: str | None = None,
|
||||
probes: tuple[DependencyProbe, ...] | None = None,
|
||||
daemon_head: str | None = None,
|
||||
use_cache: bool = True,
|
||||
) -> SystemHealthSnapshot:
|
||||
"""Assemble the read-only system-health snapshot.
|
||||
|
||||
``deep=True`` adds the network probe against Gitea; its result is cached for
|
||||
a short TTL so repeated dashboard polls do not amplify into remote load.
|
||||
"""
|
||||
repo = _repo_root()
|
||||
probe_errors: list[str] = []
|
||||
|
||||
if probes is None:
|
||||
collected: list[DependencyProbe] = []
|
||||
for probe_fn in (
|
||||
lambda: probe_control_plane_db(),
|
||||
lambda: probe_repository(repo),
|
||||
):
|
||||
try:
|
||||
collected.append(probe_fn())
|
||||
except Exception as exc: # noqa: BLE001 — a probe must not 500 the API
|
||||
probe_errors.append(redact(f"probe raised: {exc}"))
|
||||
resolved_host = host if host is not None else _default_host()
|
||||
if deep and not _offline():
|
||||
collected.append(_cached_gitea_probe(resolved_host, use_cache=use_cache))
|
||||
else:
|
||||
collected.append(_skipped_gitea(resolved_host))
|
||||
probes = tuple(collected)
|
||||
|
||||
status, ready, readiness_complete, reasons = _aggregate(probes)
|
||||
stale = assess_stale_runtime(repo, daemon_head=daemon_head)
|
||||
if stale.stale:
|
||||
if status == STATUS_OK:
|
||||
status = STATUS_DEGRADED
|
||||
reasons = reasons + (
|
||||
"runtime is stale relative to its remote-tracking commit",
|
||||
)
|
||||
|
||||
db_probe = next((p for p in probes if p.name == "control_plane_db"), None)
|
||||
schema_version = None
|
||||
if db_probe and db_probe.metadata:
|
||||
schema_version = db_probe.metadata.get("schema_version")
|
||||
|
||||
return SystemHealthSnapshot(
|
||||
status=status,
|
||||
ready=ready,
|
||||
readiness_complete=readiness_complete,
|
||||
readiness_reasons=reasons,
|
||||
service=SERVICE_NAME,
|
||||
mode="read-only",
|
||||
version=_load_version(repo, schema_version=schema_version),
|
||||
started_at=_STARTED_AT.isoformat(),
|
||||
uptime_seconds=round(time.monotonic() - _STARTED_MONOTONIC, 3),
|
||||
timestamp=datetime.now(timezone.utc).isoformat(),
|
||||
deep_probes_requested=deep,
|
||||
dependencies=probes,
|
||||
mcp_namespaces=namespace_summaries(),
|
||||
stale_runtime=stale,
|
||||
probe_errors=tuple(probe_errors),
|
||||
)
|
||||
|
||||
|
||||
def _cached_gitea_probe(host: str, *, use_cache: bool = True) -> DependencyProbe:
|
||||
ttl = _deep_probe_ttl()
|
||||
now = time.monotonic()
|
||||
if use_cache and ttl > 0:
|
||||
cached = _deep_cache.get(host)
|
||||
if cached and (now - cached[0]) < ttl:
|
||||
return cached[1]
|
||||
probe = probe_gitea(host)
|
||||
if use_cache and ttl > 0:
|
||||
_deep_cache[host] = (now, probe)
|
||||
return probe
|
||||
|
||||
|
||||
def clear_probe_cache() -> None:
|
||||
"""Drop cached deep-probe results (tests and operator-forced refresh)."""
|
||||
_deep_cache.clear()
|
||||
|
||||
|
||||
def probe_to_dict(probe: DependencyProbe) -> dict[str, Any]:
|
||||
return {
|
||||
"name": probe.name,
|
||||
"kind": probe.kind,
|
||||
"status": probe.status,
|
||||
"detail": probe.detail,
|
||||
"required": probe.required,
|
||||
"healthy": probe.healthy,
|
||||
"latency_ms": probe.latency_ms,
|
||||
"metadata": dict(probe.metadata or {}),
|
||||
}
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: SystemHealthSnapshot) -> dict[str, Any]:
|
||||
return {
|
||||
"status": snapshot.status,
|
||||
"service": snapshot.service,
|
||||
"mode": snapshot.mode,
|
||||
"api": API_PATH,
|
||||
"timestamp": snapshot.timestamp,
|
||||
"readiness": {
|
||||
"ready": snapshot.ready,
|
||||
"complete": snapshot.readiness_complete,
|
||||
"reasons": list(snapshot.readiness_reasons),
|
||||
},
|
||||
"version": {
|
||||
"git_sha": snapshot.version.git_sha,
|
||||
"git_describe": snapshot.version.git_describe,
|
||||
"control_plane_schema_version": (
|
||||
snapshot.version.control_plane_schema_version
|
||||
),
|
||||
"python_version": snapshot.version.python_version,
|
||||
"known": snapshot.version.known,
|
||||
},
|
||||
"process": {
|
||||
"started_at": snapshot.started_at,
|
||||
"uptime_seconds": snapshot.uptime_seconds,
|
||||
},
|
||||
"deep_probes_requested": snapshot.deep_probes_requested,
|
||||
"dependencies": [probe_to_dict(probe) for probe in snapshot.dependencies],
|
||||
"mcp_namespaces": [dict(row) for row in snapshot.mcp_namespaces],
|
||||
"stale_runtime": {
|
||||
"daemon_head": snapshot.stale_runtime.daemon_head,
|
||||
"checkout_head": snapshot.stale_runtime.checkout_head,
|
||||
"remote_head": snapshot.stale_runtime.remote_head,
|
||||
"stale": snapshot.stale_runtime.stale,
|
||||
"determinable": snapshot.stale_runtime.determinable,
|
||||
"mutation_safe": snapshot.stale_runtime.mutation_safe,
|
||||
"reasons": list(snapshot.stale_runtime.reasons),
|
||||
},
|
||||
"probe_errors": list(snapshot.probe_errors),
|
||||
}
|
||||
@@ -0,0 +1,906 @@
|
||||
"""Workflow-event and conversation timeline model (#637, Phase 1).
|
||||
|
||||
Operators cannot browse a unified timeline of workflow events, decisions,
|
||||
tool calls, and handoffs: the evidence is scattered across control-plane
|
||||
events, Gitea canonical handoff comments, and local logs. This module defines
|
||||
one durable, versioned event schema and per-source adapters that normalise
|
||||
those scattered records into a single ``WorkflowEvent`` stream, plus a
|
||||
read-only query layer (filter by issue / PR / session, stable ordering,
|
||||
pagination) that the ``/api/v1/timeline`` route serves.
|
||||
|
||||
Design rules honoured here:
|
||||
|
||||
- **Read-only.** Sources are read; nothing is mutated. The control-plane
|
||||
database is opened through a ``mode=ro`` URI so a missing or unwritable DB
|
||||
degrades to a reason instead of creating directories or running migrations.
|
||||
- **Fail-soft per source.** An unavailable source degrades to a status with a
|
||||
reason rather than raising, and a source that could not run is never
|
||||
rendered as an empty-and-healthy timeline.
|
||||
- **Answerable filters only.** Each source declares which filter dimensions it
|
||||
can actually answer. A filter dimension no source that ran can carry is
|
||||
refused with an explicit reason rather than silently matching nothing: an
|
||||
empty page from an unanswerable filter reads to an operator as "no such
|
||||
activity", which is a different — and false — statement.
|
||||
- **Redaction at the boundary, fail closed.** Every free-text field (event
|
||||
messages, redacted tool arguments, decision/proof text) is run through the
|
||||
console redaction policy before it leaves this module, and *before* any
|
||||
structured value is derived from it — evidence references are extracted from
|
||||
redacted text, then independently revalidated before serialization. An
|
||||
unredactable value becomes the placeholder, and a value that cannot be proven
|
||||
safe is dropped — an unredacted payload is never emitted, and a generation
|
||||
error never drops raw data to a caller or a log.
|
||||
- **Stable ordering.** Events sort by ``(timestamp, source_rank, event_key)``
|
||||
with a deterministic tiebreak, so pagination is stable across calls and
|
||||
events with equal or missing timestamps keep a fixed order.
|
||||
|
||||
Non-goals (from the issue): no full chat replay, no mutation of historical
|
||||
events, no unredacted tool-argument storage.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import sqlite3
|
||||
from dataclasses import dataclass, replace
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable, Iterable
|
||||
|
||||
import control_plane_db
|
||||
from webui import console_redaction
|
||||
|
||||
# The schema is versioned so consumers can branch on shape. Bump on any
|
||||
# breaking change to WorkflowEvent's serialized form.
|
||||
TIMELINE_SCHEMA_VERSION = 1
|
||||
|
||||
# Known event sources and their deterministic ordering rank. When two events
|
||||
# carry the same timestamp, the source rank breaks the tie before the
|
||||
# per-source event key, so a control-plane event and a handoff comment minted
|
||||
# in the same second always sort in a fixed order.
|
||||
SOURCE_CONTROL_PLANE = "control_plane"
|
||||
SOURCE_GITEA_HANDOFF = "gitea_handoff"
|
||||
_SOURCE_RANK = {
|
||||
SOURCE_CONTROL_PLANE: 0,
|
||||
SOURCE_GITEA_HANDOFF: 1,
|
||||
}
|
||||
|
||||
# The filter dimensions the query layer accepts.
|
||||
FILTER_ISSUE = "issue"
|
||||
FILTER_PR = "pr"
|
||||
FILTER_SESSION = "session"
|
||||
|
||||
# Which dimensions each source can actually answer. This is a property of the
|
||||
# underlying records, not of the query code: the control-plane ``events`` table
|
||||
# is (event_id, work_item_id, event_type, message, created_at) and carries no
|
||||
# session identity at all, so no control-plane event can ever match a session
|
||||
# filter. A CTH handoff comment can declare its session as a field, so the
|
||||
# handoff source answers all three. Filtering on a dimension the surviving
|
||||
# sources cannot carry is refused in ``load_timeline`` rather than answered
|
||||
# with an empty page.
|
||||
_SOURCE_FILTER_SUPPORT: dict[str, tuple[str, ...]] = {
|
||||
SOURCE_CONTROL_PLANE: (FILTER_ISSUE, FILTER_PR),
|
||||
SOURCE_GITEA_HANDOFF: (FILTER_ISSUE, FILTER_PR, FILTER_SESSION),
|
||||
}
|
||||
|
||||
# Why a source cannot answer a dimension, for the refusal reason an operator reads.
|
||||
_SOURCE_FILTER_LIMITS: dict[tuple[str, str], str] = {
|
||||
(SOURCE_CONTROL_PLANE, FILTER_SESSION): (
|
||||
"control-plane events carry no session identity "
|
||||
"(the events table has no session column)"
|
||||
),
|
||||
}
|
||||
|
||||
# A timestamp far in the future so events with no parseable timestamp sort
|
||||
# last (after everything real) instead of first, without raising.
|
||||
_MISSING_TS_SORT = "9999-12-31T23:59:59Z"
|
||||
|
||||
|
||||
def _parse_ts(value: str | None) -> str | None:
|
||||
"""Normalise a timestamp to ``...Z`` UTC, or None when unparseable."""
|
||||
if not value:
|
||||
return None
|
||||
text = str(value).strip()
|
||||
if not text:
|
||||
return None
|
||||
candidate = text[:-1] + "+00:00" if text.endswith("Z") else text
|
||||
try:
|
||||
parsed = datetime.fromisoformat(candidate)
|
||||
except ValueError:
|
||||
return None
|
||||
if parsed.tzinfo is None:
|
||||
parsed = parsed.replace(tzinfo=timezone.utc)
|
||||
return parsed.astimezone(timezone.utc).replace(microsecond=0).isoformat().replace("+00:00", "Z")
|
||||
|
||||
|
||||
def _redact(value: Any) -> Any:
|
||||
"""Redact a single free-text field, failing closed to the placeholder."""
|
||||
if value is None:
|
||||
return None
|
||||
return console_redaction.redact_text(str(value))
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class WorkflowEvent:
|
||||
"""One normalised timeline event.
|
||||
|
||||
Every field is optional except ``source``/``event_type``/``event_key``
|
||||
because sources carry different subsets. The class is frozen so an adapted
|
||||
event is an immutable record; a consumer that needs a variant builds a new
|
||||
one rather than mutating history.
|
||||
"""
|
||||
|
||||
source: str
|
||||
event_type: str
|
||||
event_key: str
|
||||
timestamp: str | None = None
|
||||
actor: str | None = None
|
||||
role: str | None = None
|
||||
issue_number: int | None = None
|
||||
pr_number: int | None = None
|
||||
session_id: str | None = None
|
||||
tool_name: str | None = None
|
||||
decision: str | None = None
|
||||
message: str | None = None
|
||||
correlation_id: str | None = None
|
||||
evidence_refs: tuple[str, ...] = ()
|
||||
sensitive: bool = False
|
||||
|
||||
def sort_key(self) -> tuple[str, int, str]:
|
||||
return (
|
||||
self.timestamp or _MISSING_TS_SORT,
|
||||
_SOURCE_RANK.get(self.source, 99),
|
||||
self.event_key,
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"source": self.source,
|
||||
"event_type": self.event_type,
|
||||
"event_key": self.event_key,
|
||||
"timestamp": self.timestamp,
|
||||
"actor": self.actor,
|
||||
"role": self.role,
|
||||
"issue_number": self.issue_number,
|
||||
"pr_number": self.pr_number,
|
||||
"session_id": self.session_id,
|
||||
"tool_name": self.tool_name,
|
||||
"decision": self.decision,
|
||||
"message": self.message,
|
||||
"correlation_id": self.correlation_id,
|
||||
"evidence_refs": list(self.evidence_refs),
|
||||
"sensitive": self.sensitive,
|
||||
}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Adapters — pure functions from a source's raw records to WorkflowEvents. #
|
||||
# Each is total: a malformed record is skipped, never raised on. #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
# Event types whose payload is treated as sensitive and always redaction-hard
|
||||
# (they can carry lease/session provenance or tool arguments).
|
||||
_SENSITIVE_EVENT_HINTS = ("lease", "capability", "token", "auth", "secret")
|
||||
|
||||
# Reference tokens (issue/PR/comment ids) and SHAs parsed out of proof text.
|
||||
_EVIDENCE_REF_RE = re.compile(r"(?:#|PR\s*#?|issue\s*#?|comment\s*#?)(\d+)", re.IGNORECASE)
|
||||
|
||||
# A commit reference is only recognised when the text *declares* it as one.
|
||||
# A bare lowercase hex run is not evidence of anything: at 40 characters it is
|
||||
# exactly the shape of a Gitea personal access token, and at 7 it also matches
|
||||
# ordinary words such as "defaced". Requiring an anchoring keyword keeps real
|
||||
# references ("commit abc1234", "at head a209756...", "base caaae9b6") usable
|
||||
# while refusing to lift an undeclared secret-shaped run out of free text.
|
||||
_SHA_RE = re.compile(
|
||||
r"(?i:\b(?:commit|sha|head|base|parent|revision|rev|merge[- ]base)\b[\s:=@#]*)"
|
||||
r"([0-9a-f]{7,40})\b"
|
||||
)
|
||||
|
||||
# Shapes a serialized evidence reference is allowed to take. Anything else is
|
||||
# dropped rather than emitted.
|
||||
_REF_ISSUE_SHAPE = re.compile(r"^#[0-9]{1,9}$")
|
||||
_REF_SHA_SHAPE = re.compile(r"^[0-9a-f]{7,40}$")
|
||||
|
||||
# A long undelimited hex run with no declaring context is treated as credential
|
||||
# material wherever it appears, never as an identifier.
|
||||
_BARE_SECRET_SHAPE = re.compile(r"^[0-9a-f]{32,}$")
|
||||
|
||||
# An event type reads like an identifier, but a stored one is externally
|
||||
# influenced: any producer that writes the control-plane ``events`` table
|
||||
# chooses the string. It reaches ``to_dict`` verbatim, so it is validated here
|
||||
# rather than trusted because of where it came from.
|
||||
_CP_EVENT_TYPE_SHAPE = re.compile(r"^[A-Za-z][A-Za-z0-9._:+-]{0,63}$")
|
||||
|
||||
# Emitted in place of a value that cannot be proven safe. Deliberately not a
|
||||
# plausible workflow type: an unsafe value is refused, never quietly rewritten
|
||||
# into a different valid-looking one that would misdescribe the record.
|
||||
UNSAFE_EVENT_TYPE = "unsafe:redacted"
|
||||
|
||||
# Emitted for a CTH heading that is not a declared member of ``CTH_TYPES``. The
|
||||
# contract is enforced on write (``format_cth_body``) and on assess; the read
|
||||
# path the timeline uses enforces it too rather than assuming it was.
|
||||
UNKNOWN_HANDOFF_EVENT_TYPE = "handoff:unrecognized"
|
||||
|
||||
# A source record id is a plain integer in both sources it comes from: the
|
||||
# control-plane ``events`` primary key and a Gitea comment id. ``event_key`` is
|
||||
# serialized verbatim and is the pagination tiebreak, so anything else is
|
||||
# refused rather than interpolated into it.
|
||||
_RECORD_ID_SHAPE = re.compile(r"^[0-9]{1,19}$")
|
||||
|
||||
|
||||
def _kind_to_numbers(kind: str | None, number: int | None) -> tuple[int | None, int | None]:
|
||||
"""Map a control-plane work-item (kind, number) to (issue_no, pr_no)."""
|
||||
if number is None:
|
||||
return (None, None)
|
||||
if kind == "pr":
|
||||
return (None, int(number))
|
||||
if kind == "issue":
|
||||
return (int(number), None)
|
||||
return (None, None)
|
||||
|
||||
|
||||
def _correlation_for(kind: str | None, number: int | None) -> str | None:
|
||||
if number is None or kind not in ("issue", "pr"):
|
||||
return None
|
||||
return f"{kind}#{number}"
|
||||
|
||||
|
||||
def _extract_evidence_refs(*texts: str | None) -> tuple[str, ...]:
|
||||
"""Extract issue/PR and declared-commit references from **redacted** text.
|
||||
|
||||
Callers must pass text that has already been through :func:`_redact`; this
|
||||
function derives a structured field from its input, so extracting ahead of
|
||||
redaction would republish whatever redaction was about to remove. Every
|
||||
reference is revalidated by :func:`_validated_evidence_refs` before it is
|
||||
serialized.
|
||||
"""
|
||||
refs: list[str] = []
|
||||
for text in texts:
|
||||
if not text:
|
||||
continue
|
||||
for match in _EVIDENCE_REF_RE.finditer(text):
|
||||
token = f"#{match.group(1)}"
|
||||
if token not in refs:
|
||||
refs.append(token)
|
||||
for match in _SHA_RE.finditer(text):
|
||||
token = match.group(1)
|
||||
if token not in refs:
|
||||
refs.append(token)
|
||||
return tuple(refs)
|
||||
|
||||
|
||||
def _validated_evidence_refs(refs: Iterable[str]) -> tuple[tuple[str, ...], bool]:
|
||||
"""Independently revalidate references immediately before serialization.
|
||||
|
||||
Extraction is not trusted on its own. A reference survives only when it has
|
||||
a known reference shape and is unchanged by a second redaction pass — a
|
||||
value the redaction policy would alter is credential material that must not
|
||||
be emitted as a structured field. A full 40-character SHA stays usable
|
||||
because extraction only accepts a hex run the source text explicitly
|
||||
declared as a commit. Returns ``(safe_refs, dropped_any)``; ``dropped_any``
|
||||
marks the event sensitive so the drop is visible rather than silent.
|
||||
"""
|
||||
safe: list[str] = []
|
||||
dropped = False
|
||||
for ref in refs or ():
|
||||
try:
|
||||
token = str(ref).strip()
|
||||
if not token:
|
||||
continue
|
||||
recognised = bool(_REF_ISSUE_SHAPE.match(token) or _REF_SHA_SHAPE.match(token))
|
||||
if not recognised:
|
||||
dropped = True
|
||||
continue
|
||||
if _redact(token) != token:
|
||||
dropped = True
|
||||
continue
|
||||
if token not in safe:
|
||||
safe.append(token)
|
||||
except Exception:
|
||||
# Fail closed: a reference that cannot be proven safe is dropped.
|
||||
dropped = True
|
||||
continue
|
||||
return (tuple(safe), dropped)
|
||||
|
||||
|
||||
def _safe_session_id(value: Any) -> str | None:
|
||||
"""Return a session identifier only when it is safe to emit.
|
||||
|
||||
The value is authoritative source data — a session the record names for
|
||||
itself — but it is still free text. It is dropped when redaction alters it
|
||||
or when it is a bare secret-shaped hex run, so a credential parked in a
|
||||
session field can never reach the payload or be echoed back by a filter.
|
||||
"""
|
||||
if value is None:
|
||||
return None
|
||||
text = str(value).strip()
|
||||
if not text:
|
||||
return None
|
||||
if _BARE_SECRET_SHAPE.match(text):
|
||||
return None
|
||||
return text if _redact(text) == text else None
|
||||
|
||||
|
||||
def _safe_record_id(value: Any) -> str | None:
|
||||
"""Return a source record id only when it is a plain numeric identifier.
|
||||
|
||||
``event_key`` is serialized verbatim and is the deterministic pagination
|
||||
tiebreak, so an id is interpolated into it only when it has the shape both
|
||||
real sources actually produce. A record whose identity cannot be trusted is
|
||||
refused by the caller rather than keyed on.
|
||||
"""
|
||||
if value is None or isinstance(value, bool):
|
||||
return None
|
||||
if isinstance(value, int):
|
||||
return str(value)
|
||||
text = str(value).strip()
|
||||
return text if _RECORD_ID_SHAPE.match(text) else None
|
||||
|
||||
|
||||
def _safe_cp_event_type(value: Any) -> tuple[str, bool]:
|
||||
"""Validate a stored control-plane event type. Returns ``(type, unsafe)``.
|
||||
|
||||
The stored value is externally influenced — whichever producer wrote the
|
||||
``events`` row chose the string — and ``to_dict`` serializes it verbatim, so
|
||||
it passes a boundary of its own instead of relying on the one ``message``
|
||||
passes. A value survives only when it is an ordinary identifier, is not a
|
||||
bare secret-shaped hex run, and is unchanged by a redaction pass. Anything
|
||||
else fails closed to :data:`UNSAFE_EVENT_TYPE`: the record stays visible as
|
||||
an audit entry, but the value itself is never republished — not verbatim,
|
||||
not partially sanitized, and not rewritten into some other valid-looking
|
||||
type that would misdescribe what happened.
|
||||
"""
|
||||
text = ("" if value is None else str(value)).strip()
|
||||
if not text:
|
||||
return ("", False)
|
||||
if _BARE_SECRET_SHAPE.match(text):
|
||||
return (UNSAFE_EVENT_TYPE, True)
|
||||
if not _CP_EVENT_TYPE_SHAPE.match(text):
|
||||
return (UNSAFE_EVENT_TYPE, True)
|
||||
if _redact(text) != text:
|
||||
return (UNSAFE_EVENT_TYPE, True)
|
||||
return (text, False)
|
||||
|
||||
|
||||
def _safe_echo(value: Any) -> Any:
|
||||
"""Guard a scalar that is echoed back rather than derived from a record.
|
||||
|
||||
Query scope and filter values are caller-supplied and are reflected in the
|
||||
response so an operator can see what was asked. Reflection is still
|
||||
emission: a value redaction would alter, or a bare secret-shaped hex run, is
|
||||
replaced by the placeholder instead of being echoed verbatim. Ordinary
|
||||
scope and filter values pass through untouched.
|
||||
"""
|
||||
if value is None or isinstance(value, (int, bool)):
|
||||
return value
|
||||
text = str(value)
|
||||
if _BARE_SECRET_SHAPE.match(text.strip()):
|
||||
return console_redaction.REDACTED
|
||||
return _redact(text)
|
||||
|
||||
|
||||
def adapt_cp_events(rows: Iterable[dict[str, Any]]) -> list[WorkflowEvent]:
|
||||
"""Adapt control-plane ``events`` rows (joined to work_items) into events.
|
||||
|
||||
Each row is expected to carry ``event_id``, ``event_type``, ``message``,
|
||||
``created_at`` and the joined work-item ``kind``/``number``. Rows missing
|
||||
an id or type are skipped so a partially written table never raises.
|
||||
"""
|
||||
events: list[WorkflowEvent] = []
|
||||
for row in rows or []:
|
||||
try:
|
||||
event_id = _safe_record_id(row.get("event_id"))
|
||||
raw_event_type = (row.get("event_type") or "").strip()
|
||||
if event_id is None or not raw_event_type:
|
||||
continue
|
||||
# The stored type is source data, not a trusted constant: validate
|
||||
# it before it is serialized, exactly as `message` below is redacted
|
||||
# before it is serialized.
|
||||
event_type, event_type_unsafe = _safe_cp_event_type(raw_event_type)
|
||||
kind = row.get("kind")
|
||||
number = row.get("number")
|
||||
issue_no, pr_no = _kind_to_numbers(kind, number)
|
||||
sensitive = event_type_unsafe or any(
|
||||
hint in raw_event_type.lower() for hint in _SENSITIVE_EVENT_HINTS
|
||||
)
|
||||
events.append(
|
||||
WorkflowEvent(
|
||||
source=SOURCE_CONTROL_PLANE,
|
||||
event_type=event_type,
|
||||
event_key=f"cp:{event_id}",
|
||||
timestamp=_parse_ts(row.get("created_at")),
|
||||
issue_number=issue_no,
|
||||
pr_number=pr_no,
|
||||
# No session_id: the control-plane events table is
|
||||
# (event_id, work_item_id, event_type, message, created_at)
|
||||
# and records no session. Inventing one from the work item
|
||||
# or the message text would be a guess, so this source
|
||||
# declares the session dimension unsupported instead
|
||||
# (_SOURCE_FILTER_SUPPORT) and the query layer refuses a
|
||||
# session filter it cannot honestly answer.
|
||||
message=_redact(row.get("message")),
|
||||
correlation_id=_correlation_for(kind, number),
|
||||
sensitive=sensitive,
|
||||
)
|
||||
)
|
||||
except Exception:
|
||||
# A single malformed row must not sink the whole adaptation.
|
||||
continue
|
||||
return events
|
||||
|
||||
|
||||
def adapt_cth_comments(
|
||||
comments: Iterable[dict[str, Any]],
|
||||
*,
|
||||
kind: str,
|
||||
number: int,
|
||||
) -> list[WorkflowEvent]:
|
||||
"""Adapt Gitea Canonical Thread Handoff (CTH) comments into events.
|
||||
|
||||
Only comments that parse as a CTH (``canonical_thread_handoff.parse_cth_comment``)
|
||||
become events; ordinary comments are ignored. ``kind``/``number`` scope the
|
||||
events to the issue or PR the comments belong to.
|
||||
"""
|
||||
# Imported lazily so this module has no import-time dependency on the
|
||||
# handoff parser when only the control-plane adapter is used.
|
||||
from canonical_thread_handoff import is_known_cth_type, parse_cth_comment
|
||||
|
||||
# ``kind``/``number`` are interpolated into event_key and correlation_id, so
|
||||
# they are normalised once here. A scope this adapter cannot express is
|
||||
# refused outright rather than serialized into an identifier.
|
||||
kind = (kind or "").strip().lower()
|
||||
if kind not in ("issue", "pr"):
|
||||
return []
|
||||
try:
|
||||
number = int(number)
|
||||
except (TypeError, ValueError):
|
||||
return []
|
||||
|
||||
issue_no, pr_no = _kind_to_numbers(kind, number)
|
||||
correlation = _correlation_for(kind, number)
|
||||
events: list[WorkflowEvent] = []
|
||||
for comment in comments or []:
|
||||
try:
|
||||
body = comment.get("body") or ""
|
||||
parsed = parse_cth_comment(body)
|
||||
if not parsed:
|
||||
continue
|
||||
fields = parsed.get("fields") or {}
|
||||
cth_type = parsed.get("cth_type") or ""
|
||||
comment_id = _safe_record_id(comment.get("id"))
|
||||
if comment_id is None:
|
||||
continue
|
||||
# The CTH heading is free text: the parser accepts whatever follows
|
||||
# "## CTH:", and only the write and assess paths check it against
|
||||
# the contract. Check it here too — an unrecognised heading is
|
||||
# reported as such rather than serialized into event_type, so
|
||||
# arbitrary, malformed, or secret-shaped heading content has no way
|
||||
# through. Declared types are preserved exactly.
|
||||
cth_type_known = is_known_cth_type(cth_type)
|
||||
# Redaction runs first, and every derived value is taken from the
|
||||
# redacted text — deriving evidence refs from the raw proof would
|
||||
# re-emit exactly what redaction was about to remove.
|
||||
decision = _redact(fields.get("decision"))
|
||||
proof = _redact(fields.get("proof"))
|
||||
next_action = _redact(fields.get("next action"))
|
||||
refs, refs_dropped = _validated_evidence_refs(
|
||||
_extract_evidence_refs(proof, decision)
|
||||
)
|
||||
events.append(
|
||||
WorkflowEvent(
|
||||
source=SOURCE_GITEA_HANDOFF,
|
||||
event_type=(
|
||||
f"handoff:{cth_type.strip()}"
|
||||
if cth_type_known
|
||||
else UNKNOWN_HANDOFF_EVENT_TYPE
|
||||
),
|
||||
event_key=f"cth:{kind}:{number}:{comment_id}",
|
||||
timestamp=_parse_ts(comment.get("created_at")),
|
||||
actor=_redact((comment.get("user") or {}).get("login")),
|
||||
role=_redact(fields.get("next owner")),
|
||||
issue_number=issue_no,
|
||||
pr_number=pr_no,
|
||||
# A CTH names its own session when the producer records one;
|
||||
# it is read from that declared field, never inferred from
|
||||
# unrelated text.
|
||||
session_id=_safe_session_id(fields.get("session")),
|
||||
decision=decision,
|
||||
message=next_action or _redact(fields.get("status")),
|
||||
correlation_id=correlation,
|
||||
evidence_refs=refs,
|
||||
sensitive=refs_dropped or not cth_type_known,
|
||||
)
|
||||
)
|
||||
except Exception:
|
||||
continue
|
||||
return events
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Read-only control-plane event source. #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
_CP_EVENTS_QUERY = """
|
||||
SELECT e.event_id AS event_id,
|
||||
e.event_type AS event_type,
|
||||
e.message AS message,
|
||||
e.created_at AS created_at,
|
||||
w.kind AS kind,
|
||||
w.number AS number
|
||||
FROM events e
|
||||
JOIN work_items w ON e.work_item_id = w.work_item_id
|
||||
WHERE w.remote = ? AND w.org = ? AND w.repo = ?
|
||||
"""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SourceStatus:
|
||||
"""Fail-soft status for one timeline source.
|
||||
|
||||
``supported_filters`` states which filter dimensions this source's records
|
||||
can carry; ``unsupported_filters`` names the requested dimensions it cannot,
|
||||
so an operator can see *why* a source contributed nothing rather than being
|
||||
left to read an empty list as an absence of activity.
|
||||
"""
|
||||
|
||||
name: str
|
||||
ok: bool
|
||||
reason: str | None = None
|
||||
count: int = 0
|
||||
supported_filters: tuple[str, ...] = ()
|
||||
unsupported_filters: tuple[str, ...] = ()
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"name": self.name,
|
||||
"ok": self.ok,
|
||||
"reason": self.reason,
|
||||
"count": self.count,
|
||||
"supported_filters": list(self.supported_filters),
|
||||
"unsupported_filters": list(self.unsupported_filters),
|
||||
}
|
||||
|
||||
|
||||
def _cp_status(*, ok: bool, reason: str | None = None, count: int = 0) -> SourceStatus:
|
||||
return SourceStatus(
|
||||
SOURCE_CONTROL_PLANE,
|
||||
ok=ok,
|
||||
# A failure reason is serialized like any other field and is often an
|
||||
# exception string carrying a path or a transport error, so it crosses
|
||||
# the redaction boundary too. Static reasons pass through unchanged.
|
||||
reason=_redact(reason),
|
||||
count=count,
|
||||
supported_filters=_SOURCE_FILTER_SUPPORT[SOURCE_CONTROL_PLANE],
|
||||
)
|
||||
|
||||
|
||||
def _handoff_status(*, ok: bool, reason: str | None = None, count: int = 0) -> SourceStatus:
|
||||
return SourceStatus(
|
||||
SOURCE_GITEA_HANDOFF,
|
||||
ok=ok,
|
||||
# Same boundary as the control-plane status: this reason can quote an
|
||||
# error raised by a live authenticated fetch.
|
||||
reason=_redact(reason),
|
||||
count=count,
|
||||
supported_filters=_SOURCE_FILTER_SUPPORT[SOURCE_GITEA_HANDOFF],
|
||||
)
|
||||
|
||||
|
||||
def read_cp_events(
|
||||
*,
|
||||
remote: str,
|
||||
org: str,
|
||||
repo: str,
|
||||
db_path: str | None = None,
|
||||
) -> tuple[list[WorkflowEvent], SourceStatus]:
|
||||
"""Read scoped control-plane events read-only. Never creates the DB.
|
||||
|
||||
Opens the SQLite file through a ``mode=ro`` URI: a health/timeline read
|
||||
must never create directories or run the schema migration that
|
||||
``ControlPlaneDB()`` performs on construction. A missing or unreadable DB
|
||||
degrades to a status with a reason.
|
||||
"""
|
||||
path = (db_path or control_plane_db.default_db_path()).strip()
|
||||
conn: sqlite3.Connection | None = None
|
||||
try:
|
||||
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
||||
conn.row_factory = sqlite3.Row
|
||||
cursor = conn.execute(_CP_EVENTS_QUERY, (remote, org, repo))
|
||||
rows = [dict(r) for r in cursor.fetchall()]
|
||||
except sqlite3.OperationalError as exc:
|
||||
return ([], _cp_status(ok=False, reason=f"control-plane DB unavailable: {exc}"))
|
||||
except sqlite3.Error as exc:
|
||||
return ([], _cp_status(ok=False, reason=f"control-plane read failed: {exc}"))
|
||||
finally:
|
||||
if conn is not None:
|
||||
conn.close()
|
||||
events = adapt_cp_events(rows)
|
||||
return (events, _cp_status(ok=True, count=len(events)))
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Filter, sort, paginate. #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def filter_events(
|
||||
events: Iterable[WorkflowEvent],
|
||||
*,
|
||||
issue: int | None = None,
|
||||
pr: int | None = None,
|
||||
session: str | None = None,
|
||||
) -> list[WorkflowEvent]:
|
||||
"""Filter events by issue number, PR number, and/or session id.
|
||||
|
||||
Filters are conjunctive. A filter that names a dimension an event does not
|
||||
carry excludes that event (an issue filter excludes PR-only events).
|
||||
"""
|
||||
out: list[WorkflowEvent] = []
|
||||
for ev in events:
|
||||
if issue is not None and ev.issue_number != issue:
|
||||
continue
|
||||
if pr is not None and ev.pr_number != pr:
|
||||
continue
|
||||
if session is not None and ev.session_id != session:
|
||||
continue
|
||||
out.append(ev)
|
||||
return out
|
||||
|
||||
|
||||
def sort_events(events: Iterable[WorkflowEvent]) -> list[WorkflowEvent]:
|
||||
"""Return events in stable timeline order (ascending)."""
|
||||
return sorted(events, key=lambda ev: ev.sort_key())
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TimelinePage:
|
||||
"""One page of the sorted, filtered timeline."""
|
||||
|
||||
events: tuple[WorkflowEvent, ...]
|
||||
total: int
|
||||
limit: int
|
||||
offset: int
|
||||
|
||||
@property
|
||||
def next_offset(self) -> int | None:
|
||||
nxt = self.offset + len(self.events)
|
||||
return nxt if nxt < self.total else None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"events": [ev.to_dict() for ev in self.events],
|
||||
"pagination": {
|
||||
"total": self.total,
|
||||
"limit": self.limit,
|
||||
"offset": self.offset,
|
||||
"returned": len(self.events),
|
||||
"next_offset": self.next_offset,
|
||||
"has_more": self.next_offset is not None,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
_MAX_LIMIT = 500
|
||||
_DEFAULT_LIMIT = 50
|
||||
|
||||
|
||||
def _coerce_bounds(limit: int | None, offset: int | None) -> tuple[int, int]:
|
||||
try:
|
||||
lim = int(limit) if limit is not None else _DEFAULT_LIMIT
|
||||
except (TypeError, ValueError):
|
||||
lim = _DEFAULT_LIMIT
|
||||
try:
|
||||
off = int(offset) if offset is not None else 0
|
||||
except (TypeError, ValueError):
|
||||
off = 0
|
||||
lim = max(1, min(lim, _MAX_LIMIT))
|
||||
off = max(0, off)
|
||||
return (lim, off)
|
||||
|
||||
|
||||
def paginate(events: list[WorkflowEvent], *, limit: int | None, offset: int | None) -> TimelinePage:
|
||||
lim, off = _coerce_bounds(limit, offset)
|
||||
window = events[off : off + lim]
|
||||
return TimelinePage(events=tuple(window), total=len(events), limit=lim, offset=off)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Composition — load_timeline aggregates all sources, fail-soft. #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
# A comment source is a callable that, given (kind, number), returns the raw
|
||||
# Gitea comment list for that issue/PR. The route supplies a live fail-soft
|
||||
# fetcher; tests supply a fixture. When None, the handoff source is reported as
|
||||
# not-run (never silently empty-and-healthy).
|
||||
CommentSource = Callable[[str, int], list[dict[str, Any]]]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TimelineSnapshot:
|
||||
"""One answered timeline query.
|
||||
|
||||
``ok`` is False when the query could not be answered as asked — currently
|
||||
when a requested filter dimension no surviving source can carry was
|
||||
supplied. The page is then empty *and* the snapshot says so, because an
|
||||
``ok`` empty page is a claim that no such activity exists.
|
||||
"""
|
||||
|
||||
schema_version: int
|
||||
remote: str
|
||||
org: str
|
||||
repo: str
|
||||
filters: dict[str, Any]
|
||||
page: TimelinePage
|
||||
sources: tuple[SourceStatus, ...]
|
||||
ok: bool = True
|
||||
error: dict[str, Any] | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"ok": self.ok,
|
||||
"error": self.error,
|
||||
"schema_version": self.schema_version,
|
||||
# Scope and filters are echoed caller input, not derived record
|
||||
# data. Reflecting a value is still emitting it, so both cross the
|
||||
# same boundary; ordinary scope and filter values are unchanged.
|
||||
"scope": {
|
||||
"remote": _safe_echo(self.remote),
|
||||
"org": _safe_echo(self.org),
|
||||
"repo": _safe_echo(self.repo),
|
||||
},
|
||||
"filters": {key: _safe_echo(value) for key, value in self.filters.items()},
|
||||
"sources": [s.to_dict() for s in self.sources],
|
||||
**self.page.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
def _unanswerable_reasons(
|
||||
statuses: Iterable[SourceStatus], unanswerable: Iterable[str]
|
||||
) -> list[dict[str, str]]:
|
||||
"""Explain, per source, why each unanswerable dimension went unanswered."""
|
||||
out: list[dict[str, str]] = []
|
||||
for status in statuses:
|
||||
for dim in unanswerable:
|
||||
if dim not in status.supported_filters:
|
||||
reason = _SOURCE_FILTER_LIMITS.get(
|
||||
(status.name, dim), f"this source's records carry no {dim} identity"
|
||||
)
|
||||
elif not status.ok:
|
||||
reason = (
|
||||
f"this source can carry {dim} but did not run: "
|
||||
f"{status.reason or 'unavailable'}"
|
||||
)
|
||||
else:
|
||||
continue
|
||||
out.append({"source": status.name, "filter": dim, "reason": reason})
|
||||
return out
|
||||
|
||||
|
||||
def load_timeline(
|
||||
*,
|
||||
remote: str,
|
||||
org: str,
|
||||
repo: str,
|
||||
issue: int | None = None,
|
||||
pr: int | None = None,
|
||||
session: str | None = None,
|
||||
limit: int | None = None,
|
||||
offset: int | None = None,
|
||||
db_path: str | None = None,
|
||||
comment_source: CommentSource | None = None,
|
||||
) -> TimelineSnapshot:
|
||||
"""Aggregate every timeline source into one filtered, paginated snapshot.
|
||||
|
||||
Sources are read independently and fail soft: an unavailable source
|
||||
contributes a ``SourceStatus`` with ``ok=False`` and a reason, and never
|
||||
collapses the whole timeline. The handoff source only runs when a specific
|
||||
issue or PR is requested (a handoff comment belongs to one thread) and a
|
||||
``comment_source`` is available; otherwise it is reported as ``not run``
|
||||
rather than as an empty-and-healthy source.
|
||||
|
||||
A filter dimension that no surviving source can carry — a ``session``
|
||||
filter when the only source that ran is the control plane, whose events
|
||||
record no session — is refused with ``ok=False`` and a structured error
|
||||
instead of being answered with an empty page.
|
||||
"""
|
||||
all_events: list[WorkflowEvent] = []
|
||||
statuses: list[SourceStatus] = []
|
||||
|
||||
cp_events, cp_status = read_cp_events(remote=remote, org=org, repo=repo, db_path=db_path)
|
||||
all_events.extend(cp_events)
|
||||
statuses.append(cp_status)
|
||||
|
||||
# Gitea handoff comments are thread-scoped: only fetch when the caller
|
||||
# narrowed to one issue or PR, and only when a source was provided.
|
||||
handoff_target: tuple[str, int] | None = None
|
||||
if pr is not None:
|
||||
handoff_target = ("pr", pr)
|
||||
elif issue is not None:
|
||||
handoff_target = ("issue", issue)
|
||||
|
||||
if handoff_target is None:
|
||||
statuses.append(
|
||||
_handoff_status(
|
||||
ok=False,
|
||||
reason="not run: handoff comments are thread-scoped; filter by issue or pr to include them",
|
||||
)
|
||||
)
|
||||
elif comment_source is None:
|
||||
statuses.append(
|
||||
_handoff_status(
|
||||
ok=False,
|
||||
reason="not run: no comment source configured for this timeline read",
|
||||
)
|
||||
)
|
||||
else:
|
||||
kind, number = handoff_target
|
||||
try:
|
||||
comments = comment_source(kind, number) or []
|
||||
handoff_events = adapt_cth_comments(comments, kind=kind, number=number)
|
||||
all_events.extend(handoff_events)
|
||||
statuses.append(_handoff_status(ok=True, count=len(handoff_events)))
|
||||
except Exception as exc: # fail soft: a fetch/parse error degrades this source only
|
||||
statuses.append(_handoff_status(ok=False, reason=f"handoff source failed: {exc}"))
|
||||
|
||||
requested = tuple(
|
||||
name
|
||||
for name, value in ((FILTER_ISSUE, issue), (FILTER_PR, pr), (FILTER_SESSION, session))
|
||||
if value is not None
|
||||
)
|
||||
statuses = [
|
||||
replace(
|
||||
status,
|
||||
unsupported_filters=tuple(
|
||||
dim for dim in requested if dim not in status.supported_filters
|
||||
),
|
||||
)
|
||||
for status in statuses
|
||||
]
|
||||
filters = {"issue": issue, "pr": pr, "session": session}
|
||||
|
||||
# A dimension is answerable only if a source that actually ran can carry it.
|
||||
# If none can, refuse: an empty page would assert "no such activity", which
|
||||
# is a claim this timeline is not in a position to make.
|
||||
answerable: set[str] = set()
|
||||
for status in statuses:
|
||||
if status.ok:
|
||||
answerable.update(status.supported_filters)
|
||||
unanswerable = tuple(dim for dim in requested if dim not in answerable)
|
||||
|
||||
if unanswerable:
|
||||
return TimelineSnapshot(
|
||||
schema_version=TIMELINE_SCHEMA_VERSION,
|
||||
remote=remote,
|
||||
org=org,
|
||||
repo=repo,
|
||||
filters=filters,
|
||||
page=paginate([], limit=limit, offset=offset),
|
||||
sources=tuple(statuses),
|
||||
ok=False,
|
||||
error={
|
||||
"code": "filter_not_supported",
|
||||
"unsupported_filters": list(unanswerable),
|
||||
"detail": (
|
||||
"no timeline source that ran can answer "
|
||||
+ ", ".join(f"'{dim}'" for dim in unanswerable)
|
||||
+ "; the result is refused rather than returned empty"
|
||||
),
|
||||
"sources": _unanswerable_reasons(statuses, unanswerable),
|
||||
},
|
||||
)
|
||||
|
||||
filtered = filter_events(all_events, issue=issue, pr=pr, session=session)
|
||||
ordered = sort_events(filtered)
|
||||
page = paginate(ordered, limit=limit, offset=offset)
|
||||
|
||||
return TimelineSnapshot(
|
||||
schema_version=TIMELINE_SCHEMA_VERSION,
|
||||
remote=remote,
|
||||
org=org,
|
||||
repo=repo,
|
||||
filters=filters,
|
||||
page=page,
|
||||
sources=tuple(statuses),
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: TimelineSnapshot) -> dict[str, Any]:
|
||||
return snapshot.to_dict()
|
||||
Reference in New Issue
Block a user