Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d5d121a21b | ||
|
|
1ca2b50406 | ||
|
|
2f4dec8323 | ||
|
|
3a9d634c17 | ||
|
|
a81db75402 | ||
|
|
2068bae341 | ||
|
|
7af40fb5ff | ||
|
|
619f679077 | ||
|
|
9517834913 | ||
|
|
824c42f7e3 | ||
|
|
578c44b685 | ||
|
|
3b68d15593 | ||
|
|
a4c73766f4 | ||
|
|
9f686253eb | ||
|
|
b2e28428a4 | ||
|
|
95e4aae287 | ||
|
|
dac40ab9b3 | ||
|
|
ccde9e8f11 | ||
|
|
1948d3dc21 | ||
|
|
069a9af7e6 | ||
|
|
1cbbde0089 | ||
|
|
0a78da39e5 |
+81
-17
@@ -133,10 +133,15 @@ ROLE_ACTIONS: dict[str, tuple[tuple[str, ...], tuple[str, ...]]] = {
|
||||
|
||||
|
||||
# Body phrases that prove an issue is an implementation container, not a
|
||||
# unit of direct author work (#844). Matched case-insensitively against the
|
||||
# issue body. Title alone is never sufficient (ordinary issues may mention
|
||||
# "epic" incidentally).
|
||||
# unit of direct author work (#844 / #854). Matched case-insensitively against
|
||||
# the issue body. Title alone is never sufficient (ordinary issues may mention
|
||||
# "epic", "roadmap", "vision", or "umbrella" incidentally).
|
||||
#
|
||||
# #854 extends the #844 marker set so product-vision (#652), phased-roadmap
|
||||
# (#653), and umbrella (#655) coordination records — which do not use the word
|
||||
# "epic" — are classified with the same semantic exclusion as epic containers.
|
||||
_CHILD_ONLY_BODY_MARKERS: tuple[str, ...] = (
|
||||
# Epic / child-only (#844, live #631)
|
||||
"implementation is delivered via child issues only",
|
||||
"implementation is delivered through child issues only",
|
||||
"implementation is delivered via child issues",
|
||||
@@ -150,9 +155,26 @@ _CHILD_ONLY_BODY_MARKERS: tuple[str, ...] = (
|
||||
"coordination container",
|
||||
"child-only container",
|
||||
"implementation is delegated to child",
|
||||
# Vision / roadmap / umbrella coordination (#854, live #652/#653/#655).
|
||||
# Prefer authoritative non-implementation / child-only scope language over
|
||||
# bare words like "roadmap" so ordinary implementable issues that mention
|
||||
# a parent vision or roadmap stay eligible.
|
||||
"do not implement features on this issue",
|
||||
"implementing features on this roadmap issue",
|
||||
"implementation is via linked children only",
|
||||
"no product feature claimed complete on this issue alone",
|
||||
"phased delivery roadmap and epic sequencing",
|
||||
"this issue is the enduring source of truth",
|
||||
"enduring source of truth for the",
|
||||
"canonical product vision — enduring source of truth",
|
||||
"canonical product vision - enduring source of truth",
|
||||
"state: vision-active",
|
||||
"state: roadmap-active",
|
||||
)
|
||||
|
||||
# Explicit epic / umbrella labels (structured evidence preferred over title).
|
||||
# Explicit epic / umbrella / vision / roadmap labels (structured evidence
|
||||
# preferred over title). Tracker alone is *not* included — ordinary issues
|
||||
# may carry a tracker label without being non-implementable containers.
|
||||
_EPIC_LABELS: frozenset[str] = frozenset(
|
||||
{
|
||||
"type:epic",
|
||||
@@ -161,6 +183,16 @@ _EPIC_LABELS: frozenset[str] = frozenset(
|
||||
"scope:epic",
|
||||
"type:umbrella",
|
||||
"umbrella",
|
||||
"kind:umbrella",
|
||||
"scope:umbrella",
|
||||
"type:vision",
|
||||
"vision",
|
||||
"kind:vision",
|
||||
"scope:vision",
|
||||
"type:roadmap",
|
||||
"roadmap",
|
||||
"kind:roadmap",
|
||||
"scope:roadmap",
|
||||
}
|
||||
)
|
||||
|
||||
@@ -222,16 +254,45 @@ class WorkCandidate:
|
||||
}
|
||||
|
||||
|
||||
def _title_container_prefix(title_l: str) -> str | None:
|
||||
"""Return a coordination-title prefix token if *title_l* uses one (#854).
|
||||
|
||||
Title prefixes alone never exclude; they only corroborate body/label
|
||||
evidence. Ordinary issues may say "roadmap" or "vision" mid-title.
|
||||
"""
|
||||
for prefix, token in (
|
||||
("epic:", "title_epic_prefix"),
|
||||
("epic ", "title_epic_prefix"),
|
||||
("umbrella:", "title_umbrella_prefix"),
|
||||
("umbrella ", "title_umbrella_prefix"),
|
||||
("roadmap:", "title_roadmap_prefix"),
|
||||
("roadmap ", "title_roadmap_prefix"),
|
||||
("product vision:", "title_vision_prefix"),
|
||||
("product vision ", "title_vision_prefix"),
|
||||
("vision:", "title_vision_prefix"),
|
||||
("vision ", "title_vision_prefix"),
|
||||
):
|
||||
if title_l.startswith(prefix):
|
||||
return token
|
||||
return None
|
||||
|
||||
|
||||
def classify_epic_or_child_only_container(
|
||||
c: WorkCandidate,
|
||||
) -> tuple[bool, str | None]:
|
||||
"""Return whether *c* is an epic / child-only implementation container (#844).
|
||||
"""Return whether *c* is a non-implementable coordination container (#844/#854).
|
||||
|
||||
Exclusion uses structured evidence first (labels, body scope language).
|
||||
A bare title containing the word "epic" is **not** enough — ordinary
|
||||
implementable issues may mention epics incidentally. A title that is
|
||||
explicitly prefixed ``Epic:`` only counts when the body also proves
|
||||
child-only / no-direct-implementation scope (or an epic label is present).
|
||||
A bare title containing the words "epic", "roadmap", "vision", or
|
||||
"umbrella" is **not** enough — ordinary implementable issues may mention
|
||||
those terms incidentally. Explicit title prefixes (``Epic:``, ``Roadmap:``,
|
||||
``Product vision:``, ``Umbrella:``) only count when the body also proves
|
||||
child-only / no-direct-implementation scope (or a container label is
|
||||
present).
|
||||
|
||||
Covers epic, product-vision, phased-roadmap, umbrella, and child-only
|
||||
records so the allocator never assigns coordination containers as direct
|
||||
author work.
|
||||
|
||||
PRs are never classified as containers here (they already have a head).
|
||||
"""
|
||||
@@ -245,25 +306,28 @@ def classify_epic_or_child_only_container(
|
||||
title_l = title.lower()
|
||||
|
||||
body_hits = [m for m in _CHILD_ONLY_BODY_MARKERS if m in body_l]
|
||||
title_epic_prefix = title_l.startswith("epic:") or title_l.startswith("epic ")
|
||||
title_prefix = _title_container_prefix(title_l)
|
||||
|
||||
if epic_label:
|
||||
detail = f"label={epic_label[0]}"
|
||||
if body_hits:
|
||||
detail = f"{detail}; body_marker={body_hits[0]!r}"
|
||||
if title_prefix:
|
||||
detail = f"{title_prefix}; {detail}"
|
||||
return True, detail
|
||||
|
||||
if body_hits:
|
||||
# Body proves child-only / umbrella scope. Title "Epic:" is corroborating
|
||||
# but not required — containers without the word still exclude.
|
||||
# Body proves child-only / vision / roadmap / umbrella scope. Title
|
||||
# prefixes are corroborating but not required — containers without the
|
||||
# title word still exclude.
|
||||
detail = f"body_marker={body_hits[0]!r}"
|
||||
if title_epic_prefix:
|
||||
detail = f"title_epic_prefix; {detail}"
|
||||
if title_prefix:
|
||||
detail = f"{title_prefix}; {detail}"
|
||||
return True, detail
|
||||
|
||||
# Title-only "Epic:" without body scope evidence is insufficient (#844 AC:
|
||||
# eligibility does not rely solely on the word "Epic" in a title).
|
||||
# Similarly, incidental "epic" mid-title without markers stays eligible.
|
||||
# Title-only coordination prefix without body scope evidence is
|
||||
# insufficient (#844/#854 AC: eligibility does not rely solely on a title
|
||||
# word). Incidental mid-title mentions without markers stay eligible.
|
||||
return False, None
|
||||
|
||||
|
||||
|
||||
+74
-6
@@ -57,6 +57,8 @@ status, onboarding checklist state, and the fail-closed error payloads (#635).
|
||||
| `/system-health` | System-health dashboard — readiness, version/uptime, dependencies, MCP namespaces, stale-runtime parity (#639) |
|
||||
| `/queue` | Live PR and issue queue dashboard (#429) |
|
||||
| `/api/queue` | JSON queue export with pagination metadata |
|
||||
| `/traffic` | Workflow traffic-control view — runnable, leased, blocked, needs-controller, terminal-complete (#640) |
|
||||
| `/api/traffic` | JSON traffic-control export with state classifications and next safe role actions |
|
||||
| `/projects` | Project registry list with status and onboarding progress (#427, #635) |
|
||||
| `/projects/{id}` | Project detail + onboarding checklist |
|
||||
| `/api/v1/projects` | Versioned JSON registry export (#635) |
|
||||
@@ -75,7 +77,9 @@ status, onboarding checklist state, and the fail-closed error payloads (#635).
|
||||
| `/api/actions/{id}/preview` | Mutation ledger preview (GET, read-only) |
|
||||
| `/leases` | Lease and collision visibility (#433) |
|
||||
| `/api/leases` | JSON lease/collision export |
|
||||
| `/sessions` | Phase 1 shell stub — session inventory (backed by #636) |
|
||||
| `/sessions` | Runtime and session view (#641) — health + inventory sessions/namespaces/worktrees |
|
||||
| `/api/sessions` | JSON export for the runtime/session view |
|
||||
| `/api/v1/sessions` | Versioned alias of `/api/sessions` |
|
||||
| `/inventory` | Phase 1 shell stub — unified inventory (backed by #636) |
|
||||
| `/timeline` | Phase 1 shell stub — workflow event timeline |
|
||||
| `/policy` | Phase 1 shell stub — capability/role policy placeholder |
|
||||
@@ -85,6 +89,35 @@ Most routes are GET-only. POST/PUT/PATCH/DELETE return `405` with
|
||||
`read-only-mvp`, except `/audit` and `/api/audit` which accept POST for
|
||||
local validator preview only (no Gitea mutations, no server-side storage).
|
||||
|
||||
### Traffic-control state vocabulary (#640)
|
||||
|
||||
The traffic view classifies each open issue/PR into exactly one bucket:
|
||||
|
||||
| Bucket | Meaning | Operator implication |
|
||||
|--------|---------|----------------------|
|
||||
| **runnable** | No active lease, no block reason, safe for its expected role | Next role may start work |
|
||||
| **leased** | Active author claim or reviewer PR lease | Do not stomp; wait or adopt via role tools |
|
||||
| **blocked** | Dependency, missing head pin, conflict, or unmet dependency | Author remediation first |
|
||||
| **needs_controller** | Contaminated, controller-only diagnosis, or `status:blocked` | Controller only |
|
||||
| **terminal_complete** | Reconciler / terminal-lock territory | Reconciler cleanup path |
|
||||
|
||||
`status:blocked` items route to **needs_controller**, not **blocked**:
|
||||
`expected_role_for_candidate` sends them to the controller, and the blocker
|
||||
reason renders in either bucket.
|
||||
|
||||
**Live path contracts (do not invent):**
|
||||
|
||||
- PR head pins come from `QueueItem.signals["head_sha"]` (full SHA). Display
|
||||
`extra["head_sha"]` is truncated and must never be used for routing.
|
||||
- Reviewer leases are keyed as `(pr, pr_number)` only — never via a linked
|
||||
`issue_number` on the same lease marker.
|
||||
- Issue claims come from `claim_inventory["entries"]`
|
||||
(`issue_claim_heartbeat.build_claim_inventory`). There is no `active_claims`
|
||||
key.
|
||||
- Queue display badges are only: `blocked`, `claimed`, `duplicate`, `stale`,
|
||||
`in-review`, `open`. Review verdicts (`request-changes`, `approved`) are
|
||||
**not** queue badges; traffic does not invent them from the queue loader.
|
||||
|
||||
## System health API (#634)
|
||||
|
||||
`GET /api/v1/system/health` is the structured, read-only health surface for
|
||||
@@ -253,11 +286,46 @@ The header carries two read-only status badges — an **environment** badge
|
||||
a **mode: read-only** badge — plus a **Docs** link to this document. No
|
||||
privileged action controls are present in the Phase 1 shell.
|
||||
|
||||
Not-yet-implemented surfaces (`/sessions`, `/inventory`, `/timeline`,
|
||||
`/policy`, `/insights`) resolve to graceful read-only stub pages instead of
|
||||
404s; their backing views land in later child issues of #631 (the inventory
|
||||
surfaces are backed by #636). Mutating methods on stub routes still fail closed
|
||||
with `read-only-mvp`.
|
||||
Not-yet-implemented surfaces (`/inventory`, `/timeline`, `/policy`,
|
||||
`/insights`) resolve to graceful read-only stub pages instead of 404s; their
|
||||
backing views land in later child issues of #631 (the inventory surfaces are
|
||||
backed by #636). Mutating methods on stub routes still fail closed with
|
||||
`read-only-mvp`.
|
||||
|
||||
### Runtime and sessions (#641)
|
||||
|
||||
`/sessions` is a live Phase 1 read-only view that composes:
|
||||
|
||||
* runtime health from `#430` (profile, role, identity, master parity, stale warning)
|
||||
* control-plane sessions / leases and filesystem locks / worktrees / namespaces from `#636`
|
||||
* durable contamination markers when detectable (`#630` runtime recovery, `#671` stable-branch push)
|
||||
|
||||
It surfaces stale indicators (dead PID, expired lease) and never silences an
|
||||
active contamination marker. Recovery links point only at sanctioned
|
||||
reconnect/operator restart docs (`docs/mcp-namespace-eof-recovery.md`,
|
||||
`docs/mcp-namespace-health.md`, `docs/mcp-restart-path-inventory.md`, this
|
||||
document). The page does **not** restart, kill, or take over sessions; manual
|
||||
`pkill` of MCP daemons is contamination, not recovery.
|
||||
|
||||
Honesty rules specific to this view:
|
||||
|
||||
* **Ownership columns never assert absence they cannot prove.** When the
|
||||
`leases` or `locks` section is degraded or unavailable, the Leases and
|
||||
Worktree-binding cells render `unknown (inventory <status>)` with an
|
||||
*authority unproven* badge instead of `none` / `unbound`, and a caveat names
|
||||
the unreadable sections. A worktree binding is correlated through lease work
|
||||
numbers, so it is unproven when *either* section fails to read.
|
||||
`/api/sessions` carries the same facts as `ownership_authority_complete`,
|
||||
`ownership_section_status`, and per-row `lease_authority` /
|
||||
`worktree_authority`, so a JSON consumer can tell "holds none" from "could
|
||||
not be read".
|
||||
* **Contamination text is redacted at the display boundary.** Marker payloads
|
||||
(`command_summary`, `reason_class`, `session_id`, `role`) are
|
||||
operator-supplied free text that does not arrive through inventory scrubbing,
|
||||
so they pass through `webui.inventory.scrub_text`, which collapses `$HOME` and
|
||||
redacts credential-shaped tokens and URL userinfo *anywhere* in the string.
|
||||
The write-time redactor is a narrow denylist and is not relied on. The field
|
||||
itself is kept — it is the `#630` evidence naming which daemon was killed.
|
||||
|
||||
## System-health dashboard (#639)
|
||||
|
||||
|
||||
+1015
File diff suppressed because it is too large
Load Diff
+58
-8
@@ -2068,6 +2068,7 @@ import lease_lifecycle # noqa: E402
|
||||
import lease_policy # noqa: E402
|
||||
import workflow_dashboard # noqa: E402 # #605 live queue/lease dashboard
|
||||
import restart_coordinator # noqa: E402 # #658 MCP restart coordinator/impact
|
||||
import drain_proof # noqa: E402 # #661 pre-restart drain proof and hard gate
|
||||
import incident_bridge # noqa: E402
|
||||
import sentry_observability # noqa: E402 (#606 optional Sentry observability)
|
||||
import sentry_incident_bridge # noqa: E402 (#607 Sentry→Gitea incident bridge)
|
||||
@@ -22342,6 +22343,8 @@ def gitea_request_mcp_restart(
|
||||
request_override: bool = False,
|
||||
session_id: str | None = None,
|
||||
limit: int = 200,
|
||||
drain_proof_json: str | None = None,
|
||||
request_break_glass: bool = False,
|
||||
) -> dict:
|
||||
"""Evaluate a proposed MCP restart and return an impact preview (#658).
|
||||
|
||||
@@ -22351,10 +22354,15 @@ def gitea_request_mcp_restart(
|
||||
verdict, so the console (#642/#652) and operators can see what a restart
|
||||
would disrupt *before* any concurrent LLM work is destroyed.
|
||||
|
||||
This tool is **dry-run and never restarts anything.** The mutative apply
|
||||
path is a separate child gated by a drain proof (non-goal here); calling
|
||||
with ``dry_run=False`` still performs no restart and reports that apply is
|
||||
not yet available.
|
||||
This tool **never restarts a process.** In dry-run (the default) it returns
|
||||
only the impact preview. With ``dry_run=False`` it enforces the #661 hard
|
||||
gate: the apply request must present a valid, unexpired, clean drain proof
|
||||
(``drain_proof_json``) or it is denied and a durable incident descriptor is
|
||||
returned under ``incident``. Break-glass is the only bypass and is honoured
|
||||
only when ``request_break_glass`` is set *and* the environment carries
|
||||
``GITEA_BREAKGLASS_RESTART_AUTHORIZATION``. Even an authorized gate performs
|
||||
no restart here; actual execution is a further child. The gate outcome is
|
||||
reported under ``apply_gate`` / ``apply_authorized``.
|
||||
|
||||
Operator override authority is read from the process environment
|
||||
(``GITEA_OPERATOR_RESTART_OVERRIDE_AUTHORIZATION``), never self-asserted by
|
||||
@@ -22473,12 +22481,54 @@ def gitea_request_mcp_restart(
|
||||
payload["requesting_session_id"] = sid
|
||||
payload["operator_override_requested"] = bool(request_override)
|
||||
payload["operator_override_authorized"] = operator_authorized
|
||||
# Actual restart execution remains a further child; this tool never restarts
|
||||
# a process. What #661 adds is the *hard gate*: an apply request (dry_run
|
||||
# False) must present a valid, unexpired, clean drain proof, or it is denied
|
||||
# and a durable incident is raised. Break-glass is the only bypass and its
|
||||
# authorization is read from the environment, never self-asserted.
|
||||
payload["apply_supported"] = False
|
||||
if not dry_run:
|
||||
payload["reasons"] = list(payload.get("reasons") or []) + [
|
||||
"apply requested but not supported: sanctioned restart apply is "
|
||||
"gated by a drain proof (separate child); no restart performed (#658)"
|
||||
]
|
||||
proof_obj: dict | None = None
|
||||
proof_parse_error: str | None = None
|
||||
if drain_proof_json:
|
||||
try:
|
||||
parsed = json.loads(drain_proof_json)
|
||||
proof_obj = parsed if isinstance(parsed, dict) else None
|
||||
if proof_obj is None:
|
||||
proof_parse_error = "drain_proof_json is not a JSON object"
|
||||
except (ValueError, TypeError) as exc:
|
||||
proof_parse_error = f"invalid drain_proof_json: {_redact(str(exc))}"
|
||||
|
||||
break_glass_authorized = bool(
|
||||
(
|
||||
os.environ.get("GITEA_BREAKGLASS_RESTART_AUTHORIZATION") or ""
|
||||
).strip()
|
||||
)
|
||||
break_glass = bool(request_break_glass and break_glass_authorized)
|
||||
|
||||
expected_fp = drain_proof.impact_fingerprint(report.as_dict())
|
||||
gate = drain_proof.gate_apply_restart(
|
||||
proof=proof_obj,
|
||||
break_glass=break_glass,
|
||||
expected_impact_fingerprint=expected_fp,
|
||||
requesting_session_id=sid,
|
||||
)
|
||||
gate_payload = gate.as_dict()
|
||||
if proof_parse_error and not break_glass:
|
||||
gate_payload["reasons"] = [proof_parse_error] + list(
|
||||
gate_payload.get("reasons") or []
|
||||
)
|
||||
payload["apply_gate"] = gate_payload
|
||||
payload["apply_authorized"] = gate.allow
|
||||
payload["break_glass_requested"] = bool(request_break_glass)
|
||||
payload["break_glass_authorized"] = break_glass_authorized
|
||||
# Even an authorized gate performs no restart here: execution is a later
|
||||
# child. The gate proves the apply path *would* be permitted.
|
||||
payload["reasons"] = list(payload.get("reasons") or []) + list(
|
||||
gate_payload.get("reasons") or []
|
||||
)
|
||||
if not gate.allow and gate.incident is not None:
|
||||
payload["incident"] = gate.incident
|
||||
return payload
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,907 @@
|
||||
"""Tests for the pre-restart drain proof and hard gate (#661).
|
||||
|
||||
Covers the acceptance criteria:
|
||||
|
||||
1. Restart apply without a proof fails closed.
|
||||
2. A successful drain produces a verifiable proof.
|
||||
3. An open unsafe mutation makes the proof fail (multi-session fixture).
|
||||
4. Pass / fail / expired verification paths.
|
||||
|
||||
Plus the security posture: forged/tampered proofs are rejected, break-glass is
|
||||
the only bypass and is never silent, a stale blast-radius fingerprint rejects a
|
||||
proof, and no per-process secret ever leaks into a serialized artifact.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
import drain_proof as dp
|
||||
import restart_coordinator as rc
|
||||
|
||||
|
||||
NOW = datetime(2026, 7, 24, 6, 0, 0, tzinfo=timezone.utc)
|
||||
SECRET = b"unit-test-drain-proof-secret-0123456789abcdef"
|
||||
|
||||
|
||||
def _live_pid() -> int:
|
||||
return os.getpid()
|
||||
|
||||
|
||||
def _clean_drain_state() -> dict:
|
||||
"""Every drain action succeeded, no sessions outstanding."""
|
||||
|
||||
return {
|
||||
"assignments_stopped": True,
|
||||
"checkpoints_complete": True,
|
||||
"handoffs_verified": True,
|
||||
"leases_handled": True,
|
||||
"acks": {}, # no other live sessions to acknowledge
|
||||
"ack_timeout_policy_applied": False,
|
||||
}
|
||||
|
||||
|
||||
def _safe_report() -> dict:
|
||||
"""Impact report with no other live work: a restart here is safe."""
|
||||
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
def _unsafe_mutation_report() -> dict:
|
||||
"""Multi-session report: a second session holds a live author mutation."""
|
||||
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-controller-1-req",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
{
|
||||
"session_id": "prgs-author-99",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
]
|
||||
leases = [
|
||||
{
|
||||
"lease_id": "lease-mut",
|
||||
"session_id": "prgs-author-99",
|
||||
"role": "author",
|
||||
"phase": "implementing",
|
||||
"work_kind": "issue",
|
||||
"work_number": 661,
|
||||
"worktree_path": "branches/issue-661",
|
||||
"freshness": {"freshness": "active"},
|
||||
}
|
||||
]
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": leases, "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
class BuildDrainProofTests(unittest.TestCase):
|
||||
def test_clean_drain_produces_verifiable_clean_proof(self):
|
||||
"""AC#2: a successful drain produces a verifiable proof."""
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
self.assertTrue(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [])
|
||||
self.assertEqual(
|
||||
{c.name for c in proof.checks}, set(dp.REQUIRED_CHECKS)
|
||||
)
|
||||
result = dp.verify_drain_proof(
|
||||
proof.as_dict(), now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertTrue(result.valid, result.reasons)
|
||||
self.assertFalse(result.expired)
|
||||
self.assertFalse(result.tampered)
|
||||
|
||||
def test_open_mutation_makes_proof_unclean(self):
|
||||
"""AC#3: an unsafe mutation still in flight fails the proof."""
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_NO_INFLIGHT_MUTATIONS, proof.failed_checks)
|
||||
# Leases-handled also fails: the report still shows a disruptive lease.
|
||||
self.assertIn(dp.CHECK_LEASES_HANDLED, proof.failed_checks)
|
||||
result = dp.verify_drain_proof(proof.as_dict(), now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
|
||||
def test_incomplete_inventory_fails_no_mutations_check(self):
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report={"inventory_complete": False},
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_NO_INFLIGHT_MUTATIONS, proof.failed_checks)
|
||||
|
||||
def test_missing_checkpoint_flag_fails_closed(self):
|
||||
state = _clean_drain_state()
|
||||
del state["checkpoints_complete"]
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_CHECKPOINTS_COMPLETE, proof.failed_checks)
|
||||
|
||||
def test_non_true_flags_fail_closed(self):
|
||||
"""A truthy-but-not-True value (e.g. the string 'yes') must not pass."""
|
||||
|
||||
state = _clean_drain_state()
|
||||
state["assignments_stopped"] = "yes"
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertIn(dp.CHECK_ASSIGNMENTS_STOPPED, proof.failed_checks)
|
||||
|
||||
def test_ack_timeout_policy_satisfies_ack_check(self):
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-author-99": "pending"}
|
||||
state["ack_timeout_policy_applied"] = True
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
names = {c.name: c.passed for c in proof.checks}
|
||||
self.assertTrue(names[dp.CHECK_ACKS_OR_TIMEOUT])
|
||||
|
||||
def test_outstanding_acks_without_timeout_fail(self):
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-author-99": "pending"}
|
||||
state["ack_timeout_policy_applied"] = False
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertIn(dp.CHECK_ACKS_OR_TIMEOUT, proof.failed_checks)
|
||||
|
||||
def test_all_acked_satisfies_ack_check(self):
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-author-99": "acked", "prgs-author-2": "acknowledged"}
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
names = {c.name: c.passed for c in proof.checks}
|
||||
self.assertTrue(names[dp.CHECK_ACKS_OR_TIMEOUT])
|
||||
|
||||
|
||||
class VerifyDrainProofTests(unittest.TestCase):
|
||||
def _clean_proof_dict(self) -> dict:
|
||||
return dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
|
||||
def test_missing_proof_is_invalid(self):
|
||||
result = dp.verify_drain_proof(None, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertIsNone(result.proof_id)
|
||||
|
||||
def test_expired_proof_is_invalid(self):
|
||||
"""AC#4: an expired proof fails verification."""
|
||||
|
||||
proof = self._clean_proof_dict()
|
||||
later = NOW + timedelta(seconds=dp.DEFAULT_PROOF_TTL_SECONDS + 1)
|
||||
result = dp.verify_drain_proof(proof, now=later, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.expired)
|
||||
|
||||
def test_proof_valid_just_before_expiry(self):
|
||||
proof = self._clean_proof_dict()
|
||||
almost = NOW + timedelta(seconds=dp.DEFAULT_PROOF_TTL_SECONDS - 1)
|
||||
result = dp.verify_drain_proof(proof, now=almost, secret=SECRET)
|
||||
self.assertTrue(result.valid, result.reasons)
|
||||
|
||||
def test_wrong_secret_rejected(self):
|
||||
"""A proof minted in a prior process (different secret) will not verify."""
|
||||
|
||||
proof = self._clean_proof_dict()
|
||||
result = dp.verify_drain_proof(proof, now=NOW, secret=b"other-secret")
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.tampered)
|
||||
|
||||
def test_flipping_clean_flag_is_detected(self):
|
||||
"""Forging clean=True on an unclean proof breaks the signature."""
|
||||
|
||||
unclean = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
self.assertFalse(unclean["clean"])
|
||||
unclean["clean"] = True # forge
|
||||
result = dp.verify_drain_proof(unclean, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.tampered)
|
||||
|
||||
def test_tampering_a_check_is_detected(self):
|
||||
unclean = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
for c in unclean["checks"]:
|
||||
if c["name"] == dp.CHECK_NO_INFLIGHT_MUTATIONS:
|
||||
c["passed"] = True # forge the failing check to pass
|
||||
result = dp.verify_drain_proof(unclean, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.tampered)
|
||||
|
||||
def test_missing_required_check_rejected(self):
|
||||
proof = self._clean_proof_dict()
|
||||
proof["checks"] = [
|
||||
c for c in proof["checks"] if c["name"] != dp.CHECK_HANDOFFS_OK
|
||||
]
|
||||
result = dp.verify_drain_proof(proof, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
|
||||
def test_stale_fingerprint_rejected(self):
|
||||
proof = self._clean_proof_dict()
|
||||
result = dp.verify_drain_proof(
|
||||
proof,
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint="deadbeef",
|
||||
)
|
||||
self.assertFalse(result.valid)
|
||||
|
||||
def test_matching_fingerprint_accepted(self):
|
||||
report = _safe_report()
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=report,
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
fp = dp.impact_fingerprint(report)
|
||||
result = dp.verify_drain_proof(
|
||||
proof, now=NOW, secret=SECRET, expected_impact_fingerprint=fp
|
||||
)
|
||||
self.assertTrue(result.valid, result.reasons)
|
||||
|
||||
|
||||
class GateApplyRestartTests(unittest.TestCase):
|
||||
def _clean_proof_dict(self) -> dict:
|
||||
return dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
|
||||
def test_apply_without_proof_denied(self):
|
||||
"""AC#1: restart apply without a proof fails closed + raises incident."""
|
||||
|
||||
decision = dp.gate_apply_restart(proof=None, now=NOW, secret=SECRET)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_DENY)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
self.assertEqual(
|
||||
decision.incident["kind"], "restart_drain_gate_denied"
|
||||
)
|
||||
|
||||
def test_apply_with_valid_proof_allowed(self):
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=self._clean_proof_dict(), now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertTrue(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_ALLOW)
|
||||
self.assertIsNone(decision.incident)
|
||||
|
||||
def test_apply_with_expired_proof_denied_with_incident(self):
|
||||
later = NOW + timedelta(seconds=dp.DEFAULT_PROOF_TTL_SECONDS + 5)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=self._clean_proof_dict(), now=later, secret=SECRET
|
||||
)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
|
||||
def test_apply_with_unclean_proof_denied(self):
|
||||
"""AC#3 at the gate: an unsafe-mutation proof is denied."""
|
||||
|
||||
unclean = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
decision = dp.gate_apply_restart(proof=unclean, now=NOW, secret=SECRET)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
|
||||
def test_break_glass_allows_without_proof_but_records_bypass(self):
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=None, now=NOW, secret=SECRET, break_glass=True
|
||||
)
|
||||
self.assertTrue(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_BREAK_GLASS)
|
||||
self.assertTrue(decision.break_glass)
|
||||
self.assertIsNone(decision.incident)
|
||||
self.assertTrue(decision.audit_record["break_glass"])
|
||||
|
||||
def test_denied_gate_carries_stale_fingerprint_reason(self):
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=self._clean_proof_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint="not-the-fingerprint",
|
||||
)
|
||||
self.assertFalse(decision.allow)
|
||||
|
||||
|
||||
class SecretHygieneTests(unittest.TestCase):
|
||||
def test_secret_never_serialized(self):
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
blob = dp._canonical(proof.as_dict())
|
||||
self.assertNotIn(SECRET.decode(), blob)
|
||||
# The signature is a hex digest, not the raw secret.
|
||||
self.assertNotIn(SECRET.hex(), blob)
|
||||
|
||||
def test_incident_descriptor_has_no_secret(self):
|
||||
decision = dp.gate_apply_restart(proof=None, now=NOW, secret=SECRET)
|
||||
blob = dp._canonical(decision.incident)
|
||||
self.assertNotIn(SECRET.decode(), blob)
|
||||
|
||||
|
||||
def _drained_report_with_live_sessions(count: int) -> dict:
|
||||
"""Report with ``count`` other live sessions but nothing in flight.
|
||||
|
||||
Every other checklist item passes against this report, so a failure
|
||||
isolates the acknowledgement check rather than tripping on mutations.
|
||||
"""
|
||||
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-controller-1-req",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
]
|
||||
for index in range(count):
|
||||
sessions.append(
|
||||
{
|
||||
"session_id": f"prgs-author-{index}",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
)
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
class AcknowledgementFailClosedTests(unittest.TestCase):
|
||||
"""Acknowledgement evidence must fail closed unless explicitly verified.
|
||||
|
||||
Regression cover for the reviewed fail-open on PR #882: an absent ``acks``
|
||||
key collapsed to ``{}`` and was read as "no other live sessions required to
|
||||
acknowledge", so a proof minted clean and the restart gate allowed while the
|
||||
impact report still showed other live sessions.
|
||||
"""
|
||||
|
||||
def _state(self, **overrides) -> dict:
|
||||
state = _clean_drain_state()
|
||||
state.pop("acks", None)
|
||||
state["ack_timeout_policy_applied"] = False
|
||||
state.update(overrides)
|
||||
return state
|
||||
|
||||
def _acks_check(self, proof) -> dp.DrainCheck:
|
||||
return next(c for c in proof.checks if c.name == dp.CHECK_ACKS_OR_TIMEOUT)
|
||||
|
||||
def _build(self, report: dict, state: dict):
|
||||
return dp.build_drain_proof(
|
||||
impact_report=report, drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
|
||||
def assertAcksFailClosed(self, report: dict, state: dict) -> None:
|
||||
proof = self._build(report, state)
|
||||
self.assertFalse(self._acks_check(proof).passed)
|
||||
self.assertIn(dp.CHECK_ACKS_OR_TIMEOUT, proof.failed_checks)
|
||||
self.assertFalse(proof.clean)
|
||||
|
||||
# --- missing / null / empty / malformed ------------------------------
|
||||
|
||||
def test_missing_acks_key_with_live_sessions_fails_closed(self):
|
||||
"""The exact reviewed defect: absent key, three other live sessions."""
|
||||
report = _drained_report_with_live_sessions(3)
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 3)
|
||||
state = self._state()
|
||||
self.assertNotIn("acks", state)
|
||||
proof = self._build(report, state)
|
||||
check = self._acks_check(proof)
|
||||
self.assertFalse(check.passed)
|
||||
self.assertNotIn("no other live sessions", check.detail)
|
||||
self.assertIn("fail closed", check.detail)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [dp.CHECK_ACKS_OR_TIMEOUT])
|
||||
|
||||
def test_none_acks_with_live_sessions_fails_closed(self):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(2), self._state(acks=None)
|
||||
)
|
||||
|
||||
def test_empty_acks_with_live_sessions_fails_closed(self):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(1), self._state(acks={})
|
||||
)
|
||||
|
||||
def test_malformed_acks_fail_closed(self):
|
||||
for malformed in ([], "ack", 7, ("ack",), True):
|
||||
with self.subTest(malformed=malformed):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(1),
|
||||
self._state(acks=malformed),
|
||||
)
|
||||
|
||||
# --- stale / unproven values -----------------------------------------
|
||||
|
||||
def test_stale_or_unproven_ack_values_fail_closed(self):
|
||||
for value in ("pending", "stale", "unknown", "", None, True, 1, NOW):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(1),
|
||||
self._state(acks={"prgs-author-0": value}),
|
||||
)
|
||||
|
||||
def test_partial_coverage_fails_closed(self):
|
||||
"""Fewer acknowledgements than the report's live-session count."""
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(3),
|
||||
self._state(acks={"prgs-author-0": "ack"}),
|
||||
)
|
||||
|
||||
def test_one_unacked_entry_among_many_fails_closed(self):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(2),
|
||||
self._state(acks={"prgs-author-0": "ack", "prgs-author-1": "pending"}),
|
||||
)
|
||||
|
||||
def test_unproven_live_session_count_fails_closed(self):
|
||||
"""A missing or malformed count cannot prove nobody had to acknowledge."""
|
||||
malformed_counts = (
|
||||
None,
|
||||
{},
|
||||
{"sessions_live_other": None},
|
||||
{"sessions_live_other": "3"},
|
||||
{"sessions_live_other": -1},
|
||||
{"sessions_live_other": True},
|
||||
)
|
||||
for counts in malformed_counts:
|
||||
with self.subTest(counts=counts):
|
||||
report = _drained_report_with_live_sessions(0)
|
||||
if counts is None:
|
||||
report.pop("counts", None)
|
||||
else:
|
||||
report["counts"] = counts
|
||||
self.assertAcksFailClosed(report, self._state())
|
||||
|
||||
# --- valid evidence still passes -------------------------------------
|
||||
|
||||
def test_complete_valid_acks_pass(self):
|
||||
report = _drained_report_with_live_sessions(2)
|
||||
state = self._state(
|
||||
acks={"prgs-author-0": "ack", "prgs-author-1": "acknowledged"}
|
||||
)
|
||||
proof = self._build(report, state)
|
||||
self.assertTrue(self._acks_check(proof).passed)
|
||||
self.assertTrue(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [])
|
||||
|
||||
def test_no_other_live_sessions_still_passes(self):
|
||||
"""Intended behavior retained: zero live sessions needs no acks."""
|
||||
report = _drained_report_with_live_sessions(0)
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 0)
|
||||
proof = self._build(report, self._state())
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("sessions_live_other=0", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
# --- timeout policy cannot become a second fail-open ------------------
|
||||
|
||||
def test_unproven_timeout_policy_cannot_open_the_gate(self):
|
||||
for value in (None, "true", "yes", 1, "True", [], {}):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(2),
|
||||
self._state(ack_timeout_policy_applied=value),
|
||||
)
|
||||
|
||||
def test_explicit_timeout_policy_permits(self):
|
||||
proof = self._build(
|
||||
_drained_report_with_live_sessions(2),
|
||||
self._state(ack_timeout_policy_applied=True),
|
||||
)
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("timeout policy", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
# --- the gate itself must deny ---------------------------------------
|
||||
|
||||
def test_failed_ack_check_denies_the_restart_gate(self):
|
||||
report = _drained_report_with_live_sessions(3)
|
||||
proof = self._build(report, self._state())
|
||||
self.assertFalse(proof.clean)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_DENY)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
|
||||
def test_unclean_ack_proof_fails_verification(self):
|
||||
report = _drained_report_with_live_sessions(3)
|
||||
proof = self._build(report, self._state())
|
||||
result = dp.verify_drain_proof(
|
||||
proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertFalse(result.clean)
|
||||
|
||||
|
||||
def _identity_report(*, requester: str, others: tuple[str, ...]) -> dict:
|
||||
"""Report with explicitly named requester and other live sessions.
|
||||
|
||||
Unlike :func:`_drained_report_with_live_sessions`, the session ids are
|
||||
chosen by the caller so a test can supply acknowledgements for the *wrong*
|
||||
identities while keeping the count correct.
|
||||
"""
|
||||
|
||||
sessions = [
|
||||
{
|
||||
"session_id": requester,
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
]
|
||||
for session_id in others:
|
||||
sessions.append(
|
||||
{
|
||||
"session_id": session_id,
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
)
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id=requester,
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
class AcknowledgementIdentityBindingTests(unittest.TestCase):
|
||||
"""Acknowledgement coverage must be bound to session identity, not counted.
|
||||
|
||||
Regression cover for the second reviewed fail-open on PR #882 (review 582,
|
||||
blocker B1): coverage compared ``acked_count`` against
|
||||
``counts.sessions_live_other``, so acknowledgements supplied for the
|
||||
requesting session and for ids that do not exist satisfied the obligations
|
||||
of the live sessions that never answered. The required identities are
|
||||
carried by the report itself — ``ack_state`` keys and ``affected_sessions``
|
||||
filtered on ``live and not is_requester`` — and only an acknowledgement
|
||||
keyed by one of those ids may count for it.
|
||||
"""
|
||||
|
||||
def _state(self, **overrides) -> dict:
|
||||
state = _clean_drain_state()
|
||||
state.pop("acks", None)
|
||||
state["ack_timeout_policy_applied"] = False
|
||||
state.update(overrides)
|
||||
return state
|
||||
|
||||
def _acks_check(self, proof) -> dp.DrainCheck:
|
||||
return next(c for c in proof.checks if c.name == dp.CHECK_ACKS_OR_TIMEOUT)
|
||||
|
||||
def _build(self, report: dict, state: dict):
|
||||
return dp.build_drain_proof(
|
||||
impact_report=report, drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
|
||||
def assertAcksFailClosed(self, report: dict, state: dict) -> dp.DrainCheck:
|
||||
"""Failure must propagate through the check, the proof, and the gate."""
|
||||
|
||||
proof = self._build(report, state)
|
||||
check = self._acks_check(proof)
|
||||
self.assertFalse(check.passed)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_ACKS_OR_TIMEOUT, proof.failed_checks)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertEqual(decision.verdict, dp.GATE_DENY)
|
||||
self.assertFalse(decision.allow)
|
||||
return check
|
||||
|
||||
# --- the reviewer's exact reproduction --------------------------------
|
||||
|
||||
def test_requester_plus_unknown_id_cannot_satisfy_two_live_sessions(self):
|
||||
"""Review 582 B1 verbatim: requester + a nonexistent session.
|
||||
|
||||
``sessions_live_other=2`` with ``ack_state`` naming ``other-0`` and
|
||||
``other-1``; the drain state supplies an acknowledgement from the
|
||||
requesting session itself and from a session that does not exist. The
|
||||
count matches, the identities do not.
|
||||
"""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 2)
|
||||
self.assertEqual(
|
||||
report["ack_state"], {"other-0": "pending", "other-1": "pending"}
|
||||
)
|
||||
state = self._state(acks={"req": "ack", "totally-bogus-session": "ack"})
|
||||
check = self.assertAcksFailClosed(report, state)
|
||||
self.assertIn("other-0", check.detail)
|
||||
self.assertIn("other-1", check.detail)
|
||||
self.assertIn("fail closed", check.detail)
|
||||
|
||||
# --- wrong / unknown / requester identities ---------------------------
|
||||
|
||||
def test_sufficient_count_of_wrong_ids_fails_closed(self):
|
||||
"""Right cardinality, wrong identities: two acks, neither required."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
state = self._state(acks={"ghost-a": "ack", "ghost-b": "ack"})
|
||||
check = self.assertAcksFailClosed(report, state)
|
||||
self.assertIn("do not count", check.detail)
|
||||
|
||||
def test_more_acks_than_required_still_fails_on_wrong_ids(self):
|
||||
"""Coverage cannot be bought with volume: five acks, none required."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
state = self._state(acks={f"ghost-{i}": "acknowledged" for i in range(5)})
|
||||
self.assertAcksFailClosed(report, state)
|
||||
|
||||
def test_partial_identity_match_fails_closed(self):
|
||||
"""One required id acknowledged, the rest padded with unknown ids."""
|
||||
|
||||
report = _identity_report(
|
||||
requester="req", others=("other-0", "other-1", "other-2")
|
||||
)
|
||||
state = self._state(
|
||||
acks={"other-0": "ack", "ghost-1": "ack", "ghost-2": "ack"}
|
||||
)
|
||||
check = self.assertAcksFailClosed(report, state)
|
||||
self.assertIn("other-1", check.detail)
|
||||
self.assertIn("other-2", check.detail)
|
||||
|
||||
def test_requester_ack_never_satisfies_another_sessions_obligation(self):
|
||||
"""The requester is excluded from the required set and stays excluded."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
requester_rows = [s for s in report["affected_sessions"] if s["is_requester"]]
|
||||
self.assertEqual([s["session_id"] for s in requester_rows], ["req"])
|
||||
self.assertNotIn("req", report["ack_state"])
|
||||
check = self.assertAcksFailClosed(report, self._state(acks={"req": "ack"}))
|
||||
self.assertIn("other-0", check.detail)
|
||||
|
||||
def test_fabricated_ids_do_not_count_toward_coverage(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
for bogus in ("", " ", "other-0 extra", "OTHER-0", "other-01", "0"):
|
||||
with self.subTest(bogus=bogus):
|
||||
self.assertAcksFailClosed(report, self._state(acks={bogus: "ack"}))
|
||||
|
||||
# --- per-session state must be explicitly valid ------------------------
|
||||
|
||||
def test_unproven_per_session_states_fail_closed(self):
|
||||
"""A required id present but not explicitly acknowledged fails closed."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
for value in ("pending", "stale", "unknown", "", None, True, 1, NOW):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
report,
|
||||
self._state(acks={"other-0": "ack", "other-1": value}),
|
||||
)
|
||||
|
||||
def test_report_ack_state_placeholder_is_never_read_as_an_ack(self):
|
||||
"""``ack_state`` values are the report's own placeholders, not evidence."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = {"other-0": "ack"}
|
||||
self.assertAcksFailClosed(report, self._state())
|
||||
|
||||
# --- missing / malformed / contradictory identity evidence -------------
|
||||
|
||||
def test_missing_identity_evidence_fails_closed(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
report.pop("affected_sessions", None)
|
||||
check = self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
self.assertIn("no session-identity evidence", check.detail)
|
||||
|
||||
def test_malformed_ack_state_fails_closed(self):
|
||||
for malformed in ([], "other-0", 7, None, ("other-0",)):
|
||||
with self.subTest(malformed=malformed):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = malformed
|
||||
self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack"})
|
||||
)
|
||||
|
||||
def test_non_string_ack_state_key_fails_closed(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = {7: "pending"}
|
||||
self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
|
||||
def test_malformed_affected_sessions_fails_closed(self):
|
||||
for malformed in ("sessions", 7, {"session_id": "other-0"}, [None], [7]):
|
||||
with self.subTest(malformed=malformed):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
report["affected_sessions"] = malformed
|
||||
self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack"})
|
||||
)
|
||||
|
||||
def test_affected_sessions_without_explicit_booleans_fails_closed(self):
|
||||
"""``live``/``is_requester`` must be real booleans, never inferred."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
for row in report["affected_sessions"]:
|
||||
if row["session_id"] == "other-0":
|
||||
row["is_requester"] = "false"
|
||||
self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
|
||||
def test_affected_sessions_missing_live_flag_fails_closed(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
for row in report["affected_sessions"]:
|
||||
row.pop("live", None)
|
||||
self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
|
||||
def test_contradictory_ack_state_and_affected_sessions_fails_closed(self):
|
||||
"""Both views present and disagreeing is unresolvable, not a tie-break."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
report["ack_state"] = {"other-0": "pending", "other-9": "pending"}
|
||||
check = self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack", "other-9": "ack"})
|
||||
)
|
||||
self.assertIn("contradicts itself", check.detail)
|
||||
|
||||
def test_identity_count_mismatch_fails_closed(self):
|
||||
"""Identity evidence that cannot be reconciled with the count denies."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
report["counts"] = dict(report["counts"], sessions_live_other=1)
|
||||
check = self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack", "other-1": "ack"})
|
||||
)
|
||||
self.assertIn("cannot be reconciled", check.detail)
|
||||
|
||||
def test_broken_identity_evidence_outranks_timeout_policy(self):
|
||||
"""The sanctioned timeout path cannot paper over an unreadable report."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = "not-a-mapping"
|
||||
self.assertAcksFailClosed(report, self._state(ack_timeout_policy_applied=True))
|
||||
|
||||
# --- legitimate success is preserved -----------------------------------
|
||||
|
||||
def test_every_required_session_acknowledged_passes(self):
|
||||
report = _identity_report(
|
||||
requester="req", others=("other-0", "other-1", "other-2")
|
||||
)
|
||||
state = self._state(
|
||||
acks={
|
||||
"other-0": "ack",
|
||||
"other-1": "acked",
|
||||
"other-2": "acknowledged",
|
||||
}
|
||||
)
|
||||
proof = self._build(report, state)
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertTrue(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [])
|
||||
self.assertIn("acknowledged by identity", check.detail)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertEqual(decision.verdict, dp.GATE_ALLOW)
|
||||
self.assertTrue(decision.allow)
|
||||
|
||||
def test_required_session_ack_tolerates_surrounding_whitespace(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
proof = self._build(report, self._state(acks={" other-0 ": " ACK "}))
|
||||
self.assertTrue(self._acks_check(proof).passed)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
def test_no_other_live_sessions_still_passes_with_identity_evidence(self):
|
||||
report = _identity_report(requester="req", others=())
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 0)
|
||||
self.assertEqual(report["ack_state"], {})
|
||||
proof = self._build(report, self._state())
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("sessions_live_other=0", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
def test_explicit_timeout_policy_retains_intended_behavior(self):
|
||||
"""Valid, correctly typed timeout evidence still permits the check."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
proof = self._build(report, self._state(ack_timeout_policy_applied=True))
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("timeout policy", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
def test_timeout_policy_still_strictly_typed_under_identity_binding(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
for value in (None, "true", "True", 1, [], {}):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
report, self._state(ack_timeout_policy_applied=value)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,378 @@
|
||||
"""Allocator semantic container exclusion for vision/roadmap/umbrella (#854).
|
||||
|
||||
#844 excluded epic / child-only containers (live #631) but product-vision
|
||||
(#652), phased-roadmap (#653), and umbrella (#655) coordination records still
|
||||
ranked as implementable work. This module is the live-equivalent canary:
|
||||
|
||||
* #631 / #652 / #653 / #655-shaped records are all excluded in one inventory.
|
||||
* Independently executable children remain eligible and can be selected.
|
||||
* Ordinary issues that merely mention vision / roadmap / umbrella stay eligible.
|
||||
* Excluded containers never receive assignments or workflow leases.
|
||||
* Structured skip reason ``epic_or_child_only_container`` is reported.
|
||||
* Candidate-set fingerprint remains stable after exclusions.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
|
||||
from allocator_service import (
|
||||
OUTCOME_ASSIGNED,
|
||||
OUTCOME_PREVIEW,
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
WorkCandidate,
|
||||
allocate_next_work,
|
||||
candidate_set_fingerprint,
|
||||
classify_epic_or_child_only_container,
|
||||
)
|
||||
from control_plane_db import ControlPlaneDB
|
||||
|
||||
REMOTE = "prgs"
|
||||
ORG = "Scaled-Tech-Consulting"
|
||||
REPO = "Gitea-Tools"
|
||||
|
||||
# Minimal bodies mirroring live coordination records (not full issue text).
|
||||
_EPIC_631_BODY = """
|
||||
## Scope (umbrella)
|
||||
|
||||
This epic owns the **product roadmap and linkage** for the Web Console.
|
||||
Implementation is delivered via child issues only.
|
||||
|
||||
## Explicit non-goals
|
||||
|
||||
* Do not implement product features in this epic issue itself.
|
||||
* No product feature implementation is claimed complete solely on this epic.
|
||||
"""
|
||||
|
||||
_VISION_652_BODY = """
|
||||
## Canonical product vision — enduring source of truth
|
||||
|
||||
**This issue is the enduring source of truth for the MCP Control Plane Web Console product vision.**
|
||||
|
||||
## Implementation linkage
|
||||
|
||||
* **Do not implement features on this issue.**
|
||||
* Sequencing: roadmap issue + #631 children.
|
||||
|
||||
## Canonical issue state
|
||||
|
||||
```text
|
||||
STATE: vision-active
|
||||
WHO_IS_NEXT: controller (triage/ordering) / author (implementation of linked children only)
|
||||
```
|
||||
"""
|
||||
|
||||
_ROADMAP_653_BODY = """
|
||||
## Purpose
|
||||
|
||||
This issue is the **phased delivery roadmap and epic sequencing** for the MCP Control Plane Web Console.
|
||||
|
||||
## Non-goals
|
||||
|
||||
* Implementing features on this roadmap issue.
|
||||
* Deleting vision items by omitting them from phases without #652 change log.
|
||||
|
||||
## Canonical issue state
|
||||
|
||||
```text
|
||||
STATE: roadmap-active
|
||||
WHO_IS_NEXT: author
|
||||
```
|
||||
"""
|
||||
|
||||
_UMBRELLA_655_BODY = """
|
||||
## Scope (umbrella)
|
||||
|
||||
This issue owns the **canonical restart-governance program**. Implementation is via linked children only.
|
||||
|
||||
## Acceptance criteria (umbrella)
|
||||
|
||||
6. No product feature claimed complete on this issue alone.
|
||||
"""
|
||||
|
||||
_CHILD_BODY = """
|
||||
## Problem
|
||||
|
||||
Operators need a workflow-event timeline model for Phase 1.
|
||||
|
||||
## Acceptance criteria
|
||||
|
||||
- [ ] Timeline model API exists
|
||||
"""
|
||||
|
||||
|
||||
def _issue(
|
||||
number: int,
|
||||
*,
|
||||
title: str = "",
|
||||
body: str = "",
|
||||
labels: tuple[str, ...] = ("status:ready", "type:feature"),
|
||||
priority: int = 20,
|
||||
) -> WorkCandidate:
|
||||
return WorkCandidate(
|
||||
kind="issue",
|
||||
number=number,
|
||||
state="open",
|
||||
labels=labels,
|
||||
title=title or f"issue {number}",
|
||||
body=body,
|
||||
priority=priority,
|
||||
)
|
||||
|
||||
|
||||
def _live_shaped_containers() -> list[WorkCandidate]:
|
||||
return [
|
||||
_issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
),
|
||||
_issue(
|
||||
652,
|
||||
title="Product vision: MCP Control Plane Web Console (canonical)",
|
||||
body=_VISION_652_BODY,
|
||||
),
|
||||
_issue(
|
||||
653,
|
||||
title="Roadmap: MCP Control Plane Web Console (phased delivery)",
|
||||
body=_ROADMAP_653_BODY,
|
||||
),
|
||||
_issue(
|
||||
655,
|
||||
title="Umbrella: Governed MCP restart coordination and zero-disruption recovery",
|
||||
body=_UMBRELLA_655_BODY,
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
class ClassifySemanticContainersTest(unittest.TestCase):
|
||||
def test_652_vision_is_container(self) -> None:
|
||||
c = _issue(
|
||||
652,
|
||||
title="Product vision: MCP Control Plane Web Console (canonical)",
|
||||
body=_VISION_652_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIsNotNone(detail)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_653_roadmap_is_container(self) -> None:
|
||||
c = _issue(
|
||||
653,
|
||||
title="Roadmap: MCP Control Plane Web Console (phased delivery)",
|
||||
body=_ROADMAP_653_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_655_umbrella_is_container(self) -> None:
|
||||
c = _issue(
|
||||
655,
|
||||
title="Umbrella: Governed MCP restart coordination and zero-disruption recovery",
|
||||
body=_UMBRELLA_655_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_631_still_container_after_854(self) -> None:
|
||||
c = _issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_incidental_vision_roadmap_umbrella_words_not_container(self) -> None:
|
||||
cases = (
|
||||
(
|
||||
"Document vision handoff conventions",
|
||||
"Update docs so implementable issues that mention a vision "
|
||||
"remain independently executable.",
|
||||
),
|
||||
(
|
||||
"Clarify roadmap sequencing notes",
|
||||
"Write a short note about how the roadmap issue relates to children.",
|
||||
),
|
||||
(
|
||||
"Umbrella recovery checklist for authors",
|
||||
"Authors should still implement the concrete recovery fix here.",
|
||||
),
|
||||
(
|
||||
"Product vision wording in the help text",
|
||||
"Fix a typo in the operator-facing help string that says product vision.",
|
||||
),
|
||||
)
|
||||
for title, body in cases:
|
||||
with self.subTest(title=title):
|
||||
c = _issue(900, title=title, body=body)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
self.assertIsNone(detail)
|
||||
|
||||
def test_title_prefix_alone_not_container(self) -> None:
|
||||
for title in (
|
||||
"Epic: something mentioned only in title",
|
||||
"Roadmap: title only without body scope",
|
||||
"Product vision: title only without body scope",
|
||||
"Umbrella: title only without body scope",
|
||||
):
|
||||
with self.subTest(title=title):
|
||||
c = _issue(
|
||||
901,
|
||||
title=title,
|
||||
body="Implement a concrete fix for the allocator skip list.",
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
self.assertIsNone(detail)
|
||||
|
||||
def test_roadmap_label_alone_is_container(self) -> None:
|
||||
c = _issue(
|
||||
902,
|
||||
title="Console delivery sequencing",
|
||||
body="Track phased delivery only.",
|
||||
labels=("status:ready", "type:roadmap"),
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("type:roadmap", detail or "")
|
||||
|
||||
def test_child_referencing_parent_policy_stays_eligible(self) -> None:
|
||||
"""Children may quote parent policy without becoming containers."""
|
||||
c = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=(
|
||||
_CHILD_BODY
|
||||
+ "\n\nParent #652 says do not implement on the vision issue; "
|
||||
"this child is the implementable unit."
|
||||
),
|
||||
)
|
||||
is_c, _ = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
|
||||
|
||||
class AllocateSemanticContainerExclusionTest(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(os.path.join(self._tmp.name, "cp.sqlite3"))
|
||||
|
||||
def _alloc(self, candidates, **kwargs):
|
||||
defaults = dict(
|
||||
session_id="sess-854",
|
||||
role="author",
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
profile_name="prgs-author",
|
||||
username="jcwalker3",
|
||||
claims={},
|
||||
apply=False,
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
return allocate_next_work(self.db, candidates=candidates, **defaults)
|
||||
|
||||
def test_live_equivalent_canary_excludes_all_containers_selects_child(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
inventory = containers + [child]
|
||||
res = self._alloc(inventory, apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
|
||||
skipped = {s["number"]: s for s in res["skipped"]}
|
||||
for number in (631, 652, 653, 655):
|
||||
self.assertIn(number, skipped, res["skipped"])
|
||||
self.assertEqual(
|
||||
skipped[number]["reason_code"],
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
)
|
||||
self.assertIn(
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER, skipped[number]["reason"]
|
||||
)
|
||||
|
||||
def test_containers_cannot_receive_assignment_or_lease(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
res = self._alloc(containers, apply=True)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertNotEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertIsNone(res.get("assignment"))
|
||||
self.assertIsNone(res.get("selected"))
|
||||
skipped = {s["number"]: s for s in res["skipped"]}
|
||||
for number in (631, 652, 653, 655):
|
||||
self.assertEqual(
|
||||
skipped[number]["reason_code"],
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
)
|
||||
|
||||
leases = []
|
||||
if hasattr(self.db, "list_active_leases"):
|
||||
leases = self.db.list_active_leases(
|
||||
remote=REMOTE, org=ORG, repo=REPO
|
||||
)
|
||||
if not leases and hasattr(self.db, "list_leases"):
|
||||
leases = self.db.list_leases(remote=REMOTE, org=ORG, repo=REPO)
|
||||
for lease in leases or []:
|
||||
work_number = (
|
||||
lease.get("work_number") if isinstance(lease, dict) else None
|
||||
)
|
||||
self.assertNotIn(work_number, {631, 652, 653, 655})
|
||||
|
||||
def test_apply_selects_child_not_container(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
res = self._alloc(containers + [child], apply=True)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
self.assertEqual(res["assignment"]["work_number"], 637)
|
||||
|
||||
def test_fingerprint_stable_with_containers_present(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
inventory = containers + [child]
|
||||
fp_before = candidate_set_fingerprint(inventory)
|
||||
res = self._alloc(inventory, apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
# Allocator reports the same CAS fingerprint for the full candidate set.
|
||||
reported = res.get("candidate_set_fingerprint")
|
||||
self.assertEqual(reported, fp_before)
|
||||
# Re-fingerprint of the same inventory is byte-stable.
|
||||
self.assertEqual(candidate_set_fingerprint(inventory), fp_before)
|
||||
|
||||
def test_incidental_mentions_remain_eligible(self) -> None:
|
||||
ordinary = _issue(
|
||||
700,
|
||||
title="Document roadmap handoff conventions",
|
||||
body="Write runbook text about vision vs roadmap vs child issues.",
|
||||
)
|
||||
res = self._alloc([ordinary], apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["selected"]["number"], 700)
|
||||
self.assertEqual(res["skipped"], [])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,739 @@
|
||||
"""Tests for the Runtime and session view (Phase 1, #641).
|
||||
|
||||
Covers clean and stale session rendering, contamination marker surfacing,
|
||||
worktree binding display, sanctioned recovery links (no pkill), nav/live
|
||||
status, and the JSON API export.
|
||||
|
||||
Also pins the two invariants a reviewer found violated at head a81db754:
|
||||
degraded ownership sections must render as *unknown* rather than as an
|
||||
affirmative "none"/"unbound", and contamination payload text must be redacted
|
||||
at the display boundary rather than trusted from the write-time denylist.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from tests.webui_testclient import TestClient
|
||||
|
||||
from webui.app import create_app
|
||||
from webui.inventory import (
|
||||
AUTHORITY_CONTROL_PLANE_DB,
|
||||
AUTHORITY_FILESYSTEM,
|
||||
InventorySection,
|
||||
InventorySnapshot,
|
||||
STATUS_DEGRADED,
|
||||
STATUS_OK,
|
||||
STATUS_UNAVAILABLE,
|
||||
)
|
||||
from webui.nav import NAV_GROUPS, STUB_PAGES, iter_nav_items
|
||||
from webui.runtime_health import FileHash, RuntimeSnapshot
|
||||
from webui.session_loader import (
|
||||
ContaminationMarker,
|
||||
SessionRow,
|
||||
SessionViewSnapshot,
|
||||
_build_session_rows,
|
||||
_inspect_contamination,
|
||||
load_session_view_snapshot,
|
||||
snapshot_to_dict,
|
||||
)
|
||||
from webui.session_views import render_sessions_page
|
||||
|
||||
|
||||
def _runtime(
|
||||
*,
|
||||
stale: str | None = None,
|
||||
profile: str = "prgs-author",
|
||||
role: str = "author",
|
||||
) -> RuntimeSnapshot:
|
||||
return RuntimeSnapshot(
|
||||
project_id="gitea-tools",
|
||||
repo_root="/tmp/repo",
|
||||
remote="prgs",
|
||||
host="gitea.prgs.cc",
|
||||
profile_name=profile,
|
||||
role_kind=role,
|
||||
config_model="v2-contexts",
|
||||
profile_mode="dynamic-profile",
|
||||
profile_source="config file profile",
|
||||
authenticated_username="jcwalker3",
|
||||
identity_error=None,
|
||||
repo_sha="a" * 40,
|
||||
remote_master_sha="a" * 40,
|
||||
commits_behind_master=0,
|
||||
stale_runtime_warning=stale,
|
||||
shell_health={"shell_use_allowed": True, "consecutive_spawn_failures": 0},
|
||||
workflow_hashes=(
|
||||
FileHash(label="SKILL.md", path="skills/llm-project-workflow/SKILL.md", sha256="abc"),
|
||||
),
|
||||
schema_hashes=(),
|
||||
restart_guidance="docs/mcp-namespace-eof-recovery.md",
|
||||
fetch_error=None,
|
||||
)
|
||||
|
||||
|
||||
def _inventory(
|
||||
*,
|
||||
sessions: tuple[dict, ...] = (),
|
||||
leases: tuple[dict, ...] = (),
|
||||
locks: tuple[dict, ...] = (),
|
||||
worktrees: tuple[dict, ...] = (),
|
||||
namespaces: tuple[dict, ...] = (),
|
||||
statuses: dict[str, str] | None = None,
|
||||
) -> InventorySnapshot:
|
||||
"""Build a snapshot; ``statuses`` degrades named sections (default all ok)."""
|
||||
status_of = statuses or {}
|
||||
|
||||
def _status(name: str) -> str:
|
||||
return status_of.get(name, STATUS_OK)
|
||||
|
||||
sections = (
|
||||
InventorySection(
|
||||
name="sessions",
|
||||
authority=AUTHORITY_CONTROL_PLANE_DB,
|
||||
status=_status("sessions"),
|
||||
items=sessions,
|
||||
),
|
||||
InventorySection(
|
||||
name="leases",
|
||||
authority=AUTHORITY_CONTROL_PLANE_DB,
|
||||
status=_status("leases"),
|
||||
items=leases,
|
||||
),
|
||||
InventorySection(
|
||||
name="locks",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=_status("locks"),
|
||||
items=locks,
|
||||
),
|
||||
InventorySection(
|
||||
name="worktrees",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=_status("worktrees"),
|
||||
items=worktrees,
|
||||
),
|
||||
InventorySection(
|
||||
name="namespaces",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=_status("namespaces"),
|
||||
items=namespaces
|
||||
or (
|
||||
{
|
||||
"profile_name": "prgs-author",
|
||||
"role": "author",
|
||||
"mcp_namespace": "gitea-author",
|
||||
"capability_summary": {
|
||||
"can_author": True,
|
||||
"can_review": False,
|
||||
"can_merge": False,
|
||||
},
|
||||
"active": True,
|
||||
},
|
||||
),
|
||||
reason="only the profile serving this web process is observable",
|
||||
),
|
||||
)
|
||||
index = {section.name: section for section in sections}
|
||||
return InventorySnapshot(
|
||||
generated_at="2026-07-25T00:00:00+00:00",
|
||||
sections=sections,
|
||||
collisions=(),
|
||||
correlations=(),
|
||||
scan_ms=1.0,
|
||||
_section_index=index,
|
||||
)
|
||||
|
||||
|
||||
def _clean_session() -> dict:
|
||||
return {
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"namespace": "gitea-author",
|
||||
"pid": 1111,
|
||||
"pid_alive": True,
|
||||
"status": "active",
|
||||
"started_at": "2026-07-25T00:00:00Z",
|
||||
"last_heartbeat_at": "2026-07-25T01:00:00Z",
|
||||
}
|
||||
|
||||
|
||||
def _stale_session() -> dict:
|
||||
return {
|
||||
"session_id": "prgs-author-222-stale",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"namespace": "gitea-author",
|
||||
"pid": 2222,
|
||||
"pid_alive": False,
|
||||
"status": "active",
|
||||
"started_at": "2026-07-24T00:00:00Z",
|
||||
"last_heartbeat_at": "2026-07-24T01:00:00Z",
|
||||
}
|
||||
|
||||
|
||||
class TestBuildSessionRows(unittest.TestCase):
|
||||
def test_clean_session_has_no_stale_or_contamination_flags(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
leases=(
|
||||
{
|
||||
"lease_id": "lease-clean",
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"status": "active",
|
||||
"expired": False,
|
||||
"work_kind": "issue",
|
||||
"work_number": 641,
|
||||
},
|
||||
),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"branch_name": "feat/issue-641-runtime-session-view",
|
||||
"worktree_path": "~/Development/Gitea-Tools/branches/feat-issue-641",
|
||||
"live": True,
|
||||
},
|
||||
),
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination=())
|
||||
self.assertEqual(len(rows), 1)
|
||||
row = rows[0]
|
||||
self.assertEqual(row.session_id, "prgs-author-111-clean")
|
||||
self.assertEqual(row.role, "author")
|
||||
self.assertEqual(row.namespace, "gitea-author")
|
||||
self.assertEqual(row.pid_alive, True)
|
||||
self.assertEqual(row.lease_ids, ("lease-clean",))
|
||||
self.assertEqual(row.work_refs, ("issue#641",))
|
||||
self.assertTrue(row.worktree_paths)
|
||||
self.assertEqual(row.stale_flags, ())
|
||||
self.assertEqual(row.contamination_flags, ())
|
||||
|
||||
def test_stale_session_flags_dead_pid(self):
|
||||
inventory = _inventory(sessions=(_stale_session(),))
|
||||
rows = _build_session_rows(inventory, contamination=())
|
||||
self.assertEqual(rows[0].stale_flags, ("pid-dead",))
|
||||
|
||||
def test_contamination_marker_binds_to_session(self):
|
||||
inventory = _inventory(sessions=(_clean_session(),))
|
||||
marker = ContaminationMarker(
|
||||
kind="runtime_recovery_contamination",
|
||||
on_disk=True,
|
||||
has_payload=True,
|
||||
summary="manual daemon kill",
|
||||
reason_class="manual_daemon_kill",
|
||||
session_id="prgs-author-111-clean",
|
||||
role="author",
|
||||
command_summary="pkill -f mcp_server.py",
|
||||
cleared=False,
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination=(marker,))
|
||||
self.assertIn("runtime_recovery_contamination", rows[0].contamination_flags)
|
||||
|
||||
def test_process_wide_contamination_surfaces_on_all_sessions(self):
|
||||
inventory = _inventory(sessions=(_clean_session(), _stale_session()))
|
||||
marker = ContaminationMarker(
|
||||
kind="stable_branch_contamination",
|
||||
on_disk=True,
|
||||
has_payload=True,
|
||||
summary="direct master push attempt",
|
||||
reason_class="stable_branch_push",
|
||||
session_id=None,
|
||||
cleared=False,
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination=(marker,))
|
||||
self.assertEqual(len(rows), 2)
|
||||
for row in rows:
|
||||
self.assertTrue(
|
||||
any("stable_branch_contamination" in f for f in row.contamination_flags)
|
||||
)
|
||||
|
||||
|
||||
class TestRenderSessionsPage(unittest.TestCase):
|
||||
def _snapshot(
|
||||
self,
|
||||
*,
|
||||
sessions: tuple[dict, ...],
|
||||
contamination: tuple[ContaminationMarker, ...] = (),
|
||||
stale_runtime: str | None = None,
|
||||
) -> SessionViewSnapshot:
|
||||
inventory = _inventory(
|
||||
sessions=sessions,
|
||||
leases=(
|
||||
{
|
||||
"lease_id": "lease-1",
|
||||
"session_id": sessions[0]["session_id"] if sessions else "",
|
||||
"status": "active",
|
||||
"expired": False,
|
||||
"work_kind": "issue",
|
||||
"work_number": 641,
|
||||
},
|
||||
)
|
||||
if sessions
|
||||
else (),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"worktree_path": "branches/feat-issue-641",
|
||||
},
|
||||
)
|
||||
if sessions
|
||||
else (),
|
||||
worktrees=(
|
||||
{
|
||||
"rel_path": "branches/feat-issue-641",
|
||||
"branch": "feat/issue-641-runtime-session-view",
|
||||
"classification": "active_issue_work",
|
||||
"registered_worktree": True,
|
||||
"dirty": False,
|
||||
},
|
||||
),
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination)
|
||||
return SessionViewSnapshot(
|
||||
runtime=_runtime(stale=stale_runtime),
|
||||
inventory=inventory,
|
||||
sessions=rows,
|
||||
contamination_markers=contamination,
|
||||
)
|
||||
|
||||
def test_clean_session_render(self):
|
||||
html = render_sessions_page(self._snapshot(sessions=(_clean_session(),)))
|
||||
self.assertIn("Runtime and sessions", html)
|
||||
self.assertIn("prgs-author-111-clean", html)
|
||||
self.assertIn("gitea-author", html)
|
||||
self.assertIn("branches/feat-issue-641", html)
|
||||
self.assertIn("Sanctioned recovery", html)
|
||||
self.assertIn("docs/mcp-namespace-eof-recovery.md", html)
|
||||
# Recovery section must name reconnect and forbid manual kill.
|
||||
recovery_idx = html.lower().find("sanctioned recovery")
|
||||
self.assertGreaterEqual(recovery_idx, 0)
|
||||
recovery = html[recovery_idx:].lower()
|
||||
self.assertIn("reconnect", recovery)
|
||||
self.assertIn("contamination", recovery)
|
||||
self.assertIn("not recovery", recovery)
|
||||
self.assertNotIn("run pkill", recovery)
|
||||
self.assertNotIn("killall", recovery)
|
||||
|
||||
def test_stale_session_render(self):
|
||||
html = render_sessions_page(self._snapshot(sessions=(_stale_session(),)))
|
||||
self.assertIn("prgs-author-222-stale", html)
|
||||
self.assertIn("pid-dead", html)
|
||||
self.assertIn("badge-stale", html)
|
||||
|
||||
def test_contamination_render_is_not_silent(self):
|
||||
marker = ContaminationMarker(
|
||||
kind="runtime_recovery_contamination",
|
||||
on_disk=True,
|
||||
has_payload=True,
|
||||
summary="manual kill",
|
||||
reason_class="manual_daemon_kill",
|
||||
session_id="prgs-author-111-clean",
|
||||
command_summary="pkill -f mcp_server.py",
|
||||
cleared=False,
|
||||
)
|
||||
html = render_sessions_page(
|
||||
self._snapshot(sessions=(_clean_session(),), contamination=(marker,))
|
||||
)
|
||||
self.assertIn("Contamination markers", html)
|
||||
self.assertIn("runtime_recovery_contamination", html)
|
||||
self.assertIn("ACTIVE", html)
|
||||
self.assertIn("badge-blocked", html)
|
||||
|
||||
def test_stale_runtime_banner(self):
|
||||
html = render_sessions_page(
|
||||
self._snapshot(
|
||||
sessions=(_clean_session(),),
|
||||
stale_runtime="server behind master by 3 commits",
|
||||
)
|
||||
)
|
||||
self.assertIn("Stale runtime", html)
|
||||
self.assertIn("server behind master", html)
|
||||
|
||||
|
||||
class TestSessionLoaderComposition(unittest.TestCase):
|
||||
def test_load_with_injected_sources(self):
|
||||
inventory = _inventory(sessions=(_clean_session(), _stale_session()))
|
||||
snap = load_session_view_snapshot(
|
||||
load_runtime=lambda: _runtime(),
|
||||
load_inventory=lambda: inventory,
|
||||
inspect_contamination=lambda **_k: {
|
||||
"on_disk": False,
|
||||
"has_payload": False,
|
||||
"summary": "absent",
|
||||
},
|
||||
load_contamination_payload=lambda **_k: None,
|
||||
)
|
||||
self.assertEqual(len(snap.sessions), 2)
|
||||
self.assertEqual(snap.stale_session_count, 1)
|
||||
self.assertEqual(snap.contaminated_session_count, 0)
|
||||
data = snapshot_to_dict(snap)
|
||||
self.assertEqual(data["view"], "runtime-sessions")
|
||||
self.assertEqual(data["issue"], 641)
|
||||
self.assertTrue(data["read_only"])
|
||||
self.assertEqual(data["session_counts"]["total"], 2)
|
||||
self.assertEqual(data["session_counts"]["stale"], 1)
|
||||
self.assertIn("recovery_docs", data)
|
||||
|
||||
|
||||
class TestSessionsRoutes(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.client = TestClient(create_app())
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(), _stale_session()),
|
||||
leases=(
|
||||
{
|
||||
"lease_id": "lease-x",
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"status": "active",
|
||||
"expired": False,
|
||||
"work_kind": "issue",
|
||||
"work_number": 641,
|
||||
},
|
||||
),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"worktree_path": "branches/feat-issue-641",
|
||||
},
|
||||
),
|
||||
worktrees=(
|
||||
{
|
||||
"rel_path": "branches/feat-issue-641",
|
||||
"branch": "feat/issue-641-runtime-session-view",
|
||||
"classification": "active_issue_work",
|
||||
"registered_worktree": True,
|
||||
"dirty": False,
|
||||
},
|
||||
),
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination=())
|
||||
self.snapshot = SessionViewSnapshot(
|
||||
runtime=_runtime(stale="stale for test"),
|
||||
inventory=inventory,
|
||||
sessions=rows,
|
||||
contamination_markers=(),
|
||||
)
|
||||
self._patch = mock.patch(
|
||||
"webui.app.load_session_view_snapshot",
|
||||
return_value=self.snapshot,
|
||||
)
|
||||
self._patch.start()
|
||||
|
||||
def tearDown(self):
|
||||
self._patch.stop()
|
||||
|
||||
def test_sessions_page_live(self):
|
||||
response = self.client.get("/sessions")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertIn("Runtime and sessions", response.text)
|
||||
self.assertIn("prgs-author-111-clean", response.text)
|
||||
self.assertIn("prgs-author-222-stale", response.text)
|
||||
self.assertIn("pid-dead", response.text)
|
||||
self.assertIn("Sanctioned recovery", response.text)
|
||||
self.assertNotIn("Phase 1 shell placeholder", response.text)
|
||||
self.assertNotIn("child issue of #425", response.text.lower())
|
||||
|
||||
def test_api_sessions_json(self):
|
||||
for path in ("/api/sessions", "/api/v1/sessions"):
|
||||
response = self.client.get(path)
|
||||
self.assertEqual(response.status_code, 200, path)
|
||||
data = response.json()
|
||||
self.assertEqual(data["view"], "runtime-sessions")
|
||||
self.assertEqual(data["session_counts"]["total"], 2)
|
||||
self.assertEqual(data["session_counts"]["stale"], 1)
|
||||
self.assertTrue(data["read_only"])
|
||||
self.assertEqual(data["mutations"], [])
|
||||
|
||||
def test_nav_marks_sessions_live(self):
|
||||
sessions_items = [
|
||||
item for item in iter_nav_items() if item.href == "/sessions"
|
||||
]
|
||||
self.assertEqual(len(sessions_items), 1)
|
||||
self.assertEqual(sessions_items[0].status, "live")
|
||||
self.assertNotIn("/sessions", STUB_PAGES)
|
||||
# Home page should not mark Sessions as stub.
|
||||
home = self.client.get("/")
|
||||
self.assertEqual(home.status_code, 200)
|
||||
self.assertIn('href="/sessions"', home.text)
|
||||
# Stub marker only appears next to remaining stub destinations.
|
||||
self.assertNotIn(
|
||||
'href="/sessions">Sessions</a> <span class="muted">(stub)</span>',
|
||||
home.text,
|
||||
)
|
||||
|
||||
|
||||
class TestDegradedOwnershipAuthority(unittest.TestCase):
|
||||
"""B1: a section that could not be read must never render as absence."""
|
||||
|
||||
def _snapshot(self, inventory: InventorySnapshot) -> SessionViewSnapshot:
|
||||
return SessionViewSnapshot(
|
||||
runtime=_runtime(),
|
||||
inventory=inventory,
|
||||
sessions=_build_session_rows(inventory, contamination=()),
|
||||
contamination_markers=(),
|
||||
)
|
||||
|
||||
def test_unavailable_leases_mark_row_authority_unproven(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
statuses={"leases": STATUS_UNAVAILABLE},
|
||||
)
|
||||
row = _build_session_rows(inventory, contamination=())[0]
|
||||
self.assertEqual(row.lease_ids, ())
|
||||
self.assertEqual(row.lease_authority, STATUS_UNAVAILABLE)
|
||||
# Worktree binding is correlated through lease work numbers, so it
|
||||
# inherits the unreadable lease section.
|
||||
self.assertEqual(row.worktree_authority, STATUS_UNAVAILABLE)
|
||||
self.assertFalse(row.ownership_authority_complete)
|
||||
|
||||
def test_readable_locks_are_not_reported_unbound_when_leases_degrade(self):
|
||||
# The narrow variant: locks hold a real worktree_path and read cleanly,
|
||||
# but the lease section that supplies the correlating work number does
|
||||
# not. The row must say unknown, not "unbound".
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"worktree_path": "branches/feat-issue-641",
|
||||
},
|
||||
),
|
||||
statuses={"leases": STATUS_DEGRADED},
|
||||
)
|
||||
row = _build_session_rows(inventory, contamination=())[0]
|
||||
self.assertEqual(row.worktree_paths, ())
|
||||
self.assertEqual(row.worktree_authority, STATUS_DEGRADED)
|
||||
self.assertFalse(row.ownership_authority_complete)
|
||||
|
||||
def test_degraded_render_says_unknown_not_none_or_unbound(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
statuses={"leases": STATUS_UNAVAILABLE, "locks": STATUS_UNAVAILABLE},
|
||||
)
|
||||
html = render_sessions_page(self._snapshot(inventory))
|
||||
self.assertIn("unknown (inventory unavailable)", html)
|
||||
self.assertIn("authority unproven", html)
|
||||
self.assertIn("Ownership authority incomplete", html)
|
||||
# The affirmative-absence strings must be gone from the row entirely.
|
||||
self.assertNotIn(">none<", html)
|
||||
self.assertNotIn(">unbound<", html)
|
||||
|
||||
def test_clean_inventory_still_renders_affirmative_absence(self):
|
||||
# Guards against over-correcting B1 into "everything is unknown".
|
||||
inventory = _inventory(sessions=(_clean_session(),))
|
||||
html = render_sessions_page(self._snapshot(inventory))
|
||||
self.assertIn(">none<", html)
|
||||
self.assertIn(">unbound<", html)
|
||||
# The column legend mentions "unknown (inventory …)" as static copy, so
|
||||
# assert on the per-row marker and the concrete statuses instead.
|
||||
self.assertNotIn("authority unproven", html)
|
||||
self.assertNotIn("unknown (inventory unavailable)", html)
|
||||
self.assertNotIn("unknown (inventory degraded)", html)
|
||||
self.assertNotIn("Ownership authority incomplete", html)
|
||||
|
||||
def test_json_export_carries_snapshot_and_per_row_authority(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
statuses={"locks": STATUS_UNAVAILABLE},
|
||||
)
|
||||
data = snapshot_to_dict(self._snapshot(inventory))
|
||||
self.assertFalse(data["ownership_authority_complete"])
|
||||
self.assertEqual(
|
||||
data["ownership_section_status"]["locks"], STATUS_UNAVAILABLE
|
||||
)
|
||||
self.assertEqual(data["ownership_section_status"]["leases"], STATUS_OK)
|
||||
self.assertIn("unknown, not unowned", data["ownership_note"])
|
||||
|
||||
row = data["sessions"][0]
|
||||
self.assertTrue(row["lease_authority_complete"])
|
||||
self.assertFalse(row["worktree_authority_complete"])
|
||||
self.assertEqual(row["worktree_authority"], STATUS_UNAVAILABLE)
|
||||
self.assertFalse(row["ownership_authority_complete"])
|
||||
self.assertIn("unknown, not unowned", row["ownership_note"])
|
||||
|
||||
def test_json_export_is_affirmative_when_every_source_reads(self):
|
||||
inventory = _inventory(sessions=(_clean_session(),))
|
||||
data = snapshot_to_dict(self._snapshot(inventory))
|
||||
self.assertTrue(data["ownership_authority_complete"])
|
||||
self.assertTrue(data["sessions"][0]["ownership_authority_complete"])
|
||||
|
||||
def test_missing_session_list_is_not_reported_as_no_sessions(self):
|
||||
inventory = _inventory(statuses={"sessions": STATUS_UNAVAILABLE})
|
||||
html = render_sessions_page(self._snapshot(inventory))
|
||||
self.assertIn("could not be read", html)
|
||||
self.assertIn("not evidence that no sessions exist", html)
|
||||
|
||||
def test_expired_lease_flags_row_as_stale(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
leases=(
|
||||
{
|
||||
"lease_id": "lease-expired-1",
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"status": "active",
|
||||
"expired": True,
|
||||
"work_kind": "issue",
|
||||
"work_number": 641,
|
||||
},
|
||||
),
|
||||
)
|
||||
row = _build_session_rows(inventory, contamination=())[0]
|
||||
self.assertIn("lease-expired", row.stale_flags)
|
||||
self.assertIn("active-lease-past-expiry", row.stale_flags)
|
||||
|
||||
|
||||
class TestContaminationRedaction(unittest.TestCase):
|
||||
"""B2: marker payload text is redacted at the display boundary."""
|
||||
|
||||
def _marker(self, payload: dict) -> ContaminationMarker:
|
||||
return _inspect_contamination(
|
||||
"runtime_recovery_contamination",
|
||||
remote="prgs",
|
||||
inspect=lambda **_k: {
|
||||
"on_disk": True,
|
||||
"has_payload": True,
|
||||
"summary": "",
|
||||
},
|
||||
load=lambda **_k: payload,
|
||||
)
|
||||
|
||||
def test_home_paths_are_collapsed(self):
|
||||
home = os.path.expanduser("~")
|
||||
marker = self._marker(
|
||||
{"command_summary": f"pkill -f {home}/Development/Gitea-Tools/x.py"}
|
||||
)
|
||||
self.assertNotIn(home, marker.command_summary)
|
||||
self.assertIn("~/Development/Gitea-Tools/x.py", marker.command_summary)
|
||||
|
||||
def test_secrets_missed_by_the_write_time_denylist_are_redacted(self):
|
||||
# Each of these was verified in review to survive
|
||||
# stable_branch_push_guard.redact_command untouched.
|
||||
cases = (
|
||||
("curl -H 'X-Api-Key: SUPERSECRET123' https://example.invalid", "SUPERSECRET123"),
|
||||
("cmd --password hunter2 origin master", "hunter2"),
|
||||
("PRIVATE_KEY=abc123 python deploy.py", "abc123"),
|
||||
("fetch https://user:[email protected]/x.git", "user:pw"),
|
||||
)
|
||||
for raw, secret in cases:
|
||||
with self.subTest(raw=raw):
|
||||
marker = self._marker({"command_summary": raw})
|
||||
self.assertNotIn(secret, marker.command_summary)
|
||||
self.assertIn("[redacted]", marker.command_summary)
|
||||
|
||||
def test_command_summary_is_redacted_not_removed(self):
|
||||
# It is legitimate #630 evidence: the operator must still see which
|
||||
# daemon was killed.
|
||||
marker = self._marker(
|
||||
{
|
||||
"command_summary": "pkill -f gitea_mcp_server.py",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"role": "author",
|
||||
}
|
||||
)
|
||||
self.assertIn("pkill -f gitea_mcp_server.py", marker.command_summary)
|
||||
self.assertEqual(marker.reason_class, "manual_daemon_kill")
|
||||
self.assertEqual(marker.session_id, "prgs-author-111-clean")
|
||||
self.assertEqual(marker.role, "author")
|
||||
|
||||
def test_rendered_page_exposes_no_home_path_from_a_marker(self):
|
||||
home = os.path.expanduser("~")
|
||||
marker = self._marker(
|
||||
{
|
||||
"command_summary": f"pkill -f {home}/Development/Gitea-Tools/x.py",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
}
|
||||
)
|
||||
inventory = _inventory(sessions=(_clean_session(),))
|
||||
html = render_sessions_page(
|
||||
SessionViewSnapshot(
|
||||
runtime=_runtime(),
|
||||
inventory=inventory,
|
||||
sessions=_build_session_rows(inventory, (marker,)),
|
||||
contamination_markers=(marker,),
|
||||
)
|
||||
)
|
||||
self.assertIn("Contamination markers", html)
|
||||
self.assertNotIn(home, html)
|
||||
|
||||
|
||||
class TestSessionsPageEscaping(unittest.TestCase):
|
||||
"""Hostile values from every rendered source stay inert (N2)."""
|
||||
|
||||
HOSTILE = '<script>alert("xss")</script>'
|
||||
|
||||
def test_hostile_session_and_marker_values_are_escaped(self):
|
||||
session = dict(_clean_session())
|
||||
session["session_id"] = f"sid-{self.HOSTILE}"
|
||||
session["role"] = self.HOSTILE
|
||||
session["profile"] = self.HOSTILE
|
||||
session["namespace"] = self.HOSTILE
|
||||
session["status"] = self.HOSTILE
|
||||
inventory = _inventory(
|
||||
sessions=(session,),
|
||||
leases=(
|
||||
{
|
||||
"lease_id": self.HOSTILE,
|
||||
"session_id": session["session_id"],
|
||||
"status": "active",
|
||||
"expired": False,
|
||||
"work_kind": self.HOSTILE,
|
||||
"work_number": 641,
|
||||
},
|
||||
),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"worktree_path": self.HOSTILE,
|
||||
},
|
||||
),
|
||||
)
|
||||
marker = ContaminationMarker(
|
||||
kind="runtime_recovery_contamination",
|
||||
on_disk=True,
|
||||
has_payload=True,
|
||||
summary=self.HOSTILE,
|
||||
reason_class=self.HOSTILE,
|
||||
session_id=session["session_id"],
|
||||
role=self.HOSTILE,
|
||||
command_summary=self.HOSTILE,
|
||||
cleared=False,
|
||||
)
|
||||
html = render_sessions_page(
|
||||
SessionViewSnapshot(
|
||||
runtime=_runtime(),
|
||||
inventory=inventory,
|
||||
sessions=_build_session_rows(inventory, (marker,)),
|
||||
contamination_markers=(marker,),
|
||||
)
|
||||
)
|
||||
self.assertNotIn("<script>", html)
|
||||
self.assertNotIn('alert("xss")', html)
|
||||
self.assertIn("<script>", html)
|
||||
|
||||
def test_hostile_values_in_a_degraded_render_are_escaped(self):
|
||||
inventory = _inventory(
|
||||
sessions=(dict(_clean_session(), session_id=f"sid-{self.HOSTILE}"),),
|
||||
statuses={"leases": STATUS_UNAVAILABLE, "locks": STATUS_DEGRADED},
|
||||
)
|
||||
html = render_sessions_page(
|
||||
SessionViewSnapshot(
|
||||
runtime=_runtime(),
|
||||
inventory=inventory,
|
||||
sessions=_build_session_rows(inventory, contamination=()),
|
||||
contamination_markers=(),
|
||||
)
|
||||
)
|
||||
self.assertNotIn("<script>", html)
|
||||
self.assertIn("<script>", html)
|
||||
self.assertIn("unknown (inventory", html)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,408 @@
|
||||
"""Tests for web UI workflow traffic-control view (#640)."""
|
||||
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from webui.app import create_app
|
||||
from webui.traffic_loader import (
|
||||
TrafficItem,
|
||||
TrafficSnapshot,
|
||||
load_traffic_snapshot,
|
||||
snapshot_to_dict,
|
||||
)
|
||||
from webui.traffic_views import render_traffic_page
|
||||
from allocator_service import WorkCandidate
|
||||
|
||||
|
||||
class TestTrafficClassification(unittest.TestCase):
|
||||
def test_runnable_candidate_classification(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Web Console: Workflow traffic-control view (Phase 1)",
|
||||
priority=20,
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
self.assertEqual(len(snap.runnable), 1)
|
||||
self.assertEqual(snap.runnable[0].number, 640)
|
||||
self.assertTrue(snap.runnable[0].is_safe)
|
||||
self.assertEqual(snap.runnable[0].traffic_state, "runnable")
|
||||
|
||||
def test_blocked_dependency_candidate_classification(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=643,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Web Console: Requests & intent preview (Phase 2)",
|
||||
priority=20,
|
||||
dependency_unmet=True,
|
||||
dependency_reason="issue#643 depends on unresolved issue(s) #640; they are not closed",
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
self.assertEqual(len(snap.blocked), 1)
|
||||
self.assertEqual(snap.blocked[0].number, 643)
|
||||
self.assertFalse(snap.blocked[0].is_safe)
|
||||
self.assertEqual(snap.blocked[0].traffic_state, "blocked")
|
||||
self.assertIn("depends on unresolved issue(s) #640", snap.blocked[0].block_reason)
|
||||
|
||||
def test_leased_candidate_classification(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:in-progress",),
|
||||
title="Web Console: Workflow traffic-control view (Phase 1)",
|
||||
priority=20,
|
||||
)
|
||||
lease = {
|
||||
"kind": "issue",
|
||||
"number": 640,
|
||||
"session_id": "prgs-author-12345",
|
||||
"role": "author",
|
||||
"status": "active",
|
||||
}
|
||||
snap = load_traffic_snapshot(candidates=[cand], leases=[lease])
|
||||
self.assertEqual(len(snap.leased), 1)
|
||||
self.assertEqual(snap.leased[0].number, 640)
|
||||
self.assertEqual(snap.leased[0].traffic_state, "leased")
|
||||
self.assertIsNotNone(snap.leased[0].lease_info)
|
||||
|
||||
def test_needs_controller_candidate_classification(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=700,
|
||||
state="open",
|
||||
labels=("status:blocked",),
|
||||
title="Controller intervention needed",
|
||||
priority=10,
|
||||
blocked=True,
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
self.assertEqual(len(snap.needs_controller), 1)
|
||||
self.assertEqual(snap.needs_controller[0].number, 700)
|
||||
|
||||
|
||||
class TestTrafficLoader(unittest.TestCase):
|
||||
def test_snapshot_to_dict_export(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Traffic control test",
|
||||
priority=20,
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
data = snapshot_to_dict(snap)
|
||||
self.assertEqual(data["project_id"], "gitea-tools")
|
||||
self.assertEqual(len(data["runnable"]), 1)
|
||||
self.assertTrue(data["inventory_complete"])
|
||||
|
||||
def test_fail_closed_error_handling(self):
|
||||
with mock.patch("webui.traffic_loader.load_queue_snapshot", side_effect=RuntimeError("Gitea connection failed")):
|
||||
snap = load_traffic_snapshot()
|
||||
self.assertIsNotNone(snap.fetch_error)
|
||||
self.assertIn("Failed to load traffic state", snap.fetch_error)
|
||||
self.assertEqual(len(snap.runnable), 0)
|
||||
self.assertFalse(snap.inventory_complete)
|
||||
|
||||
|
||||
class TestTrafficLivePath(unittest.TestCase):
|
||||
"""Live path tests: inject QueueSnapshot + LeaseSnapshot (no candidates=).
|
||||
|
||||
Covers the production ``load_traffic_snapshot()`` branch that ``/traffic``
|
||||
and ``/api/traffic`` actually execute (#640 B1–B5).
|
||||
"""
|
||||
|
||||
FULL_SHA = "069a9af7e6aa2c2994e07199d1b0814819457017"
|
||||
|
||||
def _queue(
|
||||
self,
|
||||
*,
|
||||
prs=(),
|
||||
issues=(),
|
||||
):
|
||||
from webui.queue_loader import QueueSnapshot
|
||||
|
||||
return QueueSnapshot(
|
||||
project_id="gitea-tools",
|
||||
repo_label="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
prs=tuple(prs),
|
||||
issues=tuple(issues),
|
||||
pr_pagination=None,
|
||||
issue_pagination=None,
|
||||
fetch_error=None,
|
||||
)
|
||||
|
||||
def _lease(
|
||||
self,
|
||||
*,
|
||||
claim_inventory=None,
|
||||
reviewer_leases=(),
|
||||
):
|
||||
from webui.lease_loader import LeaseSnapshot
|
||||
|
||||
return LeaseSnapshot(
|
||||
project_id="gitea-tools",
|
||||
repo_label="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
issue_lock=None,
|
||||
claim_inventory=claim_inventory or {"entries": [], "counts": {}},
|
||||
reviewer_leases=tuple(reviewer_leases),
|
||||
duplicate_prs=(),
|
||||
duplicate_branches=(),
|
||||
collision_history=(),
|
||||
fetch_error=None,
|
||||
)
|
||||
|
||||
def test_live_pr_uses_full_head_sha_and_is_runnable(self):
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
pr = QueueItem(
|
||||
number=885,
|
||||
title="traffic control",
|
||||
badges=("in-review",),
|
||||
extra={"head_sha": self.FULL_SHA[:12], "linked_issue": "640"},
|
||||
signals={
|
||||
"head_sha": self.FULL_SHA,
|
||||
"mergeable": True,
|
||||
"labels": (),
|
||||
"linked_issue": 640,
|
||||
},
|
||||
)
|
||||
q = self._queue(prs=[pr])
|
||||
l = self._lease()
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: q,
|
||||
fetch_lease_snapshot=lambda: l,
|
||||
)
|
||||
self.assertIsNone(snap.fetch_error)
|
||||
self.assertEqual(len(snap.runnable), 1)
|
||||
item = snap.runnable[0]
|
||||
self.assertEqual(item.kind, "pr")
|
||||
self.assertEqual(item.number, 885)
|
||||
self.assertEqual(item.head_sha, self.FULL_SHA)
|
||||
self.assertNotEqual(item.head_sha, self.FULL_SHA[:12])
|
||||
self.assertIsNone(item.block_reason)
|
||||
self.assertEqual(len(snap.blocked), 0)
|
||||
|
||||
def test_live_pr_without_head_sha_is_blocked(self):
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
pr = QueueItem(
|
||||
number=1,
|
||||
title="missing pin",
|
||||
badges=("open",),
|
||||
extra={"head_sha": ""},
|
||||
signals={"head_sha": "", "mergeable": True, "labels": ()},
|
||||
)
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: self._queue(prs=[pr]),
|
||||
fetch_lease_snapshot=lambda: self._lease(),
|
||||
)
|
||||
self.assertEqual(len(snap.blocked) + len(snap.needs_controller), 1)
|
||||
item = (snap.blocked or snap.needs_controller)[0]
|
||||
self.assertIn("missing head_sha", (item.block_reason or "").lower())
|
||||
|
||||
def test_reviewer_lease_keys_by_pr_not_linked_issue(self):
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
pr = QueueItem(
|
||||
number=885,
|
||||
title="leased pr",
|
||||
badges=("in-review",),
|
||||
extra={"head_sha": self.FULL_SHA[:12]},
|
||||
signals={"head_sha": self.FULL_SHA, "mergeable": True, "labels": ()},
|
||||
)
|
||||
issue = QueueItem(
|
||||
number=640,
|
||||
title="linked issue",
|
||||
badges=("open",),
|
||||
extra={},
|
||||
signals={"labels": ()},
|
||||
)
|
||||
# Marker-shaped record: has both pr_number and issue_number; must
|
||||
# attach to the PR only (B2).
|
||||
reviewer_lease = {
|
||||
"pr_number": 885,
|
||||
"issue_number": 640,
|
||||
"phase": "validating",
|
||||
"reviewer_identity": "sysadmin",
|
||||
"session_id": "review-sess-1",
|
||||
}
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: self._queue(prs=[pr], issues=[issue]),
|
||||
fetch_lease_snapshot=lambda: self._lease(reviewer_leases=[reviewer_lease]),
|
||||
)
|
||||
leased_prs = [i for i in snap.leased if i.kind == "pr" and i.number == 885]
|
||||
self.assertEqual(len(leased_prs), 1)
|
||||
self.assertEqual(leased_prs[0].lease_info.get("pr_number"), 885)
|
||||
# Issue 640 must not inherit the reviewer lease just because issue_number
|
||||
# is present on the marker.
|
||||
for item in list(snap.leased) + list(snap.runnable) + list(snap.blocked):
|
||||
if item.kind == "issue" and item.number == 640:
|
||||
self.assertIsNone(
|
||||
item.lease_info,
|
||||
"reviewer lease must not attach to linked issue #640",
|
||||
)
|
||||
break
|
||||
else:
|
||||
self.fail("expected issue #640 in traffic snapshot")
|
||||
|
||||
def test_claim_inventory_entries_key_marks_issue_leased(self):
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
issue = QueueItem(
|
||||
number=640,
|
||||
title="claimed issue",
|
||||
badges=("claimed",),
|
||||
extra={},
|
||||
signals={"labels": ("status:in-progress",)},
|
||||
)
|
||||
inventory = {
|
||||
"entries": [
|
||||
{
|
||||
"issue_number": 640,
|
||||
"status": "active",
|
||||
"latest_heartbeat": {"session_id": "author-sess-9"},
|
||||
"reasons": ["claim has structured heartbeat proof"],
|
||||
}
|
||||
],
|
||||
"counts": {"active": 1},
|
||||
"in_progress_total": 1,
|
||||
}
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: self._queue(issues=[issue]),
|
||||
fetch_lease_snapshot=lambda: self._lease(claim_inventory=inventory),
|
||||
)
|
||||
leased_issues = [i for i in snap.leased if i.kind == "issue" and i.number == 640]
|
||||
self.assertEqual(len(leased_issues), 1)
|
||||
self.assertEqual(leased_issues[0].traffic_state, "leased")
|
||||
|
||||
def test_active_claims_key_is_ignored(self):
|
||||
"""B3 regression: fictional ``active_claims`` must not create lease_info."""
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
issue = QueueItem(
|
||||
number=640,
|
||||
title="open issue",
|
||||
badges=("open",),
|
||||
extra={},
|
||||
signals={"labels": ()},
|
||||
)
|
||||
# Only the broken key — must NOT produce lease_info. Entries-less
|
||||
# inventory is empty (entries is the real claim_inventory key).
|
||||
inventory = {
|
||||
"active_claims": [
|
||||
{
|
||||
"kind": "issue",
|
||||
"number": 640,
|
||||
"issue_number": 640,
|
||||
"status": "active",
|
||||
},
|
||||
],
|
||||
"counts": {},
|
||||
}
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: self._queue(issues=[issue]),
|
||||
fetch_lease_snapshot=lambda: self._lease(claim_inventory=inventory),
|
||||
)
|
||||
items = [
|
||||
i
|
||||
for i in (
|
||||
list(snap.runnable)
|
||||
+ list(snap.leased)
|
||||
+ list(snap.blocked)
|
||||
+ list(snap.needs_controller)
|
||||
)
|
||||
if i.kind == "issue" and i.number == 640
|
||||
]
|
||||
self.assertEqual(len(items), 1)
|
||||
self.assertIsNone(
|
||||
items[0].lease_info,
|
||||
"active_claims is not a real inventory key; entries-only",
|
||||
)
|
||||
|
||||
|
||||
class TestTrafficRoutesAndRendering(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.client = TestClient(create_app())
|
||||
|
||||
def test_traffic_html_page_rendering(self):
|
||||
cand1 = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Traffic View Implementation",
|
||||
priority=20,
|
||||
)
|
||||
cand2 = WorkCandidate(
|
||||
kind="issue",
|
||||
number=643,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Dependent Feature",
|
||||
priority=20,
|
||||
dependency_unmet=True,
|
||||
dependency_reason="issue#643 depends on unresolved issue(s) #640; they are not closed",
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand1, cand2])
|
||||
with mock.patch("webui.app.load_traffic_snapshot", return_value=snap):
|
||||
response = self.client.get("/traffic")
|
||||
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertIn("Workflow Traffic Control", response.text)
|
||||
self.assertIn("1. Runnable Lanes", response.text)
|
||||
self.assertIn("3. Blocked Items", response.text)
|
||||
self.assertIn("Traffic View Implementation", response.text)
|
||||
self.assertIn("depends on unresolved issue(s) #640", response.text)
|
||||
|
||||
def test_api_traffic_json_route(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Traffic View API Test",
|
||||
priority=20,
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
with mock.patch("webui.app.load_traffic_snapshot", return_value=snap):
|
||||
response = self.client.get("/api/traffic")
|
||||
|
||||
self.assertEqual(response.status_code, 200)
|
||||
data = response.json()
|
||||
self.assertEqual(data["project_id"], "gitea-tools")
|
||||
self.assertEqual(len(data["runnable"]), 1)
|
||||
self.assertEqual(data["runnable"][0]["number"], 640)
|
||||
|
||||
def test_render_traffic_fail_closed_page(self):
|
||||
snap = TrafficSnapshot(
|
||||
project_id="gitea-tools",
|
||||
repo_label="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
runnable=(),
|
||||
leased=(),
|
||||
blocked=(),
|
||||
needs_controller=(),
|
||||
terminal_complete=(),
|
||||
next_roles=(),
|
||||
fetch_error="Gitea credentials unavailable for gitea.prgs.cc",
|
||||
inventory_complete=False,
|
||||
)
|
||||
html = render_traffic_page(snap)
|
||||
self.assertIn("Traffic data unavailable", html)
|
||||
self.assertIn("Fail closed", html)
|
||||
self.assertNotIn("1. Runnable Lanes", html)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -42,10 +42,17 @@ from webui.lease_loader import load_lease_snapshot, snapshot_to_dict as lease_sn
|
||||
from webui.lease_views import render_leases_page
|
||||
from webui.queue_loader import load_queue_snapshot, snapshot_to_dict as queue_snapshot_to_dict
|
||||
from webui.queue_views import render_queue_page
|
||||
from webui.traffic_loader import load_traffic_snapshot, snapshot_to_dict as traffic_snapshot_to_dict
|
||||
from webui.traffic_views import render_traffic_page
|
||||
from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as worktree_snapshot_to_dict
|
||||
from webui.worktree_views import render_worktrees_page
|
||||
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
|
||||
from webui.runtime_views import render_runtime_page
|
||||
from webui.session_loader import (
|
||||
load_session_view_snapshot,
|
||||
snapshot_to_dict as session_view_snapshot_to_dict,
|
||||
)
|
||||
from webui.session_views import render_sessions_page
|
||||
from webui.inventory import (
|
||||
SECTION_NAMES as _INVENTORY_SECTIONS,
|
||||
load_inventory_snapshot,
|
||||
@@ -80,10 +87,12 @@ def _stub_page(title: str, description: str) -> HTMLResponse:
|
||||
|
||||
|
||||
_LEGACY_PAGES = (
|
||||
("/traffic", "Traffic", "workflow traffic-control view (#640)"),
|
||||
("/queue", "Queue", "live PR and issue dashboard (#429)"),
|
||||
("/projects", "Projects", "registry and onboarding (#427)"),
|
||||
("/prompts", "Prompts", "canonical workflow prompt library (#428)"),
|
||||
("/runtime", "Runtime", "MCP health and stale-runtime detection (#430)"),
|
||||
("/sessions", "Sessions", "runtime and session view (#641)"),
|
||||
("/audit", "Audit", "final-report paste and validator preview (#431)"),
|
||||
("/worktrees", "Worktrees", "branch hygiene dashboard (#432)"),
|
||||
("/leases", "Leases", "collision and lease visibility (#433)"),
|
||||
@@ -200,6 +209,15 @@ async def api_queue(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(queue_snapshot_to_dict(load_queue_snapshot()))
|
||||
|
||||
|
||||
async def traffic(_request: Request) -> HTMLResponse:
|
||||
snapshot = load_traffic_snapshot()
|
||||
return HTMLResponse(render_traffic_page(snapshot))
|
||||
|
||||
|
||||
async def api_traffic(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(traffic_snapshot_to_dict(load_traffic_snapshot()))
|
||||
|
||||
|
||||
def _load_project_registry() -> tuple[ProjectRegistry | None, RegistryError | None]:
|
||||
"""Load the registry, converting validation failure into a fail-closed pair."""
|
||||
try:
|
||||
@@ -313,6 +331,17 @@ async def api_runtime(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(runtime_snapshot_to_dict(load_runtime_snapshot()))
|
||||
|
||||
|
||||
async def sessions(_request: Request) -> HTMLResponse:
|
||||
"""Runtime and session view (#641) — read-only composition of health + inventory."""
|
||||
snapshot = load_session_view_snapshot()
|
||||
return HTMLResponse(render_sessions_page(snapshot))
|
||||
|
||||
|
||||
async def api_sessions(_request: Request) -> JSONResponse:
|
||||
"""JSON export for the runtime/session view (#641)."""
|
||||
return JSONResponse(session_view_snapshot_to_dict(load_session_view_snapshot()))
|
||||
|
||||
|
||||
async def _parse_audit_form(request: Request) -> tuple[str, str | None]:
|
||||
if request.method == "GET":
|
||||
return "", None
|
||||
@@ -736,6 +765,8 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/system-health", system_health, methods=["GET"]),
|
||||
Route("/queue", queue, methods=["GET"]),
|
||||
Route("/api/queue", api_queue, methods=["GET"]),
|
||||
Route("/traffic", traffic, methods=["GET"]),
|
||||
Route("/api/traffic", api_traffic, methods=["GET"]),
|
||||
Route("/projects", projects, methods=["GET"]),
|
||||
Route("/projects/{project_id}", project_detail, methods=["GET"]),
|
||||
Route("/api/projects", api_projects, methods=["GET"]),
|
||||
@@ -750,6 +781,9 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/api/prompts", api_prompts, methods=["GET"]),
|
||||
Route("/runtime", runtime, methods=["GET"]),
|
||||
Route("/api/runtime", api_runtime, methods=["GET"]),
|
||||
Route("/sessions", sessions, methods=["GET"]),
|
||||
Route("/api/sessions", api_sessions, methods=["GET"]),
|
||||
Route("/api/v1/sessions", api_sessions, methods=["GET"]),
|
||||
Route("/api/v1/timeline", api_v1_timeline, methods=["GET"]),
|
||||
Route("/analytics", analytics, methods=["GET"]),
|
||||
Route("/api/analytics", api_v1_analytics, methods=["GET"]),
|
||||
|
||||
@@ -76,6 +76,35 @@ _CREDENTIAL_KEY_RE = re.compile(
|
||||
)
|
||||
_REDACTED = "[redacted]"
|
||||
|
||||
#: Credential-shaped *name* as it appears inside a free-form command line. This
|
||||
#: is deliberately broader than :data:`_CREDENTIAL_KEY_RE` — it also matches a
|
||||
#: bare ``key`` component, so ``PRIVATE_KEY=`` is caught. Over-redacting a
|
||||
#: displayed string is safe; under-redacting one is not.
|
||||
_TEXT_CREDENTIAL_NAME = (
|
||||
r"[A-Za-z0-9_.\-]*"
|
||||
r"(?:token|secret|password|passwd|key|authorization|bearer|credential)"
|
||||
r"[A-Za-z0-9_.\-]*"
|
||||
)
|
||||
#: A value following such a name: single-quoted, double-quoted, or bare. The
|
||||
#: bare form stops at a quote so an enclosing quote survives the redaction.
|
||||
_TEXT_CREDENTIAL_VALUE = r"'[^']*'|\"[^\"]*\"|[^\s'\"]+"
|
||||
|
||||
_TEXT_CREDENTIAL_FLAG_RE = re.compile(
|
||||
rf"(?P<key>(?<![\w\-])--?{_TEXT_CREDENTIAL_NAME})"
|
||||
rf"(?P<sep>[=\s]+)"
|
||||
rf"(?P<value>{_TEXT_CREDENTIAL_VALUE})",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_TEXT_CREDENTIAL_ASSIGN_RE = re.compile(
|
||||
rf"(?P<key>(?<![\w\-]){_TEXT_CREDENTIAL_NAME})"
|
||||
rf"(?P<sep>\s*[:=]\s*)"
|
||||
rf"(?P<value>{_TEXT_CREDENTIAL_VALUE})",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_TEXT_URL_USERINFO_RE = re.compile(
|
||||
r"(?P<scheme>\b[A-Za-z][A-Za-z0-9+.\-]*://)[^\s/@]+@"
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class InventorySection:
|
||||
@@ -225,6 +254,42 @@ def scrub(value: Any, *, key: str | None = None) -> Any:
|
||||
return repr(value)
|
||||
|
||||
|
||||
def collapse_home(text: str) -> str:
|
||||
"""Collapse every ``$HOME`` occurrence *inside* a string, not just a prefix."""
|
||||
home = os.path.expanduser("~")
|
||||
if not home or home == "/":
|
||||
return text
|
||||
return text.replace(home, "~")
|
||||
|
||||
|
||||
def scrub_text(value: Any) -> Any:
|
||||
"""Redact a free-form text blob such as a recorded command line.
|
||||
|
||||
:func:`scrub` keys off structured field *names* and whole-value prefixes,
|
||||
which is right for inventory records but blind to a secret embedded in the
|
||||
middle of a sentence. This collapses ``$HOME`` and redacts credential-shaped
|
||||
tokens and URL userinfo *anywhere* in the string, so operator-supplied text
|
||||
rendered verbatim — contamination ``command_summary`` (#630) above all — is
|
||||
held to the same standard as every other field on the page.
|
||||
|
||||
Returns ``None`` unchanged so callers can keep "absent" distinct from "".
|
||||
"""
|
||||
if value is None:
|
||||
return None
|
||||
text = value if isinstance(value, str) else str(value)
|
||||
text = collapse_home(text)
|
||||
text = _TEXT_URL_USERINFO_RE.sub(
|
||||
lambda m: f"{m.group('scheme')}{_REDACTED}@", text
|
||||
)
|
||||
text = _TEXT_CREDENTIAL_FLAG_RE.sub(
|
||||
lambda m: f"{m.group('key')}{m.group('sep')}{_REDACTED}", text
|
||||
)
|
||||
text = _TEXT_CREDENTIAL_ASSIGN_RE.sub(
|
||||
lambda m: f"{m.group('key')}{m.group('sep')}{_REDACTED}", text
|
||||
)
|
||||
return text
|
||||
|
||||
|
||||
# ── control-plane database (read-only) ───────────────────────────────────────
|
||||
|
||||
|
||||
|
||||
@@ -182,10 +182,17 @@ def _extract_reviewer_leases(
|
||||
parsed = parse_reviewer_lease_comment(comment.get("body") or "")
|
||||
if not parsed:
|
||||
continue
|
||||
subject_pr = parsed.get("pr_number") or pr_number
|
||||
leases.append(
|
||||
{
|
||||
**parsed,
|
||||
"pr_number": parsed.get("pr_number") or pr_number,
|
||||
"pr_number": subject_pr,
|
||||
# The lease subject is the PR, never the linked issue: a
|
||||
# reviewer lease on PR #N must not be attributed to issue #N
|
||||
# or to the issue that PR closes (#640).
|
||||
"kind": "pr",
|
||||
"number": subject_pr,
|
||||
"role": "reviewer",
|
||||
"comment_id": comment.get("id"),
|
||||
"author": (comment.get("user") or {}).get("login"),
|
||||
"created_at": comment.get("created_at"),
|
||||
|
||||
+2
-6
@@ -41,13 +41,14 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
NavItem("/system-health", "System health"),
|
||||
)),
|
||||
NavGroup("Traffic", (
|
||||
NavItem("/traffic", "Traffic control"),
|
||||
NavItem("/queue", "Queue"),
|
||||
NavItem("/leases", "Leases"),
|
||||
NavItem("/actions", "Actions"),
|
||||
)),
|
||||
NavGroup("Runtime/Sessions", (
|
||||
NavItem("/runtime", "Runtime health"),
|
||||
NavItem("/sessions", "Sessions", "stub"),
|
||||
NavItem("/sessions", "Sessions"),
|
||||
)),
|
||||
NavGroup("Projects", (
|
||||
NavItem("/projects", "Projects"),
|
||||
@@ -75,11 +76,6 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
# issues of epic #631. Each maps a path to (title, description). Routes are
|
||||
# registered so nav links resolve to a graceful, read-only stub page.
|
||||
STUB_PAGES: dict[str, tuple[str, str]] = {
|
||||
"/sessions": (
|
||||
"Sessions",
|
||||
"Active session, capability, and role inventory. Backed by the unified "
|
||||
"inventory API (#636) once it lands.",
|
||||
),
|
||||
"/inventory": (
|
||||
"Inventory",
|
||||
"Unified sessions, leases, locks, namespaces, and worktree inventory. "
|
||||
|
||||
+34
-3
@@ -4,7 +4,7 @@ from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable
|
||||
from urllib.parse import urlparse
|
||||
@@ -31,10 +31,20 @@ class PaginationMeta:
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class QueueItem:
|
||||
"""One queue row.
|
||||
|
||||
``extra`` holds *display* strings for the queue page (values are truncated
|
||||
or humanized for rendering). ``signals`` holds the *authoritative* typed
|
||||
values taken straight from the Gitea payload, for consumers that classify
|
||||
or pin state rather than render it (#640). Never derive identity or
|
||||
concurrency decisions from ``extra``.
|
||||
"""
|
||||
|
||||
number: int
|
||||
title: str
|
||||
badges: tuple[str, ...]
|
||||
extra: dict[str, str]
|
||||
signals: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -134,21 +144,37 @@ def _format_pr_item(pr: dict, badges: tuple[str, ...]) -> QueueItem:
|
||||
"mergeable" if mergeable is True else "conflicted" if mergeable is False else "unknown"
|
||||
)
|
||||
linked = _extract_linked_issue(pr.get("title"), pr.get("body"))
|
||||
head_sha = str(head.get("sha") or "")
|
||||
labels = tuple(
|
||||
str(lb.get("name") or "") for lb in (pr.get("labels") or []) if lb.get("name")
|
||||
)
|
||||
return QueueItem(
|
||||
number=int(pr["number"]),
|
||||
title=str(pr.get("title") or ""),
|
||||
badges=badges,
|
||||
extra={
|
||||
"branch": f"{head.get('ref', '?')} → {base.get('ref', '?')}",
|
||||
"head_sha": str(head.get("sha") or "")[:12],
|
||||
# Display only — truncated. Pin against signals["head_sha"] instead.
|
||||
"head_sha": head_sha[:12],
|
||||
"mergeable": merge_label,
|
||||
"linked_issue": str(linked) if linked is not None else "",
|
||||
},
|
||||
signals={
|
||||
"head_sha": head_sha,
|
||||
"head_ref": str(head.get("ref") or ""),
|
||||
"base_ref": str(base.get("ref") or ""),
|
||||
"mergeable": mergeable if isinstance(mergeable, bool) else None,
|
||||
"labels": labels,
|
||||
"linked_issue": linked,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def _format_issue_item(issue: dict, badges: tuple[str, ...]) -> QueueItem:
|
||||
labels = ", ".join(lb.get("name", "") for lb in issue.get("labels", []))
|
||||
label_names = tuple(
|
||||
str(lb.get("name") or "") for lb in (issue.get("labels") or []) if lb.get("name")
|
||||
)
|
||||
labels = ", ".join(label_names)
|
||||
assignee = (issue.get("assignee") or {}).get("login", "")
|
||||
return QueueItem(
|
||||
number=int(issue["number"]),
|
||||
@@ -159,6 +185,11 @@ def _format_issue_item(issue: dict, badges: tuple[str, ...]) -> QueueItem:
|
||||
"assignee": assignee or "unassigned",
|
||||
"state": str(issue.get("state") or ""),
|
||||
},
|
||||
signals={
|
||||
"labels": label_names,
|
||||
"assignee": assignee,
|
||||
"state": str(issue.get("state") or ""),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -88,5 +88,7 @@ def render_runtime_page(snapshot: RuntimeSnapshot) -> str:
|
||||
"<p class='muted'>MVP is read-only — restart MCP servers from your IDE/operator "
|
||||
"workflow. Related issue: <code>#420</code>. Guidance: "
|
||||
f"<code>{html.escape(snapshot.restart_guidance)}</code></p>"
|
||||
"<p class='muted'>This page does not expose tokens or perform MCP restarts.</p>"
|
||||
)
|
||||
"<p class='muted'>This page does not expose tokens or perform MCP restarts. "
|
||||
"Correlated sessions, worktree bindings, and contamination markers: "
|
||||
"<a href='/sessions'>/sessions</a> (#641).</p>"
|
||||
)
|
||||
|
||||
@@ -0,0 +1,517 @@
|
||||
"""Compose runtime health + inventory into a sessions/runtime view (#641).
|
||||
|
||||
Phase 1 is read-only. It correlates namespaces, sessions, capabilities,
|
||||
worktree bindings, lease ownership, stale flags, and contamination markers
|
||||
when they are detectable on disk (#630 / #671). It never restarts, kills, or
|
||||
takes over a session.
|
||||
|
||||
Sources:
|
||||
|
||||
* :mod:`webui.runtime_health` — profile, role, stale runtime, shell health.
|
||||
* :mod:`webui.inventory` — sessions, leases, locks, worktrees, namespaces
|
||||
and collision signals from the control-plane DB + filesystem.
|
||||
* :mod:`mcp_session_state` — durable contamination markers (inspect only).
|
||||
|
||||
Secrets are never read. Inventory-sourced values arrive already redacted by
|
||||
:func:`webui.inventory.scrub`; free-form marker text this module loads itself is
|
||||
put through :func:`webui.inventory.scrub_text`, which collapses ``$HOME`` and
|
||||
redacts credential-shaped tokens *inside* a string rather than only at its start.
|
||||
|
||||
Ownership columns are authority-aware. A lease or lock section that could not be
|
||||
read renders as ``unknown``, never as ``none`` or ``unbound``: a lease the reader
|
||||
could not load is not an absent lease (see
|
||||
:attr:`webui.inventory.InventorySnapshot.ownership_authority_complete`).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable
|
||||
|
||||
import mcp_session_state
|
||||
from webui.inventory import (
|
||||
OWNERSHIP_SECTIONS,
|
||||
STATUS_OK,
|
||||
InventorySection,
|
||||
InventorySnapshot,
|
||||
load_inventory_snapshot,
|
||||
scrub_text,
|
||||
snapshot_to_dict as inventory_snapshot_to_dict,
|
||||
)
|
||||
from webui.runtime_health import (
|
||||
RuntimeSnapshot,
|
||||
load_runtime_snapshot,
|
||||
snapshot_to_dict as runtime_snapshot_to_dict,
|
||||
)
|
||||
|
||||
# Sanctioned recovery pointers only — never pkill / killall (#630).
|
||||
SANCTIONED_RECOVERY_DOCS: tuple[dict[str, str], ...] = (
|
||||
{
|
||||
"label": "MCP namespace EOF recovery (reconnect only)",
|
||||
"path": "docs/mcp-namespace-eof-recovery.md",
|
||||
"note": "IDE/client reconnect or operator-owned restart; never kill daemons.",
|
||||
},
|
||||
{
|
||||
"label": "MCP namespace health",
|
||||
"path": "docs/mcp-namespace-health.md",
|
||||
"note": "client_namespace probe proves namespace health.",
|
||||
},
|
||||
{
|
||||
"label": "Restart path inventory",
|
||||
"path": "docs/mcp-restart-path-inventory.md",
|
||||
"note": "Catalog of sanctioned reconnect/restart paths.",
|
||||
},
|
||||
{
|
||||
"label": "Local web UI recovery",
|
||||
"path": "docs/webui-local-dev.md",
|
||||
"note": "Operator console start and documented recovery sequence.",
|
||||
},
|
||||
)
|
||||
|
||||
_CONTAMINATION_KINDS: tuple[str, ...] = (
|
||||
mcp_session_state.KIND_RUNTIME_RECOVERY_CONTAMINATION,
|
||||
mcp_session_state.KIND_STABLE_BRANCH_CONTAMINATION,
|
||||
)
|
||||
|
||||
#: Status recorded on a row when the backing inventory section is absent
|
||||
#: entirely — distinct from a section that reported itself degraded.
|
||||
AUTHORITY_MISSING = "missing"
|
||||
|
||||
|
||||
def _section_status(section: InventorySection | None) -> str:
|
||||
"""Status of an ownership section, treating an absent section as missing."""
|
||||
if section is None:
|
||||
return AUTHORITY_MISSING
|
||||
return section.status
|
||||
|
||||
|
||||
def _combined_authority(*statuses: str) -> str:
|
||||
"""Worst status of the sections a derived column depends on.
|
||||
|
||||
A column proved from two sections is only trustworthy when *both* read
|
||||
cleanly, so the first non-``ok`` status wins.
|
||||
"""
|
||||
for status in statuses:
|
||||
if status != STATUS_OK:
|
||||
return status
|
||||
return STATUS_OK
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ContaminationMarker:
|
||||
"""A detectable durable contamination marker (audit-safe summary)."""
|
||||
|
||||
kind: str
|
||||
on_disk: bool
|
||||
has_payload: bool
|
||||
summary: str
|
||||
reason_class: str | None = None
|
||||
session_id: str | None = None
|
||||
role: str | None = None
|
||||
command_summary: str | None = None
|
||||
cleared: bool = False
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"kind": self.kind,
|
||||
"on_disk": self.on_disk,
|
||||
"has_payload": self.has_payload,
|
||||
"summary": self.summary,
|
||||
"reason_class": self.reason_class,
|
||||
"session_id": self.session_id,
|
||||
"role": self.role,
|
||||
"command_summary": self.command_summary,
|
||||
"cleared": self.cleared,
|
||||
"active": self.on_disk and self.has_payload and not self.cleared,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SessionRow:
|
||||
"""One correlated session row for the sessions table."""
|
||||
|
||||
session_id: str
|
||||
role: str | None
|
||||
profile: str | None
|
||||
namespace: str | None
|
||||
pid: int | None
|
||||
pid_alive: bool | None
|
||||
status: str | None
|
||||
started_at: str | None
|
||||
last_heartbeat_at: str | None
|
||||
lease_ids: tuple[str, ...] = ()
|
||||
work_refs: tuple[str, ...] = ()
|
||||
worktree_paths: tuple[str, ...] = ()
|
||||
stale_flags: tuple[str, ...] = ()
|
||||
contamination_flags: tuple[str, ...] = ()
|
||||
#: Status of the section backing ``lease_ids``/``work_refs``. While this is
|
||||
#: not ``ok`` those tuples mean "could not be read", never "none held".
|
||||
lease_authority: str = STATUS_OK
|
||||
#: Worst status across the sections backing ``worktree_paths`` (locks are
|
||||
#: correlated through lease work numbers, so both must read cleanly).
|
||||
worktree_authority: str = STATUS_OK
|
||||
|
||||
@property
|
||||
def ownership_authority_complete(self) -> bool:
|
||||
"""True only when this row's ownership columns are provable."""
|
||||
return self.lease_authority == STATUS_OK and self.worktree_authority == STATUS_OK
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"session_id": self.session_id,
|
||||
"role": self.role,
|
||||
"profile": self.profile,
|
||||
"namespace": self.namespace,
|
||||
"pid": self.pid,
|
||||
"pid_alive": self.pid_alive,
|
||||
"status": self.status,
|
||||
"started_at": self.started_at,
|
||||
"last_heartbeat_at": self.last_heartbeat_at,
|
||||
"lease_ids": list(self.lease_ids),
|
||||
"work_refs": list(self.work_refs),
|
||||
"worktree_paths": list(self.worktree_paths),
|
||||
"stale_flags": list(self.stale_flags),
|
||||
"contamination_flags": list(self.contamination_flags),
|
||||
"is_stale": bool(self.stale_flags),
|
||||
"is_contaminated": bool(self.contamination_flags),
|
||||
"lease_authority": self.lease_authority,
|
||||
"worktree_authority": self.worktree_authority,
|
||||
"lease_authority_complete": self.lease_authority == STATUS_OK,
|
||||
"worktree_authority_complete": self.worktree_authority == STATUS_OK,
|
||||
"ownership_authority_complete": self.ownership_authority_complete,
|
||||
"ownership_note": (
|
||||
"Lease and worktree columns are proved from sections that read "
|
||||
"cleanly."
|
||||
if self.ownership_authority_complete
|
||||
else "An ownership source could not be read; empty lease_ids or "
|
||||
"worktree_paths on this row mean unknown, not unowned."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SessionViewSnapshot:
|
||||
"""Composed runtime + session inventory view (#641)."""
|
||||
|
||||
runtime: RuntimeSnapshot
|
||||
inventory: InventorySnapshot
|
||||
sessions: tuple[SessionRow, ...]
|
||||
contamination_markers: tuple[ContaminationMarker, ...]
|
||||
recovery_docs: tuple[dict[str, str], ...] = SANCTIONED_RECOVERY_DOCS
|
||||
fetch_error: str | None = None
|
||||
|
||||
@property
|
||||
def stale_session_count(self) -> int:
|
||||
return sum(1 for row in self.sessions if row.stale_flags)
|
||||
|
||||
@property
|
||||
def contaminated_session_count(self) -> int:
|
||||
return sum(1 for row in self.sessions if row.contamination_flags)
|
||||
|
||||
@property
|
||||
def active_contamination(self) -> tuple[ContaminationMarker, ...]:
|
||||
return tuple(m for m in self.contamination_markers if m.to_dict()["active"])
|
||||
|
||||
@property
|
||||
def ownership_authority_complete(self) -> bool:
|
||||
"""False while any ownership section is degraded, absent, or unavailable."""
|
||||
return self.inventory.ownership_authority_complete
|
||||
|
||||
@property
|
||||
def ownership_section_status(self) -> dict[str, str]:
|
||||
"""Per-section status for the three ownership-bearing sections."""
|
||||
return {
|
||||
name: _section_status(self.inventory.section(name))
|
||||
for name in OWNERSHIP_SECTIONS
|
||||
}
|
||||
|
||||
|
||||
def _inspect_contamination(
|
||||
kind: str,
|
||||
*,
|
||||
remote: str | None,
|
||||
inspect: Callable[..., dict[str, Any]] | None = None,
|
||||
load: Callable[..., dict[str, Any] | None] | None = None,
|
||||
) -> ContaminationMarker:
|
||||
"""Inspect one contamination kind; never raises into the page render path."""
|
||||
inspect_fn = inspect or mcp_session_state.inspect_state_envelope
|
||||
load_fn = load or mcp_session_state.load_state
|
||||
try:
|
||||
envelope = inspect_fn(kind=kind, remote=remote)
|
||||
except Exception as exc: # noqa: BLE001 — fail soft for the dashboard
|
||||
return ContaminationMarker(
|
||||
kind=kind,
|
||||
on_disk=False,
|
||||
has_payload=False,
|
||||
summary=f"contamination inspect failed: {type(exc).__name__}",
|
||||
)
|
||||
|
||||
reason_class = None
|
||||
session_id = None
|
||||
role = None
|
||||
command_summary = None
|
||||
cleared = False
|
||||
summary = scrub_text(str(envelope.get("summary") or ""))
|
||||
|
||||
if envelope.get("on_disk") and envelope.get("has_payload"):
|
||||
try:
|
||||
payload = load_fn(kind=kind, remote=remote) or {}
|
||||
except Exception: # noqa: BLE001
|
||||
payload = {}
|
||||
if isinstance(payload, dict):
|
||||
reason_class = payload.get("reason_class")
|
||||
session_id = payload.get("session_id")
|
||||
role = payload.get("role")
|
||||
command_summary = payload.get("command_summary") or payload.get("detail")
|
||||
cleared = bool(payload.get("cleared_by_reconciler"))
|
||||
if not summary:
|
||||
summary = scrub_text(
|
||||
f"{kind}: {reason_class or 'present'}"
|
||||
+ (" (cleared)" if cleared else "")
|
||||
)
|
||||
|
||||
# Marker payloads are operator-supplied free text and are the one thing on
|
||||
# this page that does not arrive through inventory scrubbing. The write-time
|
||||
# redactor is a narrow denylist, so redact again at the display boundary:
|
||||
# it leaves $HOME paths, `-H 'X-Api-Key: …'`, `--password …`, and
|
||||
# `PRIVATE_KEY=…` intact. The field itself stays — it is #630 evidence.
|
||||
return ContaminationMarker(
|
||||
kind=kind,
|
||||
on_disk=bool(envelope.get("on_disk")),
|
||||
has_payload=bool(envelope.get("has_payload")),
|
||||
summary=summary or f"{kind}: not present",
|
||||
reason_class=scrub_text(str(reason_class)) if reason_class else None,
|
||||
session_id=scrub_text(str(session_id)) if session_id else None,
|
||||
role=scrub_text(str(role)) if role else None,
|
||||
command_summary=scrub_text(str(command_summary)) if command_summary else None,
|
||||
cleared=cleared,
|
||||
)
|
||||
|
||||
|
||||
def _load_contamination_markers(
|
||||
*,
|
||||
remote: str | None,
|
||||
inspect: Callable[..., dict[str, Any]] | None = None,
|
||||
load: Callable[..., dict[str, Any] | None] | None = None,
|
||||
) -> tuple[ContaminationMarker, ...]:
|
||||
return tuple(
|
||||
_inspect_contamination(kind, remote=remote, inspect=inspect, load=load)
|
||||
for kind in _CONTAMINATION_KINDS
|
||||
)
|
||||
|
||||
|
||||
def _build_session_rows(
|
||||
inventory: InventorySnapshot,
|
||||
contamination: tuple[ContaminationMarker, ...],
|
||||
) -> tuple[SessionRow, ...]:
|
||||
sessions_section = inventory.section("sessions")
|
||||
leases_section = inventory.section("leases")
|
||||
locks_section = inventory.section("locks")
|
||||
|
||||
# Ownership columns may only assert absence when their source read cleanly.
|
||||
lease_authority = _section_status(leases_section)
|
||||
# Locks carry claimant profile/username, not control-plane session ids, so a
|
||||
# worktree binding is correlated through lease work numbers: it depends on
|
||||
# the locks *and* the leases section.
|
||||
worktree_authority = _combined_authority(
|
||||
lease_authority, _section_status(locks_section)
|
||||
)
|
||||
|
||||
leases_by_session: dict[str, list[dict[str, Any]]] = {}
|
||||
for lease in (leases_section.items if leases_section else ()):
|
||||
sid = str(lease.get("session_id") or "")
|
||||
if sid:
|
||||
leases_by_session.setdefault(sid, []).append(lease)
|
||||
|
||||
active_markers = [m for m in contamination if m.to_dict()["active"]]
|
||||
marker_session_ids = {
|
||||
m.session_id for m in active_markers if m.session_id
|
||||
}
|
||||
|
||||
rows: list[SessionRow] = []
|
||||
for raw in sessions_section.items if sessions_section else ():
|
||||
sid = str(raw.get("session_id") or "")
|
||||
if not sid:
|
||||
continue
|
||||
session_leases = leases_by_session.get(sid, [])
|
||||
lease_ids = tuple(
|
||||
str(lease["lease_id"])
|
||||
for lease in session_leases
|
||||
if lease.get("lease_id")
|
||||
)
|
||||
work_refs: list[str] = []
|
||||
work_numbers: list[int] = []
|
||||
for lease in session_leases:
|
||||
kind = lease.get("work_kind")
|
||||
number = lease.get("work_number")
|
||||
if kind and number is not None:
|
||||
work_refs.append(f"{kind}#{number}")
|
||||
try:
|
||||
work_numbers.append(int(number))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
|
||||
worktree_paths: list[str] = []
|
||||
for lock in (locks_section.items if locks_section else ()):
|
||||
try:
|
||||
issue_no = int(lock.get("issue_number"))
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
if issue_no in work_numbers and lock.get("worktree_path"):
|
||||
worktree_paths.append(str(lock["worktree_path"]))
|
||||
|
||||
stale_flags: list[str] = []
|
||||
if raw.get("pid_alive") is False:
|
||||
stale_flags.append("pid-dead")
|
||||
status = str(raw.get("status") or "").lower()
|
||||
if status and status not in {"active", "alive", "running", "ok"}:
|
||||
stale_flags.append(f"status:{status}")
|
||||
for lease in session_leases:
|
||||
if lease.get("expired") is True:
|
||||
stale_flags.append("lease-expired")
|
||||
if str(lease.get("status") or "").lower() == "active" and lease.get(
|
||||
"expired"
|
||||
) is True:
|
||||
stale_flags.append("active-lease-past-expiry")
|
||||
|
||||
contamination_flags: list[str] = []
|
||||
if sid in marker_session_ids:
|
||||
for marker in active_markers:
|
||||
if marker.session_id == sid:
|
||||
contamination_flags.append(marker.kind)
|
||||
# Process-wide contamination with no session binding still surfaces
|
||||
# against every live session so it cannot be silent (#630).
|
||||
for marker in active_markers:
|
||||
if not marker.session_id and marker.kind not in contamination_flags:
|
||||
contamination_flags.append(f"{marker.kind}:process-wide")
|
||||
|
||||
rows.append(
|
||||
SessionRow(
|
||||
session_id=sid,
|
||||
role=raw.get("role"),
|
||||
profile=raw.get("profile"),
|
||||
namespace=raw.get("namespace"),
|
||||
pid=raw.get("pid") if isinstance(raw.get("pid"), int) else None,
|
||||
pid_alive=raw.get("pid_alive")
|
||||
if isinstance(raw.get("pid_alive"), bool)
|
||||
else None,
|
||||
status=raw.get("status"),
|
||||
started_at=raw.get("started_at"),
|
||||
last_heartbeat_at=raw.get("last_heartbeat_at"),
|
||||
lease_ids=lease_ids,
|
||||
work_refs=tuple(work_refs),
|
||||
worktree_paths=tuple(worktree_paths),
|
||||
stale_flags=tuple(dict.fromkeys(stale_flags)),
|
||||
contamination_flags=tuple(dict.fromkeys(contamination_flags)),
|
||||
lease_authority=lease_authority,
|
||||
worktree_authority=worktree_authority,
|
||||
)
|
||||
)
|
||||
|
||||
return tuple(rows)
|
||||
|
||||
|
||||
def load_session_view_snapshot(
|
||||
*,
|
||||
load_runtime: Callable[..., RuntimeSnapshot] | None = None,
|
||||
load_inventory: Callable[..., InventorySnapshot] | None = None,
|
||||
inspect_contamination: Callable[..., dict[str, Any]] | None = None,
|
||||
load_contamination_payload: Callable[..., dict[str, Any] | None] | None = None,
|
||||
) -> SessionViewSnapshot:
|
||||
"""Build the composed sessions/runtime view. Fail-soft on partial sources."""
|
||||
runtime_loader = load_runtime or load_runtime_snapshot
|
||||
inventory_loader = load_inventory or load_inventory_snapshot
|
||||
fetch_error: str | None = None
|
||||
|
||||
try:
|
||||
runtime = runtime_loader()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
fetch_error = f"runtime snapshot failed: {type(exc).__name__}: {exc}"
|
||||
# Minimal placeholder so the page still renders inventory + recovery.
|
||||
from webui.runtime_health import RuntimeSnapshot as _RS
|
||||
|
||||
runtime = _RS(
|
||||
project_id="unknown",
|
||||
repo_root="",
|
||||
remote="",
|
||||
host="",
|
||||
profile_name="unknown",
|
||||
role_kind="unknown",
|
||||
config_model="unknown",
|
||||
profile_mode="unknown",
|
||||
profile_source="unknown",
|
||||
authenticated_username=None,
|
||||
identity_error=str(exc),
|
||||
repo_sha=None,
|
||||
remote_master_sha=None,
|
||||
commits_behind_master=None,
|
||||
stale_runtime_warning=None,
|
||||
shell_health={},
|
||||
workflow_hashes=(),
|
||||
schema_hashes=(),
|
||||
restart_guidance="docs/mcp-namespace-eof-recovery.md",
|
||||
fetch_error=str(exc),
|
||||
)
|
||||
|
||||
try:
|
||||
inventory = inventory_loader()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
msg = f"inventory snapshot failed: {type(exc).__name__}: {exc}"
|
||||
fetch_error = f"{fetch_error}; {msg}" if fetch_error else msg
|
||||
inventory = load_inventory_snapshot(
|
||||
db_path="/nonexistent-for-fail-soft",
|
||||
lock_dir="/nonexistent-for-fail-soft",
|
||||
)
|
||||
|
||||
remote = getattr(runtime, "remote", None)
|
||||
contamination = _load_contamination_markers(
|
||||
remote=remote,
|
||||
inspect=inspect_contamination,
|
||||
load=load_contamination_payload,
|
||||
)
|
||||
sessions = _build_session_rows(inventory, contamination)
|
||||
return SessionViewSnapshot(
|
||||
runtime=runtime,
|
||||
inventory=inventory,
|
||||
sessions=sessions,
|
||||
contamination_markers=contamination,
|
||||
fetch_error=fetch_error,
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: SessionViewSnapshot) -> dict[str, Any]:
|
||||
"""JSON export for ``/api/sessions`` (read-only)."""
|
||||
return {
|
||||
"api_version": "v1",
|
||||
"view": "runtime-sessions",
|
||||
"issue": 641,
|
||||
"fetch_error": snapshot.fetch_error,
|
||||
"runtime": runtime_snapshot_to_dict(snapshot.runtime),
|
||||
"inventory": inventory_snapshot_to_dict(snapshot.inventory),
|
||||
"sessions": [row.to_dict() for row in snapshot.sessions],
|
||||
"session_counts": {
|
||||
"total": len(snapshot.sessions),
|
||||
"stale": snapshot.stale_session_count,
|
||||
"contaminated": snapshot.contaminated_session_count,
|
||||
},
|
||||
"ownership_authority_complete": snapshot.ownership_authority_complete,
|
||||
"ownership_section_status": snapshot.ownership_section_status,
|
||||
"ownership_note": (
|
||||
"Every ownership source read cleanly; a session with no lease and no "
|
||||
"worktree path genuinely holds neither."
|
||||
if snapshot.ownership_authority_complete
|
||||
else "An ownership source is degraded or unavailable. Empty lease_ids "
|
||||
"and worktree_paths mean unknown, not unowned; consult each row's "
|
||||
"lease_authority and worktree_authority."
|
||||
),
|
||||
"contamination_markers": [
|
||||
marker.to_dict() for marker in snapshot.contamination_markers
|
||||
],
|
||||
"active_contamination": [
|
||||
marker.to_dict() for marker in snapshot.active_contamination
|
||||
],
|
||||
"recovery_docs": [dict(doc) for doc in snapshot.recovery_docs],
|
||||
"read_only": True,
|
||||
"phase": 1,
|
||||
"mutations": [],
|
||||
}
|
||||
@@ -0,0 +1,403 @@
|
||||
"""HTML views for the Runtime and session view (Phase 1, #641).
|
||||
|
||||
Read-only composition of runtime health (#430) and inventory sessions /
|
||||
namespaces / worktrees (#636). Surfaces stale and contamination indicators
|
||||
when detectable. Recovery links point only at sanctioned reconnect/restart
|
||||
docs — never at manual process kill (#630).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from html import escape
|
||||
from typing import Sequence
|
||||
|
||||
from webui.inventory import STATUS_OK
|
||||
from webui.layout import render_page
|
||||
from webui.session_loader import (
|
||||
ContaminationMarker,
|
||||
SessionRow,
|
||||
SessionViewSnapshot,
|
||||
)
|
||||
|
||||
|
||||
def _badge(text: str, css: str) -> str:
|
||||
return f'<span class="badge {css}">{escape(text)}</span>'
|
||||
|
||||
|
||||
def _flags(flags: Sequence[str], *, css: str) -> str:
|
||||
if not flags:
|
||||
return '<span class="muted">—</span>'
|
||||
return " ".join(_badge(flag, css) for flag in flags)
|
||||
|
||||
|
||||
def _unproven_cell(status: str) -> str:
|
||||
"""Render an ownership column whose backing inventory section failed to read.
|
||||
|
||||
Never "none" and never "unbound": an unreadable source proves nothing about
|
||||
ownership, and claiming otherwise is the exact failure the
|
||||
``ownership_authority_complete`` invariant exists to prevent.
|
||||
"""
|
||||
return (
|
||||
f'<span class="muted">unknown (inventory {escape(status)})</span><br>'
|
||||
f'{_badge("authority unproven", "badge-health-degraded")}'
|
||||
)
|
||||
|
||||
|
||||
def _runtime_banner(snapshot: SessionViewSnapshot) -> str:
|
||||
runtime = snapshot.runtime
|
||||
stale = runtime.stale_runtime_warning
|
||||
stale_html = ""
|
||||
if stale:
|
||||
stale_html = (
|
||||
f'<div class="health-card health-stale" style="margin-top:0.75rem;">'
|
||||
f"<strong>Stale runtime:</strong> {escape(stale)}</div>"
|
||||
)
|
||||
identity = runtime.authenticated_username or "unresolved"
|
||||
if runtime.identity_error:
|
||||
identity = f"unresolved ({runtime.identity_error})"
|
||||
|
||||
return f"""<div class="health-card">
|
||||
<h3>Runtime context</h3>
|
||||
<table class="detail">
|
||||
<tr><th>Profile</th><td><code>{escape(runtime.profile_name)}</code></td></tr>
|
||||
<tr><th>Role kind</th><td>{escape(runtime.role_kind)}</td></tr>
|
||||
<tr><th>Identity</th><td>{escape(str(identity))}</td></tr>
|
||||
<tr><th>Remote / host</th>
|
||||
<td><code>{escape(runtime.remote)}</code> · <code>{escape(runtime.host)}</code></td>
|
||||
</tr>
|
||||
<tr><th>Local HEAD</th>
|
||||
<td><code>{escape(runtime.repo_sha or "unknown")}</code></td>
|
||||
</tr>
|
||||
<tr><th>Remote master</th>
|
||||
<td><code>{escape(runtime.remote_master_sha or "unknown")}</code></td>
|
||||
</tr>
|
||||
<tr><th>Commits behind</th>
|
||||
<td>{escape(str(runtime.commits_behind_master if runtime.commits_behind_master is not None else "unknown"))}</td>
|
||||
</tr>
|
||||
</table>
|
||||
<p class="muted">Full runtime detail: <a href="/runtime">/runtime</a> ·
|
||||
Inventory API: <a href="/api/v1/inventory"><code>/api/v1/inventory</code></a></p>
|
||||
{stale_html}
|
||||
</div>"""
|
||||
|
||||
|
||||
def _summary_bar(snapshot: SessionViewSnapshot) -> str:
|
||||
total = len(snapshot.sessions)
|
||||
stale = snapshot.stale_session_count
|
||||
contaminated = snapshot.contaminated_session_count
|
||||
active_markers = len(snapshot.active_contamination)
|
||||
inv_status = snapshot.inventory.status
|
||||
authority_complete = snapshot.ownership_authority_complete
|
||||
authority_text = "complete" if authority_complete else "incomplete"
|
||||
authority_css = "badge-health-ok" if authority_complete else "badge-health-degraded"
|
||||
return f"""<div class="health-card" style="display:flex; flex-wrap:wrap; gap:1rem; align-items:center;">
|
||||
<div><strong>Sessions:</strong> <span class="badge badge-health-ok">{total}</span></div>
|
||||
<div><strong>Stale:</strong> <span class="badge badge-stale">{stale}</span></div>
|
||||
<div><strong>Contaminated:</strong> <span class="badge badge-blocked">{contaminated}</span></div>
|
||||
<div><strong>Active markers:</strong> <span class="badge badge-health-unproven">{active_markers}</span></div>
|
||||
<div><strong>Inventory:</strong> <span class="badge badge-health-skipped">{escape(inv_status)}</span></div>
|
||||
<div><strong>Ownership authority:</strong> <span class="badge {authority_css}">{authority_text}</span></div>
|
||||
</div>"""
|
||||
|
||||
|
||||
def _ownership_caveat(snapshot: SessionViewSnapshot) -> str:
|
||||
"""Name the unreadable ownership sections, or render nothing when all read."""
|
||||
if snapshot.ownership_authority_complete:
|
||||
return ""
|
||||
degraded = ", ".join(
|
||||
f"{name}: {status}"
|
||||
for name, status in snapshot.ownership_section_status.items()
|
||||
if status != STATUS_OK
|
||||
)
|
||||
return (
|
||||
'<div class="health-card health-stale">'
|
||||
"<strong>Ownership authority incomplete:</strong> "
|
||||
f"{escape(degraded)}. Columns marked <em>unknown</em> could not be read. "
|
||||
"No session below may be treated as holding no lease or no worktree "
|
||||
"binding — absence of evidence is not evidence of absence.</div>"
|
||||
)
|
||||
|
||||
|
||||
def _render_session_row(row: SessionRow) -> str:
|
||||
pid = "—" if row.pid is None else str(row.pid)
|
||||
pid_alive = "—" if row.pid_alive is None else ("alive" if row.pid_alive else "dead")
|
||||
pid_css = (
|
||||
"badge-health-ok"
|
||||
if row.pid_alive is True
|
||||
else ("badge-blocked" if row.pid_alive is False else "badge-health-skipped")
|
||||
)
|
||||
# An empty tuple only means "holds none" when its source read cleanly.
|
||||
if row.lease_authority != STATUS_OK:
|
||||
lease_cell = _unproven_cell(row.lease_authority)
|
||||
else:
|
||||
leases = (
|
||||
", ".join(f"<code>{escape(lid)}</code>" for lid in row.lease_ids)
|
||||
if row.lease_ids
|
||||
else '<span class="muted">none</span>'
|
||||
)
|
||||
work = (
|
||||
", ".join(escape(ref) for ref in row.work_refs)
|
||||
if row.work_refs
|
||||
else '<span class="muted">—</span>'
|
||||
)
|
||||
lease_cell = (
|
||||
f'{leases}<div class="muted" style="font-size:0.82rem; '
|
||||
f'margin-top:0.2rem;">{work}</div>'
|
||||
)
|
||||
|
||||
if row.worktree_authority != STATUS_OK:
|
||||
worktrees = _unproven_cell(row.worktree_authority)
|
||||
else:
|
||||
worktrees = (
|
||||
"<br>".join(f"<code>{escape(path)}</code>" for path in row.worktree_paths)
|
||||
if row.worktree_paths
|
||||
else '<span class="muted">unbound</span>'
|
||||
)
|
||||
return f"""<tr>
|
||||
<td><code>{escape(row.session_id)}</code></td>
|
||||
<td>
|
||||
<div><code>{escape(str(row.role or "—"))}</code> / <code>{escape(str(row.profile or "—"))}</code></div>
|
||||
<div class="muted" style="font-size:0.82rem;">ns: <code>{escape(str(row.namespace or "—"))}</code></div>
|
||||
</td>
|
||||
<td>
|
||||
<code>{escape(pid)}</code>
|
||||
{_badge(pid_alive, pid_css)}
|
||||
</td>
|
||||
<td>{escape(str(row.status or "—"))}<div class="muted" style="font-size:0.82rem;">{escape(str(row.last_heartbeat_at or ""))}</div></td>
|
||||
<td>{lease_cell}</td>
|
||||
<td>{worktrees}</td>
|
||||
<td>{_flags(row.stale_flags, css="badge-stale")}</td>
|
||||
<td>{_flags(row.contamination_flags, css="badge-blocked")}</td>
|
||||
</tr>"""
|
||||
|
||||
|
||||
def _sessions_table(snapshot: SessionViewSnapshot) -> str:
|
||||
rows: Sequence[SessionRow] = snapshot.sessions
|
||||
if not rows:
|
||||
sessions_status = snapshot.ownership_section_status.get("sessions", STATUS_OK)
|
||||
if sessions_status != STATUS_OK:
|
||||
return (
|
||||
'<p class="muted">Session inventory is '
|
||||
f"<strong>{escape(sessions_status)}</strong> — the session list "
|
||||
"could not be read. This is not evidence that no sessions "
|
||||
"exist.</p>"
|
||||
)
|
||||
return (
|
||||
'<p class="muted">No control-plane sessions recorded. Inventory may '
|
||||
"be unavailable, or no MCP workers have registered yet.</p>"
|
||||
)
|
||||
body = "".join(_render_session_row(row) for row in rows)
|
||||
return f"""<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Session</th>
|
||||
<th>Role / profile / namespace</th>
|
||||
<th>PID</th>
|
||||
<th>Status</th>
|
||||
<th>Leases / work</th>
|
||||
<th>Worktree binding</th>
|
||||
<th>Stale</th>
|
||||
<th>Contamination</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{body}
|
||||
</tbody>
|
||||
</table>"""
|
||||
|
||||
|
||||
def _namespaces_section(snapshot: SessionViewSnapshot) -> str:
|
||||
section = snapshot.inventory.section("namespaces")
|
||||
if section is None:
|
||||
return (
|
||||
'<div class="prompt-card"><h3>Namespaces</h3>'
|
||||
'<p class="muted">Namespaces section not loaded.</p></div>'
|
||||
)
|
||||
if not section.ok:
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Namespaces {_badge(section.status, "badge-health-degraded")}</h3>
|
||||
<p class="muted">{escape(section.reason or "unavailable")}</p>
|
||||
</div>"""
|
||||
|
||||
rows = []
|
||||
for item in section.items:
|
||||
caps = item.get("capability_summary") or {}
|
||||
cap_bits = ", ".join(
|
||||
name for name, ok in sorted(caps.items()) if ok
|
||||
) or "none"
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td><code>{escape(str(item.get('mcp_namespace') or '—'))}</code></td>"
|
||||
f"<td><code>{escape(str(item.get('profile_name') or '—'))}</code></td>"
|
||||
f"<td>{escape(str(item.get('role') or '—'))}</td>"
|
||||
f"<td>{escape(cap_bits)}</td>"
|
||||
f"<td>{'yes' if item.get('active') else 'no'}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
reason = (
|
||||
f'<p class="muted">{escape(section.reason)}</p>'
|
||||
if section.reason
|
||||
else ""
|
||||
)
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Namespaces / capabilities</h3>
|
||||
{reason}
|
||||
<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Namespace</th>
|
||||
<th>Profile</th>
|
||||
<th>Role</th>
|
||||
<th>Capabilities</th>
|
||||
<th>Active in process</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{"".join(rows) if rows else '<tr><td colspan="5" class="muted">No namespace rows.</td></tr>'}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>"""
|
||||
|
||||
|
||||
def _worktrees_section(snapshot: SessionViewSnapshot) -> str:
|
||||
section = snapshot.inventory.section("worktrees")
|
||||
if section is None:
|
||||
return ""
|
||||
if not section.ok and not section.items:
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Worktrees {_badge(section.status, "badge-health-degraded")}</h3>
|
||||
<p class="muted">{escape(section.reason or "unavailable")}</p>
|
||||
</div>"""
|
||||
|
||||
rows = []
|
||||
for item in section.items[:50]:
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td><code>{escape(str(item.get('rel_path') or item.get('path') or '—'))}</code></td>"
|
||||
f"<td><code>{escape(str(item.get('branch') or '—'))}</code></td>"
|
||||
f"<td>{escape(str(item.get('classification') or '—'))}</td>"
|
||||
f"<td>{'yes' if item.get('registered_worktree') else 'no'}</td>"
|
||||
f"<td>{'dirty' if item.get('dirty') else 'clean'}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
more = ""
|
||||
if len(section.items) > 50:
|
||||
more = f'<p class="muted">Showing 50 of {len(section.items)}. Full list: <a href="/worktrees">/worktrees</a>.</p>'
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Worktree bindings</h3>
|
||||
<p class="muted">Registered issue worktrees under <code>branches/</code>. Hygiene detail: <a href="/worktrees">/worktrees</a>.</p>
|
||||
<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Path</th>
|
||||
<th>Branch</th>
|
||||
<th>Classification</th>
|
||||
<th>Registered</th>
|
||||
<th>State</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{"".join(rows) if rows else '<tr><td colspan="5" class="muted">No worktrees recorded.</td></tr>'}
|
||||
</tbody>
|
||||
</table>
|
||||
{more}
|
||||
</div>"""
|
||||
|
||||
|
||||
def _contamination_section(markers: Sequence[ContaminationMarker]) -> str:
|
||||
if not markers:
|
||||
return (
|
||||
'<div class="prompt-card"><h3>Contamination markers</h3>'
|
||||
'<p class="muted">No contamination kinds inspected.</p></div>'
|
||||
)
|
||||
rows = []
|
||||
for marker in markers:
|
||||
active = marker.to_dict()["active"]
|
||||
status = "ACTIVE" if active else ("cleared" if marker.cleared else "absent")
|
||||
css = "badge-blocked" if active else "badge-health-ok"
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td><code>{escape(marker.kind)}</code></td>"
|
||||
f"<td>{_badge(status, css)}</td>"
|
||||
f"<td>{escape(marker.reason_class or '—')}</td>"
|
||||
f"<td><code>{escape(marker.session_id or '—')}</code></td>"
|
||||
f"<td>{escape(marker.command_summary or marker.summary)}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Contamination markers (#630 / #671)</h3>
|
||||
<p class="muted">Durable markers only — never silent when present. Clearance is reconciler-only.</p>
|
||||
<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Kind</th>
|
||||
<th>State</th>
|
||||
<th>Reason class</th>
|
||||
<th>Session</th>
|
||||
<th>Summary</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{"".join(rows)}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>"""
|
||||
|
||||
|
||||
def _recovery_section(snapshot: SessionViewSnapshot) -> str:
|
||||
items = []
|
||||
for doc in snapshot.recovery_docs:
|
||||
items.append(
|
||||
"<li>"
|
||||
f"<code>{escape(doc['path'])}</code> — "
|
||||
f"<strong>{escape(doc['label'])}</strong>: {escape(doc['note'])}"
|
||||
"</li>"
|
||||
)
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Sanctioned recovery (read-only)</h3>
|
||||
<p class="muted">This view does <strong>not</strong> restart, kill, or take over sessions.
|
||||
Manual <code>pkill</code> / <code>kill</code> of MCP daemons is contamination (#630), not recovery.</p>
|
||||
<ul class="reasons">
|
||||
{"".join(items)}
|
||||
<li>Prefer IDE/client reconnect (<code>/mcp reconnect</code>) or an operator-owned restart recorded in the restart inventory.</li>
|
||||
</ul>
|
||||
</div>"""
|
||||
|
||||
|
||||
def render_sessions_page(snapshot: SessionViewSnapshot) -> str:
|
||||
"""Render the full HTML body for the runtime/session view."""
|
||||
error_block = ""
|
||||
if snapshot.fetch_error:
|
||||
error_block = (
|
||||
f'<div class="health-card health-stale"><strong>Partial load:</strong> '
|
||||
f"{escape(snapshot.fetch_error)}</div>"
|
||||
)
|
||||
if snapshot.runtime.fetch_error:
|
||||
error_block += (
|
||||
f'<div class="health-card health-stale"><strong>Runtime note:</strong> '
|
||||
f"{escape(snapshot.runtime.fetch_error)}</div>"
|
||||
)
|
||||
|
||||
body = f"""
|
||||
{error_block}
|
||||
{_runtime_banner(snapshot)}
|
||||
{_summary_bar(snapshot)}
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>Sessions</h3>
|
||||
<p class="muted">Control-plane sessions correlated with leases and worktree bindings.
|
||||
Stale and contamination flags are fail-soft: absence of a marker is not proof of cleanliness when inventory is degraded.
|
||||
Lease and worktree columns read <em>unknown (inventory …)</em> when their source could not be loaded.</p>
|
||||
{_ownership_caveat(snapshot)}
|
||||
{_sessions_table(snapshot)}
|
||||
</div>
|
||||
|
||||
{_namespaces_section(snapshot)}
|
||||
{_worktrees_section(snapshot)}
|
||||
{_contamination_section(snapshot.contamination_markers)}
|
||||
{_recovery_section(snapshot)}
|
||||
"""
|
||||
return render_page(title="Sessions", body_html=f"""<h2>Runtime and sessions</h2>
|
||||
<p class="meta">Phase 1 read-only view (#641). Combines runtime health (#430) with
|
||||
unified inventory sessions/namespaces/worktrees (#636). No restart or session-takeover controls.</p>
|
||||
{body}""")
|
||||
@@ -278,8 +278,10 @@ def _recovery_card() -> str:
|
||||
"controls arrive in Phase 2 (#642); until then recovery runs through "
|
||||
"the sanctioned client reconnect / operator restart path.</p>"
|
||||
"<ul class='reasons'>"
|
||||
"<li><a href='/runtime'>Runtime and session view</a> — active profile, "
|
||||
"workflow hashes, and shell health.</li>"
|
||||
"<li><a href='/runtime'>Runtime health</a> — active profile, workflow "
|
||||
"hashes, and shell health.</li>"
|
||||
"<li><a href='/sessions'>Runtime and sessions</a> — namespaces, session "
|
||||
"rows, worktree bindings, and contamination markers (#641).</li>"
|
||||
"<li>Reconnect the MCP client from the IDE, then re-run the blocked "
|
||||
"cycle. Never kill the daemon process manually: unmanaged kills are "
|
||||
"recorded as runtime contamination (#630).</li>"
|
||||
|
||||
@@ -0,0 +1,449 @@
|
||||
"""Traffic-control view loader for Phase 1 operator web console (#640).
|
||||
|
||||
Combines queue snapshots, inventory leases, dependency graph classifications,
|
||||
and workflow dashboard rules to deliver full traffic-control visibility:
|
||||
runnable, leased (in-progress), blocked (dependency/lock), needs-controller,
|
||||
and terminal-complete candidates.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable, Sequence
|
||||
|
||||
from webui.project_registry import find_project, load_registry
|
||||
from webui.queue_loader import load_queue_snapshot, QueueSnapshot
|
||||
from webui.lease_loader import load_lease_snapshot, LeaseSnapshot
|
||||
from workflow_dashboard import (
|
||||
DashboardSnapshot,
|
||||
QueueEntry,
|
||||
RoleNextAction,
|
||||
build_workflow_dashboard,
|
||||
DASHBOARD_ROLES,
|
||||
)
|
||||
from allocator_service import WorkCandidate
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TrafficItem:
|
||||
kind: str # "issue" or "pr"
|
||||
number: int
|
||||
title: str
|
||||
traffic_state: str # "runnable", "leased", "blocked", "needs_controller", "terminal_complete"
|
||||
expected_role: str
|
||||
safe_for_roles: tuple[str, ...]
|
||||
badges: tuple[str, ...]
|
||||
block_reason: str | None = None
|
||||
lease_info: dict[str, Any] | None = None
|
||||
head_sha: str | None = None
|
||||
|
||||
@property
|
||||
def is_safe(self) -> bool:
|
||||
return self.block_reason is None and bool(self.safe_for_roles)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"kind": self.kind,
|
||||
"number": self.number,
|
||||
"title": self.title,
|
||||
"traffic_state": self.traffic_state,
|
||||
"expected_role": self.expected_role,
|
||||
"safe_for_roles": list(self.safe_for_roles),
|
||||
"badges": list(self.badges),
|
||||
"block_reason": self.block_reason,
|
||||
"lease_info": self.lease_info,
|
||||
"head_sha": self.head_sha,
|
||||
"is_safe": self.is_safe,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TrafficSnapshot:
|
||||
project_id: str
|
||||
repo_label: str
|
||||
runnable: tuple[TrafficItem, ...]
|
||||
leased: tuple[TrafficItem, ...]
|
||||
blocked: tuple[TrafficItem, ...]
|
||||
needs_controller: tuple[TrafficItem, ...]
|
||||
terminal_complete: tuple[TrafficItem, ...]
|
||||
next_roles: tuple[dict[str, Any], ...]
|
||||
fetch_error: str | None = None
|
||||
inventory_complete: bool = True
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"project_id": self.project_id,
|
||||
"repo_label": self.repo_label,
|
||||
"runnable": [i.as_dict() for i in self.runnable],
|
||||
"leased": [i.as_dict() for i in self.leased],
|
||||
"blocked": [i.as_dict() for i in self.blocked],
|
||||
"needs_controller": [i.as_dict() for i in self.needs_controller],
|
||||
"terminal_complete": [i.as_dict() for i in self.terminal_complete],
|
||||
"next_roles": list(self.next_roles),
|
||||
"fetch_error": self.fetch_error,
|
||||
"inventory_complete": self.inventory_complete,
|
||||
}
|
||||
|
||||
|
||||
def _classify_traffic_item(
|
||||
entry: QueueEntry,
|
||||
*,
|
||||
lease_info: dict[str, Any] | None = None,
|
||||
) -> TrafficItem:
|
||||
"""Classify a QueueEntry into a TrafficItem with explicit traffic state."""
|
||||
badges = list(entry.badges)
|
||||
block_reason = entry.block_reason
|
||||
expected_role = entry.expected_role
|
||||
|
||||
entry_is_safe = entry.block_reason is None and bool(entry.safe_for_roles)
|
||||
# Lease state is checked first: an item that is both leased and blocked is
|
||||
# reported as leased. That is safe by construction — a leased item is never
|
||||
# placed in the runnable lane — and it keeps the operator's attention on the
|
||||
# session that currently owns the work. The blocker text still renders.
|
||||
if lease_info is not None or "in-progress" in badges or "claimed" in badges:
|
||||
state = "leased"
|
||||
elif expected_role == "reconciler" or "terminal-lock" in badges:
|
||||
state = "terminal_complete"
|
||||
elif expected_role == "controller" or "contaminated" in badges or "needs-controller" in badges:
|
||||
state = "needs_controller"
|
||||
elif (
|
||||
block_reason is not None
|
||||
or "blocked" in badges
|
||||
or "dependency-unmet" in badges
|
||||
or "blocked-by-terminal" in badges
|
||||
or "status:blocked" in badges
|
||||
):
|
||||
state = "blocked"
|
||||
elif entry_is_safe:
|
||||
state = "runnable"
|
||||
else:
|
||||
state = "needs_controller"
|
||||
|
||||
return TrafficItem(
|
||||
kind=entry.kind,
|
||||
number=entry.number,
|
||||
title=entry.title,
|
||||
traffic_state=state,
|
||||
expected_role=expected_role,
|
||||
safe_for_roles=entry.safe_for_roles,
|
||||
badges=tuple(badges),
|
||||
block_reason=block_reason,
|
||||
lease_info=lease_info,
|
||||
head_sha=entry.head_sha,
|
||||
)
|
||||
|
||||
|
||||
# Claim statuses from ``issue_claim_heartbeat.build_claim_inventory`` that mean
|
||||
# a live worker currently holds the issue. Everything else (``stale``,
|
||||
# ``phantom``, ``reclaimable``, ``not_claimed``) is reported through the
|
||||
# dashboard's stale-lease channel and is never rendered as an active lease.
|
||||
_ACTIVE_CLAIM_STATUSES = frozenset({"active", "awaiting_review"})
|
||||
|
||||
# Statuses that positively mean "not an active lease" for any lease record.
|
||||
_INACTIVE_LEASE_STATUSES = frozenset(
|
||||
{"expired", "stale", "released", "moot", "reclaimable", "phantom", "not_claimed"}
|
||||
)
|
||||
|
||||
|
||||
def _candidates_from_queue_snapshot(q_snap: QueueSnapshot) -> list[WorkCandidate]:
|
||||
"""Build allocator candidates from the queue loader's authoritative signals.
|
||||
|
||||
Display badges (``blocked``/``claimed``/``duplicate``/``stale``/
|
||||
``in-review``/``open``) are rendering hints, not routing state, so nothing
|
||||
here branches on them. Every routing field comes from
|
||||
``QueueItem.signals`` — the raw Gitea payload values.
|
||||
|
||||
The queue loader reads ``/pulls`` and ``/issues`` only; it never fetches
|
||||
review verdicts. ``request_changes_current_head`` / ``approval_on_current_head``
|
||||
are therefore left at their fail-safe ``False`` rather than being guessed
|
||||
from badges: an unproven approval must never route a PR to the merger.
|
||||
"""
|
||||
candidates: list[WorkCandidate] = []
|
||||
|
||||
for pr in q_snap.prs:
|
||||
signals = pr.signals or {}
|
||||
head_sha = str(signals.get("head_sha") or "").strip()
|
||||
mergeable = signals.get("mergeable")
|
||||
labels = tuple(str(x) for x in (signals.get("labels") or ()))
|
||||
candidates.append(
|
||||
WorkCandidate(
|
||||
kind="pr",
|
||||
number=pr.number,
|
||||
state="open",
|
||||
labels=labels,
|
||||
title=pr.title,
|
||||
# Full 40-char SHA from head.sha — never the 12-char display value.
|
||||
head_sha=head_sha or None,
|
||||
priority=5,
|
||||
mergeable=mergeable is True,
|
||||
blocked=mergeable is False or "status:blocked" in labels,
|
||||
)
|
||||
)
|
||||
|
||||
for issue in q_snap.issues:
|
||||
signals = issue.signals or {}
|
||||
labels = tuple(str(x) for x in (signals.get("labels") or ()))
|
||||
lowered = {label.lower() for label in labels}
|
||||
candidates.append(
|
||||
WorkCandidate(
|
||||
kind="issue",
|
||||
number=issue.number,
|
||||
state="open",
|
||||
labels=labels,
|
||||
title=issue.title,
|
||||
priority=20 if "status:ready" in lowered else 10,
|
||||
blocked="status:blocked" in lowered,
|
||||
# A live claim by another session is not this session's work.
|
||||
already_claimed_elsewhere="status:in-progress" in lowered,
|
||||
)
|
||||
)
|
||||
|
||||
return candidates
|
||||
|
||||
|
||||
def _claim_lease_records(inventory: dict[str, Any] | None) -> list[dict[str, Any]]:
|
||||
"""Normalize ``build_claim_inventory`` entries into lease records.
|
||||
|
||||
The inventory contract is ``{"entries", "counts", "heartbeat_lease_minutes",
|
||||
"reclaim_after_minutes", "in_progress_total"}``. Each entry is keyed by
|
||||
``issue_number``; the subject kind is therefore always ``issue``.
|
||||
"""
|
||||
entries = (inventory or {}).get("entries") or ()
|
||||
records: list[dict[str, Any]] = []
|
||||
for entry in entries:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
number = entry.get("issue_number")
|
||||
if number is None:
|
||||
continue
|
||||
try:
|
||||
number_int = int(number)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
heartbeat = entry.get("latest_heartbeat") or {}
|
||||
record = dict(entry)
|
||||
record.update(
|
||||
{
|
||||
"kind": "issue",
|
||||
"number": number_int,
|
||||
"role": "author",
|
||||
"lease_source": "issue-claim-heartbeat",
|
||||
}
|
||||
)
|
||||
if isinstance(heartbeat, dict):
|
||||
if heartbeat.get("session_id") and not record.get("session_id"):
|
||||
record["session_id"] = heartbeat.get("session_id")
|
||||
if heartbeat.get("author") and not record.get("author"):
|
||||
record["author"] = heartbeat.get("author")
|
||||
records.append(record)
|
||||
return records
|
||||
|
||||
|
||||
def _lease_subject(lease: dict[str, Any]) -> tuple[str, int] | None:
|
||||
"""Return the ``(kind, number)`` a lease record actually covers.
|
||||
|
||||
Fails closed: a record that does not identify exactly one subject is
|
||||
dropped rather than attributed to a guessed work item (#640 — never invent
|
||||
a lease, and never attach a PR lease to a same-numbered issue).
|
||||
"""
|
||||
kind = str(lease.get("kind") or lease.get("work_kind") or "").strip().lower()
|
||||
pr_number = lease.get("pr_number")
|
||||
issue_number = lease.get("issue_number")
|
||||
|
||||
if kind not in ("pr", "issue"):
|
||||
if pr_number is not None and issue_number is None:
|
||||
kind = "pr"
|
||||
elif issue_number is not None and pr_number is None:
|
||||
kind = "issue"
|
||||
else:
|
||||
return None
|
||||
|
||||
number = lease.get("number")
|
||||
if number is None:
|
||||
number = lease.get("work_number")
|
||||
if number is None:
|
||||
number = pr_number if kind == "pr" else issue_number
|
||||
if number is None:
|
||||
return None
|
||||
try:
|
||||
return kind, int(number)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _is_active_lease(lease: dict[str, Any]) -> bool:
|
||||
"""True when the record proves a worker currently holds the item."""
|
||||
if lease.get("stale") or lease.get("expired"):
|
||||
return False
|
||||
status = str(lease.get("status") or lease.get("lease_status") or "").strip().lower()
|
||||
if status in _INACTIVE_LEASE_STATUSES:
|
||||
return False
|
||||
if lease.get("lease_source") == "issue-claim-heartbeat":
|
||||
return status in _ACTIVE_CLAIM_STATUSES
|
||||
return True
|
||||
|
||||
|
||||
def load_traffic_snapshot(
|
||||
*,
|
||||
candidates: Sequence[WorkCandidate] | None = None,
|
||||
leases: Sequence[dict[str, Any]] | None = None,
|
||||
terminal_pr: int | None = None,
|
||||
fetch_queue_snapshot: Callable[[], QueueSnapshot] | None = None,
|
||||
fetch_lease_snapshot: Callable[[], LeaseSnapshot] | None = None,
|
||||
project_id: str = "gitea-tools",
|
||||
) -> TrafficSnapshot:
|
||||
"""Load and compute the traffic-control snapshot."""
|
||||
try:
|
||||
reg = load_registry()
|
||||
proj = find_project(reg, project_id)
|
||||
repo_label = proj.remote_repo if proj else "Scaled-Tech-Consulting/Gitea-Tools"
|
||||
except Exception:
|
||||
repo_label = "Scaled-Tech-Consulting/Gitea-Tools"
|
||||
|
||||
# Injected candidates path (pure unit testing)
|
||||
if candidates is not None:
|
||||
dashboard = build_workflow_dashboard(
|
||||
candidates=candidates,
|
||||
leases=leases,
|
||||
terminal_pr=terminal_pr,
|
||||
inventory_complete=True,
|
||||
)
|
||||
return _build_traffic_snapshot_from_dashboard(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
dashboard=dashboard,
|
||||
leases=leases or (),
|
||||
)
|
||||
|
||||
# Live snapshot loading
|
||||
q_loader = fetch_queue_snapshot or load_queue_snapshot
|
||||
l_loader = fetch_lease_snapshot or load_lease_snapshot
|
||||
|
||||
try:
|
||||
q_snap = q_loader()
|
||||
l_snap = l_loader()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return TrafficSnapshot(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
runnable=(),
|
||||
leased=(),
|
||||
blocked=(),
|
||||
needs_controller=(),
|
||||
terminal_complete=(),
|
||||
next_roles=(),
|
||||
fetch_error=f"Failed to load traffic state: {exc}",
|
||||
inventory_complete=False,
|
||||
)
|
||||
|
||||
if q_snap.fetch_error or l_snap.fetch_error:
|
||||
err = q_snap.fetch_error or l_snap.fetch_error
|
||||
return TrafficSnapshot(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
runnable=(),
|
||||
leased=(),
|
||||
blocked=(),
|
||||
needs_controller=(),
|
||||
terminal_complete=(),
|
||||
next_roles=(),
|
||||
fetch_error=err,
|
||||
inventory_complete=False,
|
||||
)
|
||||
|
||||
candidate_list = _candidates_from_queue_snapshot(q_snap)
|
||||
|
||||
raw_leases: list[dict[str, Any]] = _claim_lease_records(l_snap.claim_inventory)
|
||||
for r_lease in l_snap.reviewer_leases or ():
|
||||
if not isinstance(r_lease, dict):
|
||||
continue
|
||||
# Always pin reviewer leases to the PR subject, even if a linked
|
||||
# issue_number is present on the marker (#640 B2).
|
||||
normalized = dict(r_lease)
|
||||
subject = normalized.get("pr_number") or normalized.get("number")
|
||||
if subject is None:
|
||||
continue
|
||||
try:
|
||||
pr_num = int(subject)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
normalized["kind"] = "pr"
|
||||
normalized["number"] = pr_num
|
||||
normalized["pr_number"] = pr_num
|
||||
normalized.setdefault("role", "reviewer")
|
||||
raw_leases.append(normalized)
|
||||
|
||||
dashboard = build_workflow_dashboard(
|
||||
candidates=candidate_list,
|
||||
leases=raw_leases,
|
||||
inventory_complete=q_snap.pr_pagination.inventory_complete if q_snap.pr_pagination else True,
|
||||
)
|
||||
|
||||
return _build_traffic_snapshot_from_dashboard(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
dashboard=dashboard,
|
||||
leases=raw_leases,
|
||||
)
|
||||
|
||||
|
||||
def _build_traffic_snapshot_from_dashboard(
|
||||
*,
|
||||
project_id: str,
|
||||
repo_label: str,
|
||||
dashboard: DashboardSnapshot,
|
||||
leases: Sequence[dict[str, Any]],
|
||||
) -> TrafficSnapshot:
|
||||
"""Classify dashboard entries into the 5 traffic state buckets."""
|
||||
all_entries = dashboard.open_prs + dashboard.open_issues
|
||||
|
||||
# Map each active lease onto the exact work item it covers. Records whose
|
||||
# subject cannot be determined, and claims that are stale/phantom/
|
||||
# reclaimable, are deliberately dropped instead of guessed.
|
||||
lease_map: dict[tuple[str, int], dict[str, Any]] = {}
|
||||
for lease in leases:
|
||||
if not isinstance(lease, dict) or not _is_active_lease(lease):
|
||||
continue
|
||||
subject = _lease_subject(lease)
|
||||
if subject is not None:
|
||||
lease_map[subject] = lease
|
||||
|
||||
runnable: list[TrafficItem] = []
|
||||
leased: list[TrafficItem] = []
|
||||
blocked: list[TrafficItem] = []
|
||||
needs_controller: list[TrafficItem] = []
|
||||
terminal_complete: list[TrafficItem] = []
|
||||
|
||||
for entry in all_entries:
|
||||
l_info = lease_map.get((entry.kind, entry.number))
|
||||
item = _classify_traffic_item(entry, lease_info=l_info)
|
||||
|
||||
if item.traffic_state == "leased":
|
||||
leased.append(item)
|
||||
elif item.traffic_state == "terminal_complete":
|
||||
terminal_complete.append(item)
|
||||
elif item.traffic_state == "blocked":
|
||||
blocked.append(item)
|
||||
elif item.traffic_state == "needs_controller":
|
||||
needs_controller.append(item)
|
||||
else:
|
||||
runnable.append(item)
|
||||
|
||||
next_roles = [dashboard.next_safe_by_role[r].as_dict() for r in DASHBOARD_ROLES if r in dashboard.next_safe_by_role]
|
||||
|
||||
return TrafficSnapshot(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
runnable=tuple(runnable),
|
||||
leased=tuple(leased),
|
||||
blocked=tuple(blocked),
|
||||
needs_controller=tuple(needs_controller),
|
||||
terminal_complete=tuple(terminal_complete),
|
||||
next_roles=tuple(next_roles),
|
||||
fetch_error=None,
|
||||
inventory_complete=dashboard.inventory_complete,
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: TrafficSnapshot) -> dict[str, Any]:
|
||||
return snapshot.as_dict()
|
||||
@@ -0,0 +1,170 @@
|
||||
"""HTML rendering for Phase 1 Traffic-Control View (#640)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from html import escape
|
||||
from typing import Sequence
|
||||
|
||||
from webui.layout import render_page
|
||||
from webui.traffic_loader import TrafficItem, TrafficSnapshot
|
||||
|
||||
|
||||
def _render_badges(badges: Sequence[str]) -> str:
|
||||
if not badges:
|
||||
return ""
|
||||
out = []
|
||||
for b in badges:
|
||||
cls = "badge"
|
||||
b_lower = b.lower()
|
||||
if "blocked" in b_lower or "unmet" in b_lower:
|
||||
cls += " badge-blocked"
|
||||
elif "claimed" in b_lower or "in-progress" in b_lower or "leased" in b_lower:
|
||||
cls += " badge-claimed"
|
||||
elif "review" in b_lower or "ready" in b_lower:
|
||||
cls += " badge-in-review"
|
||||
elif "duplicate" in b_lower:
|
||||
cls += " badge-duplicate"
|
||||
elif "stale" in b_lower:
|
||||
cls += " badge-stale"
|
||||
out.append(f'<span class="{cls}">{escape(b)}</span>')
|
||||
return f'<div class="badges">{"".join(out)}</div>'
|
||||
|
||||
|
||||
def _render_traffic_item_row(item: TrafficItem) -> str:
|
||||
kind_label = escape(item.kind.upper())
|
||||
num_str = f"#{item.number}"
|
||||
title_str = escape(item.title)
|
||||
role_str = escape(item.expected_role)
|
||||
badges_html = _render_badges(item.badges)
|
||||
|
||||
reason_html = ""
|
||||
if item.block_reason:
|
||||
reason_html = f'<div class="muted" style="font-size:0.82rem; margin-top:0.2rem;"><strong>Blocker:</strong> {escape(item.block_reason)}</div>'
|
||||
|
||||
lease_html = ""
|
||||
if item.lease_info:
|
||||
owner = escape(str(item.lease_info.get("session_id") or item.lease_info.get("reviewer_identity") or "active worker"))
|
||||
lease_html = f'<div class="muted" style="font-size:0.82rem; margin-top:0.2rem;"><strong>Lease:</strong> {owner}</div>'
|
||||
|
||||
return f"""<tr>
|
||||
<td><code>{kind_label} {num_str}</code></td>
|
||||
<td>
|
||||
<div><strong>{title_str}</strong> {badges_html}</div>
|
||||
{reason_html}
|
||||
{lease_html}
|
||||
</td>
|
||||
<td><code>{role_str}</code></td>
|
||||
</tr>"""
|
||||
|
||||
|
||||
def _render_traffic_table(items: Sequence[TrafficItem], empty_message: str) -> str:
|
||||
if not items:
|
||||
return f'<p class="muted">{escape(empty_message)}</p>'
|
||||
|
||||
rows = "".join(_render_traffic_item_row(item) for item in items)
|
||||
return f"""<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th style="width: 15%;">Item</th>
|
||||
<th style="width: 65%;">Title & Details</th>
|
||||
<th style="width: 20%;">Next Role</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{rows}
|
||||
</tbody>
|
||||
</table>"""
|
||||
|
||||
|
||||
def _render_next_roles(next_roles: Sequence[dict]) -> str:
|
||||
if not next_roles:
|
||||
return ""
|
||||
|
||||
cards = []
|
||||
for r in next_roles:
|
||||
role = escape(r.get("role", "unknown"))
|
||||
status = r.get("status", "idle")
|
||||
prompt = escape(r.get("prompt", ""))
|
||||
|
||||
status_cls = "badge-health-ok" if status == "safe" else ("badge-blocked" if "blocked" in status else "badge-health-skipped")
|
||||
cards.append(f"""<div class="health-card" style="margin-bottom:0.75rem;">
|
||||
<div style="display:flex; justify-content:space-between; align-items:center;">
|
||||
<h3>Role: <code>{role}</code></h3>
|
||||
<span class="badge {status_cls}">status: {escape(status)}</span>
|
||||
</div>
|
||||
<p class="meta" style="margin:0.35rem 0 0;">{prompt}</p>
|
||||
</div>""")
|
||||
|
||||
return f"""<div style="margin: 1.5rem 0;">
|
||||
<h3>Next Safe Role Actions</h3>
|
||||
{"".join(cards)}
|
||||
</div>"""
|
||||
|
||||
|
||||
def render_traffic_page(snapshot: TrafficSnapshot) -> str:
|
||||
"""Render the full HTML view for workflow traffic control."""
|
||||
if snapshot.fetch_error:
|
||||
body = f"""<h2>Workflow Traffic Control</h2>
|
||||
<p class="meta">Repository: <code>{escape(snapshot.repo_label)}</code></p>
|
||||
<div class="health-card health-stale">
|
||||
<h3>Traffic data unavailable</h3>
|
||||
<p class="health-headline">{escape(snapshot.fetch_error)}</p>
|
||||
<p class="muted">Fail closed: traffic state cannot be established cleanly. Check credentials or remote connectivity.</p>
|
||||
</div>"""
|
||||
return render_page(title="Traffic Control", body_html=body)
|
||||
|
||||
runnable_count = len(snapshot.runnable)
|
||||
leased_count = len(snapshot.leased)
|
||||
blocked_count = len(snapshot.blocked)
|
||||
controller_count = len(snapshot.needs_controller)
|
||||
terminal_count = len(snapshot.terminal_complete)
|
||||
|
||||
summary_bar = f"""<div class="health-card" style="display:flex; flex-wrap:wrap; gap:1rem; align-items:center;">
|
||||
<div><strong>Runnable:</strong> <span class="badge badge-health-ok">{runnable_count}</span></div>
|
||||
<div><strong>Leased:</strong> <span class="badge badge-claimed">{leased_count}</span></div>
|
||||
<div><strong>Blocked:</strong> <span class="badge badge-blocked">{blocked_count}</span></div>
|
||||
<div><strong>Needs Controller:</strong> <span class="badge badge-duplicate">{controller_count}</span></div>
|
||||
<div><strong>Terminal Complete:</strong> <span class="badge badge-stale">{terminal_count}</span></div>
|
||||
</div>"""
|
||||
|
||||
next_roles_html = _render_next_roles(snapshot.next_roles)
|
||||
|
||||
sections_html = f"""
|
||||
<div class="prompt-card">
|
||||
<h3>1. Runnable Lanes (Ready for Allocation)</h3>
|
||||
<p class="muted">Safe work items with no unmet dependencies or active leases. Safe for allocation.</p>
|
||||
{_render_traffic_table(snapshot.runnable, "No runnable items ready for allocation.")}
|
||||
</div>
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>2. In-Progress Work (Active Leases)</h3>
|
||||
<p class="muted">Work items currently leased and actively being worked by an assigned role session.</p>
|
||||
{_render_traffic_table(snapshot.leased, "No active leases in flight.")}
|
||||
</div>
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>3. Blocked Items (Dependencies / Locks)</h3>
|
||||
<p class="muted">Items blocked by unmet dependency issues, a missing head pin, a merge conflict, or an active terminal review lock. Items labelled status:blocked route to section 4. Never presented as safe.</p>
|
||||
{_render_traffic_table(snapshot.blocked, "No blocked items.")}
|
||||
</div>
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>4. Needs Controller Intervention</h3>
|
||||
<p class="muted">Items requiring controller routing, diagnosis, or cross-role assignment.</p>
|
||||
{_render_traffic_table(snapshot.needs_controller, "No items requiring controller intervention.")}
|
||||
</div>
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>5. Terminal / Complete Candidates</h3>
|
||||
<p class="muted">Items ready for terminal reconciliation or post-merge worktree cleanup.</p>
|
||||
{_render_traffic_table(snapshot.terminal_complete, "No terminal complete candidates.")}
|
||||
</div>
|
||||
"""
|
||||
|
||||
body = f"""<h2>Workflow Traffic Control</h2>
|
||||
<p class="meta">Repository: <code>{escape(snapshot.repo_label)}</code></p>
|
||||
{summary_bar}
|
||||
{next_roles_html}
|
||||
{sections_html}"""
|
||||
|
||||
return render_page(title="Traffic Control", body_html=body)
|
||||
Reference in New Issue
Block a user