Compare commits
7
Commits
68220678d4
...
077cedc0c9
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
077cedc0c9 | ||
|
|
c83c18e52a | ||
|
|
1666e55e60 | ||
|
|
e9fdb356dd | ||
|
|
94b022b24d | ||
|
|
9fbe556466 | ||
|
|
46703eac3f |
+321
-1
@@ -1052,6 +1052,125 @@ def _prs_from_list_response(list_prs_response: list | dict | None) -> list | Non
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
# ── Workspace-vs-worktree mutation consistency (Issue #313) ──────────────────
|
||||||
|
#
|
||||||
|
# A worktree reset/checkout/clean is a workspace mutation even when the
|
||||||
|
# reviewer edited no files by hand. Final reports must therefore never say
|
||||||
|
# "Workspace mutations: none" once a worktree mutation occurred, and every
|
||||||
|
# observed worktree / git ref mutation must appear under its precise
|
||||||
|
# category field.
|
||||||
|
|
||||||
|
WORKTREE_MUTATION_MARKERS = (
|
||||||
|
"reset",
|
||||||
|
"checkout",
|
||||||
|
"switch",
|
||||||
|
"restore",
|
||||||
|
"clean",
|
||||||
|
"stash",
|
||||||
|
"merge",
|
||||||
|
"rebase",
|
||||||
|
"cherry-pick",
|
||||||
|
"worktree add",
|
||||||
|
"worktree remove",
|
||||||
|
"worktree prune",
|
||||||
|
)
|
||||||
|
|
||||||
|
GIT_REF_MUTATION_MARKERS = (
|
||||||
|
"fetch",
|
||||||
|
"pull",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _report_labeled_fields(report_text):
|
||||||
|
"""Return {lowercase label: value} for every 'Label: value' line."""
|
||||||
|
fields = {}
|
||||||
|
for line in (report_text or "").splitlines():
|
||||||
|
stripped = line.strip().lstrip("-*").strip()
|
||||||
|
if ":" in stripped:
|
||||||
|
k, v = stripped.split(":", 1)
|
||||||
|
fields[k.strip().lower()] = v.strip()
|
||||||
|
return fields
|
||||||
|
|
||||||
|
|
||||||
|
def _field_claims_none(fields, label):
|
||||||
|
"""True when *label* is present and its value is empty or 'none...'."""
|
||||||
|
value = fields.get(label)
|
||||||
|
if value is None:
|
||||||
|
return False
|
||||||
|
value = value.strip().lower()
|
||||||
|
return not value or value.startswith("none")
|
||||||
|
|
||||||
|
|
||||||
|
def _classify_git_commands(observed_commands):
|
||||||
|
"""Split observed git commands into worktree and ref mutations."""
|
||||||
|
worktree_cmds = []
|
||||||
|
ref_cmds = []
|
||||||
|
for cmd in observed_commands or []:
|
||||||
|
lowered = " ".join((cmd or "").lower().split())
|
||||||
|
if "git" not in lowered:
|
||||||
|
continue
|
||||||
|
if any(marker in lowered for marker in WORKTREE_MUTATION_MARKERS):
|
||||||
|
worktree_cmds.append(cmd)
|
||||||
|
elif any(marker in lowered for marker in GIT_REF_MUTATION_MARKERS):
|
||||||
|
ref_cmds.append(cmd)
|
||||||
|
return worktree_cmds, ref_cmds
|
||||||
|
|
||||||
|
|
||||||
|
def assess_workspace_mutation_consistency(report_text, observed_commands=None):
|
||||||
|
"""#313: reject 'Workspace mutations: none' after worktree mutations.
|
||||||
|
|
||||||
|
*observed_commands* is the optional list of git commands the workflow
|
||||||
|
actually ran. Even without it, a report that itself lists a non-none
|
||||||
|
``Worktree mutations`` value while claiming ``Workspace mutations:
|
||||||
|
none`` is internally contradictory and fails.
|
||||||
|
|
||||||
|
Returns {'complete', 'downgraded', 'reasons'}; fails closed on any
|
||||||
|
contradiction or unreported observed mutation.
|
||||||
|
"""
|
||||||
|
fields = _report_labeled_fields(report_text)
|
||||||
|
worktree_cmds, ref_cmds = _classify_git_commands(observed_commands)
|
||||||
|
|
||||||
|
reported_worktree = fields.get("worktree mutations")
|
||||||
|
reported_worktree_nonnone = bool(
|
||||||
|
reported_worktree and not reported_worktree.strip().lower().startswith("none")
|
||||||
|
)
|
||||||
|
|
||||||
|
reasons = []
|
||||||
|
|
||||||
|
workspace_none = _field_claims_none(fields, "workspace mutations")
|
||||||
|
if workspace_none and (worktree_cmds or reported_worktree_nonnone):
|
||||||
|
detail = worktree_cmds[0] if worktree_cmds else reported_worktree
|
||||||
|
reasons.append(
|
||||||
|
"report claims 'Workspace mutations: none' but worktree "
|
||||||
|
f"mutations occurred ({detail}); use precise categories such as "
|
||||||
|
"'File edits by reviewer: none' plus 'Worktree mutations: <cmd>'"
|
||||||
|
)
|
||||||
|
|
||||||
|
if worktree_cmds and (
|
||||||
|
"worktree mutations" not in fields
|
||||||
|
or _field_claims_none(fields, "worktree mutations")
|
||||||
|
):
|
||||||
|
reasons.append(
|
||||||
|
"observed worktree mutation not reported under 'Worktree "
|
||||||
|
f"mutations': {worktree_cmds[0]}"
|
||||||
|
)
|
||||||
|
|
||||||
|
if ref_cmds and (
|
||||||
|
"git ref mutations" not in fields
|
||||||
|
or _field_claims_none(fields, "git ref mutations")
|
||||||
|
):
|
||||||
|
reasons.append(
|
||||||
|
"observed git ref mutation not reported under 'Git ref "
|
||||||
|
f"mutations': {ref_cmds[0]}"
|
||||||
|
)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"complete": not reasons,
|
||||||
|
"downgraded": bool(reasons),
|
||||||
|
"reasons": reasons,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def assess_role_boundary(proof=None, *, task_role=None, namespaces_used=None,
|
def assess_role_boundary(proof=None, *, task_role=None, namespaces_used=None,
|
||||||
justification=None):
|
justification=None):
|
||||||
"""Assess reviewer/author role separation for blind queue workflows.
|
"""Assess reviewer/author role separation for blind queue workflows.
|
||||||
@@ -1232,6 +1351,7 @@ def build_final_report(checkout_proof, inventory, validation, contamination,
|
|||||||
report_text=None, review_decision_lock=None,
|
report_text=None, review_decision_lock=None,
|
||||||
controller_handoff=None, capability_proof=None,
|
controller_handoff=None, capability_proof=None,
|
||||||
sweep_proof=None, worktree_proof=None,
|
sweep_proof=None, worktree_proof=None,
|
||||||
|
queue_ordering=None,
|
||||||
baseline_validation=None, project_root=None):
|
baseline_validation=None, project_root=None):
|
||||||
"""Required behavior 6 + acceptance criteria: one report, distinct proofs.
|
"""Required behavior 6 + acceptance criteria: one report, distinct proofs.
|
||||||
|
|
||||||
@@ -1259,6 +1379,22 @@ def build_final_report(checkout_proof, inventory, validation, contamination,
|
|||||||
)
|
)
|
||||||
|
|
||||||
empty_queue_report = assess_empty_queue_report(report_text)
|
empty_queue_report = assess_empty_queue_report(report_text)
|
||||||
|
if queue_ordering is None and report_text:
|
||||||
|
queue_ordering_proof = assess_queue_ordering_proof(
|
||||||
|
report_text=report_text,
|
||||||
|
)
|
||||||
|
elif queue_ordering is not None:
|
||||||
|
queue_ordering_proof = assess_queue_ordering_proof(
|
||||||
|
report_text=report_text,
|
||||||
|
queue_ordering=queue_ordering,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
queue_ordering_proof = {
|
||||||
|
"proven": True,
|
||||||
|
"block": False,
|
||||||
|
"reasons": [],
|
||||||
|
"violations": [],
|
||||||
|
}
|
||||||
queue_status_report = (
|
queue_status_report = (
|
||||||
assess_queue_status_report(report_text)
|
assess_queue_status_report(report_text)
|
||||||
if report_text and _QUEUE_STATUS_REPORT_HINT.search(report_text)
|
if report_text and _QUEUE_STATUS_REPORT_HINT.search(report_text)
|
||||||
@@ -1410,6 +1546,11 @@ def build_final_report(checkout_proof, inventory, validation, contamination,
|
|||||||
"empty-queue report missing or failed trust-gate proof (#198)"
|
"empty-queue report missing or failed trust-gate proof (#198)"
|
||||||
)
|
)
|
||||||
downgrade_reasons.extend(empty_queue_report.get("reasons", []))
|
downgrade_reasons.extend(empty_queue_report.get("reasons", []))
|
||||||
|
if not queue_ordering_proof.get("proven"):
|
||||||
|
downgrade_reasons.append(
|
||||||
|
"queue ordering proof missing or failed (#321)"
|
||||||
|
)
|
||||||
|
downgrade_reasons.extend(queue_ordering_proof.get("reasons", []))
|
||||||
if not proof_wording.get("proven"):
|
if not proof_wording.get("proven"):
|
||||||
downgrade_reasons.append(
|
downgrade_reasons.append(
|
||||||
"unsupported proof wording in final report (#330)"
|
"unsupported proof wording in final report (#330)"
|
||||||
@@ -1436,10 +1577,12 @@ def build_final_report(checkout_proof, inventory, validation, contamination,
|
|||||||
# #179: no merge without a proven final live-state recheck.
|
# #179: no merge without a proven final live-state recheck.
|
||||||
and live_state_proven
|
and live_state_proven
|
||||||
and worktree_proven
|
and worktree_proven
|
||||||
|
and queue_ordering_proof.get("proven")
|
||||||
and baseline_validation.get("proven")
|
and baseline_validation.get("proven")
|
||||||
)
|
)
|
||||||
|
|
||||||
violations = []
|
violations = []
|
||||||
|
violations.extend(queue_ordering_proof.get("violations", []))
|
||||||
violations.extend(baseline_validation.get("violations", []))
|
violations.extend(baseline_validation.get("violations", []))
|
||||||
if merge_performed and not merge_allowed:
|
if merge_performed and not merge_allowed:
|
||||||
violations.append(
|
violations.append(
|
||||||
@@ -1492,6 +1635,11 @@ def build_final_report(checkout_proof, inventory, validation, contamination,
|
|||||||
else True
|
else True
|
||||||
),
|
),
|
||||||
"empty_queue_trust_gate_status": empty_queue_report.get("status"),
|
"empty_queue_trust_gate_status": empty_queue_report.get("status"),
|
||||||
|
"queue_ordering_proven": bool(queue_ordering_proof.get("proven")),
|
||||||
|
"queue_ordering_policy": queue_ordering_proof.get("policy"),
|
||||||
|
"queue_ordering_violations": list(
|
||||||
|
queue_ordering_proof.get("violations") or []
|
||||||
|
),
|
||||||
"proof_wording_proven": bool(proof_wording.get("proven")),
|
"proof_wording_proven": bool(proof_wording.get("proven")),
|
||||||
"proof_wording_violations": list(proof_wording.get("violations") or []),
|
"proof_wording_violations": list(proof_wording.get("violations") or []),
|
||||||
"queue_status_report_proven": bool(queue_status_report.get("proven")),
|
"queue_status_report_proven": bool(queue_status_report.get("proven")),
|
||||||
@@ -3070,6 +3218,162 @@ def build_review_mutation_proof(run_log: list[dict]) -> dict:
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Reviewer queue ordering proof (#321)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
_QUEUE_SELECTION_CLAIM_RE = re.compile(
|
||||||
|
r"\b(?:next|oldest)\s+eligible\s+pr\b",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
_ORDERING_POLICY_LINE_RE = re.compile(
|
||||||
|
r"(?:queue\s+ordering\s+policy|selection\s+policy|ordering\s+policy)"
|
||||||
|
r"\s*:\s*(.+)",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
_SKIPPED_PR_LINE_RE = re.compile(
|
||||||
|
r"\bskipped\s+pr\s*#?(\d+)\b",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_ordering_policy(report_text: str) -> str:
|
||||||
|
for line in (report_text or "").splitlines():
|
||||||
|
match = _ORDERING_POLICY_LINE_RE.search(line)
|
||||||
|
if match:
|
||||||
|
return match.group(1).strip().rstrip(".")
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def _skipped_pr_numbers(report_text: str) -> set[int]:
|
||||||
|
skipped: set[int] = set()
|
||||||
|
for match in _SKIPPED_PR_LINE_RE.finditer(report_text or ""):
|
||||||
|
try:
|
||||||
|
skipped.add(int(match.group(1)))
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
return skipped
|
||||||
|
|
||||||
|
|
||||||
|
def assess_queue_ordering_proof(
|
||||||
|
*,
|
||||||
|
report_text: str | None = None,
|
||||||
|
queue_ordering: dict | None = None,
|
||||||
|
) -> dict:
|
||||||
|
"""#321: reviewer queue selection must prove ordering before claiming next PR."""
|
||||||
|
text = report_text or ""
|
||||||
|
lower = text.lower()
|
||||||
|
proof = queue_ordering or {}
|
||||||
|
reasons: list[str] = []
|
||||||
|
violations: list[str] = []
|
||||||
|
|
||||||
|
claims_selection = bool(_QUEUE_SELECTION_CLAIM_RE.search(text))
|
||||||
|
selected = proof.get("selected_pr_number")
|
||||||
|
if selected is None and "selected pr #" in lower:
|
||||||
|
for token in lower.split():
|
||||||
|
if token.startswith("#") and token[1:].isdigit():
|
||||||
|
selected = int(token[1:])
|
||||||
|
break
|
||||||
|
|
||||||
|
if not claims_selection and selected is None:
|
||||||
|
return {
|
||||||
|
"proven": True,
|
||||||
|
"block": False,
|
||||||
|
"claims_selection": False,
|
||||||
|
"policy": None,
|
||||||
|
"reasons": [],
|
||||||
|
"violations": [],
|
||||||
|
"safe_next_action": "proceed",
|
||||||
|
}
|
||||||
|
|
||||||
|
policy = (proof.get("policy") or _parse_ordering_policy(text)).strip()
|
||||||
|
if not policy:
|
||||||
|
reasons.append(
|
||||||
|
"queue selection claimed without stated ordering policy (#321)"
|
||||||
|
)
|
||||||
|
|
||||||
|
policy_lower = policy.lower()
|
||||||
|
api_numbers: list[int] = []
|
||||||
|
for n in proof.get("api_pr_numbers") or []:
|
||||||
|
if isinstance(n, int):
|
||||||
|
api_numbers.append(n)
|
||||||
|
elif isinstance(n, str) and n.isdigit():
|
||||||
|
api_numbers.append(int(n))
|
||||||
|
skipped_entries = list(proof.get("skipped_earlier") or [])
|
||||||
|
skipped_numbers = {
|
||||||
|
int(entry["pr_number"])
|
||||||
|
for entry in skipped_entries
|
||||||
|
if isinstance(entry, dict) and entry.get("pr_number") is not None
|
||||||
|
}
|
||||||
|
skipped_numbers.update(_skipped_pr_numbers(text))
|
||||||
|
|
||||||
|
if selected is not None and api_numbers:
|
||||||
|
if "oldest" in policy_lower and "number" in policy_lower:
|
||||||
|
earlier_actionable = [
|
||||||
|
n for n in api_numbers
|
||||||
|
if n < int(selected) and n not in skipped_numbers
|
||||||
|
]
|
||||||
|
for pr_number in earlier_actionable:
|
||||||
|
entry = next(
|
||||||
|
(
|
||||||
|
e for e in skipped_entries
|
||||||
|
if isinstance(e, dict)
|
||||||
|
and int(e.get("pr_number", -1)) == pr_number
|
||||||
|
),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
reason = (entry or {}).get("reason", "").strip()
|
||||||
|
if not reason and f"skipped pr #{pr_number}" not in lower:
|
||||||
|
violations.append(
|
||||||
|
f"earlier actionable PR #{pr_number} skipped without "
|
||||||
|
"proof-backed reason (#321)"
|
||||||
|
)
|
||||||
|
reasons.append(
|
||||||
|
f"oldest-first policy requires skip reasoning for "
|
||||||
|
f"earlier PR #{pr_number} (#321)"
|
||||||
|
)
|
||||||
|
if api_numbers and api_numbers[0] != min(api_numbers):
|
||||||
|
if (
|
||||||
|
"api order" not in lower
|
||||||
|
and "sorted" not in lower
|
||||||
|
and "newest-first" not in lower
|
||||||
|
and not skipped_entries
|
||||||
|
):
|
||||||
|
reasons.append(
|
||||||
|
"API response order differs from oldest-first "
|
||||||
|
"selection but report does not prove explicit sort "
|
||||||
|
"or skip reasoning (#321)"
|
||||||
|
)
|
||||||
|
|
||||||
|
if "priority" in policy_lower:
|
||||||
|
override = (proof.get("priority_override") or "").strip()
|
||||||
|
if not override and "priority" not in lower:
|
||||||
|
reasons.append(
|
||||||
|
"priority-label ordering policy requires override "
|
||||||
|
"explanation in report (#321)"
|
||||||
|
)
|
||||||
|
|
||||||
|
proven = not reasons and not violations
|
||||||
|
return {
|
||||||
|
"proven": proven,
|
||||||
|
"block": bool(violations),
|
||||||
|
"claims_selection": True,
|
||||||
|
"policy": policy or None,
|
||||||
|
"selected_pr_number": selected,
|
||||||
|
"reasons": reasons,
|
||||||
|
"violations": violations,
|
||||||
|
"safe_next_action": (
|
||||||
|
"state queue ordering policy and proof-backed skip reasoning "
|
||||||
|
"for every earlier actionable PR"
|
||||||
|
if not proven
|
||||||
|
else "proceed"
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Reviewer baseline validation (#325)
|
# Reviewer baseline validation (#325)
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -3384,6 +3688,17 @@ _PROOF_WORDING_RULES: tuple[dict, ...] = (
|
|||||||
"text_evidence": _PAGINATION_FINALITY_EVIDENCE,
|
"text_evidence": _PAGINATION_FINALITY_EVIDENCE,
|
||||||
"allow_prior_label": False,
|
"allow_prior_label": False,
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"id": "same_as_master",
|
||||||
|
"pattern": re.compile(r"\bsame as master\b", re.I),
|
||||||
|
"session_keys": ("baseline_worktree_proof", "clean_baseline_proof"),
|
||||||
|
"text_evidence": re.compile(
|
||||||
|
r"baseline worktree|clean baseline|diff base.*master|"
|
||||||
|
r"base[- ]equivalent.*master|worktree proof",
|
||||||
|
re.I,
|
||||||
|
),
|
||||||
|
"allow_prior_label": False,
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"id": "no_file_edits",
|
"id": "no_file_edits",
|
||||||
"pattern": re.compile(
|
"pattern": re.compile(
|
||||||
@@ -3449,7 +3764,12 @@ def assess_proof_wording(
|
|||||||
*,
|
*,
|
||||||
session_evidence: dict | None = None,
|
session_evidence: dict | None = None,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Issue #330: reject unsupported proof phrases in final reports."""
|
"""Issue #330: reject unsupported proof phrases in final reports.
|
||||||
|
|
||||||
|
Forbidden wording such as ``inventory complete``, ``live proof``, and
|
||||||
|
``same as master`` is allowed only when matching current-session evidence
|
||||||
|
appears in the report text or in *session_evidence*.
|
||||||
|
"""
|
||||||
text = report_text or ""
|
text = report_text or ""
|
||||||
evidence = dict(session_evidence or {})
|
evidence = dict(session_evidence or {})
|
||||||
violations: list[str] = []
|
violations: list[str] = []
|
||||||
|
|||||||
@@ -0,0 +1,93 @@
|
|||||||
|
"""Doc-contract checks for proof wording enforcement (#330)."""
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
from review_proofs import assess_proof_wording, build_final_report
|
||||||
|
|
||||||
|
|
||||||
|
class TestProofWording(unittest.TestCase):
|
||||||
|
def test_rejects_inventory_complete_without_pagination_proof(self):
|
||||||
|
result = assess_proof_wording(
|
||||||
|
"Open PR inventory is complete because only 3 PRs were returned."
|
||||||
|
)
|
||||||
|
self.assertFalse(result["proven"])
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
|
||||||
|
def test_rejects_default_page_size_assumption(self):
|
||||||
|
result = assess_proof_wording(
|
||||||
|
"Inventory exhaustive: fewer than the default Gitea page size returned."
|
||||||
|
)
|
||||||
|
self.assertFalse(result["proven"])
|
||||||
|
self.assertIn("default_page_size_assumption", result["flagged_rules"])
|
||||||
|
|
||||||
|
def test_allows_inventory_complete_with_finality_metadata(self):
|
||||||
|
result = assess_proof_wording(
|
||||||
|
"Inventory complete. pagination_complete: true, is_final_page: true"
|
||||||
|
)
|
||||||
|
self.assertTrue(result["proven"])
|
||||||
|
|
||||||
|
def test_allows_inventory_complete_with_session_evidence(self):
|
||||||
|
result = assess_proof_wording(
|
||||||
|
"Queue inventory complete.",
|
||||||
|
session_evidence={"pagination_complete": True},
|
||||||
|
)
|
||||||
|
self.assertTrue(result["proven"])
|
||||||
|
|
||||||
|
def test_rejects_live_proof_without_session_evidence(self):
|
||||||
|
result = assess_proof_wording("Live proof confirms mergeable state.")
|
||||||
|
self.assertFalse(result["proven"])
|
||||||
|
self.assertIn("live_proof", result["flagged_rules"])
|
||||||
|
|
||||||
|
def test_allows_prior_blocker_reuse_wording(self):
|
||||||
|
result = assess_proof_wording(
|
||||||
|
"Prior blocker reused; head SHA unchanged since blocking review. "
|
||||||
|
"Live proof not claimed for this skip."
|
||||||
|
)
|
||||||
|
self.assertTrue(result["proven"])
|
||||||
|
|
||||||
|
def test_allows_live_proof_with_current_session_evidence(self):
|
||||||
|
result = assess_proof_wording(
|
||||||
|
"Live proof: gitea_view_pr revalidated in the current session."
|
||||||
|
)
|
||||||
|
self.assertTrue(result["proven"])
|
||||||
|
|
||||||
|
def test_rejects_same_as_master_without_baseline_proof(self):
|
||||||
|
result = assess_proof_wording("Diff is same as master.")
|
||||||
|
self.assertFalse(result["proven"])
|
||||||
|
self.assertIn("same_as_master", result["flagged_rules"])
|
||||||
|
|
||||||
|
def test_allows_same_as_master_with_baseline_wording(self):
|
||||||
|
result = assess_proof_wording(
|
||||||
|
"Clean baseline worktree proof: diff base matches master."
|
||||||
|
)
|
||||||
|
self.assertTrue(result["proven"])
|
||||||
|
|
||||||
|
def test_rejects_file_edits_none_without_ledger(self):
|
||||||
|
result = assess_proof_wording("Workspace mutations: file edits none.")
|
||||||
|
self.assertFalse(result["proven"])
|
||||||
|
self.assertIn("no_file_edits", result["flagged_rules"])
|
||||||
|
|
||||||
|
def test_allows_file_edits_none_with_mutation_ledger(self):
|
||||||
|
result = assess_proof_wording(
|
||||||
|
"Mutation ledger: tracked file edits: none",
|
||||||
|
session_evidence={"mutation_ledger_clean": True},
|
||||||
|
)
|
||||||
|
self.assertTrue(result["proven"])
|
||||||
|
|
||||||
|
def test_build_final_report_downgrades_on_proof_wording_violation(self):
|
||||||
|
report = build_final_report(
|
||||||
|
{"proven": True},
|
||||||
|
{"complete": True},
|
||||||
|
{"claimable": True, "verdict": "strong"},
|
||||||
|
{"status": "clean"},
|
||||||
|
True,
|
||||||
|
False,
|
||||||
|
True,
|
||||||
|
report_text="Inventory complete after listing 4 open PRs.",
|
||||||
|
)
|
||||||
|
self.assertEqual(report["grade"], "downgraded")
|
||||||
|
self.assertFalse(report["proof_wording_proven"])
|
||||||
|
self.assertTrue(report["proof_wording_violations"])
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -24,6 +24,7 @@ from review_proofs import ( # noqa: E402
|
|||||||
ISSUE_SELECTION_CONTINUATION_EXPLICIT,
|
ISSUE_SELECTION_CONTINUATION_EXPLICIT,
|
||||||
ISSUE_SELECTION_REPRESENTED_BY_OPEN_PR,
|
ISSUE_SELECTION_REPRESENTED_BY_OPEN_PR,
|
||||||
assess_author_pr_report,
|
assess_author_pr_report,
|
||||||
|
assess_queue_ordering_proof,
|
||||||
assess_reviewer_baseline_validation_proof,
|
assess_reviewer_baseline_validation_proof,
|
||||||
assess_capability_evidence,
|
assess_capability_evidence,
|
||||||
assess_capability_proof,
|
assess_capability_proof,
|
||||||
@@ -2161,6 +2162,111 @@ class TestCreateIssueFinalReport(unittest.TestCase):
|
|||||||
self.assertTrue(result["downgraded"])
|
self.assertTrue(result["downgraded"])
|
||||||
|
|
||||||
|
|
||||||
|
class TestQueueOrderingProof(unittest.TestCase):
|
||||||
|
"""Issue #321: reviewer queue selection must prove ordering."""
|
||||||
|
|
||||||
|
def test_next_eligible_claim_without_policy_fails(self):
|
||||||
|
result = assess_queue_ordering_proof(
|
||||||
|
report_text="Selected the next eligible PR: PR #286.",
|
||||||
|
)
|
||||||
|
self.assertFalse(result["proven"])
|
||||||
|
self.assertTrue(result["claims_selection"])
|
||||||
|
|
||||||
|
def test_next_eligible_claim_with_policy_passes(self):
|
||||||
|
result = assess_queue_ordering_proof(
|
||||||
|
report_text=(
|
||||||
|
"Queue ordering policy: oldest PR number first\n"
|
||||||
|
"Selected the next eligible PR: PR #286."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
self.assertTrue(result["proven"])
|
||||||
|
self.assertEqual(result["policy"], "oldest PR number first")
|
||||||
|
|
||||||
|
def test_newest_first_api_with_oldest_first_selection_passes(self):
|
||||||
|
result = assess_queue_ordering_proof(
|
||||||
|
report_text=(
|
||||||
|
"Queue ordering policy: oldest PR number first\n"
|
||||||
|
"API order was newest-first (PR #291 before PR #286).\n"
|
||||||
|
"Selected the next eligible PR: PR #286."
|
||||||
|
),
|
||||||
|
queue_ordering={
|
||||||
|
"policy": "oldest PR number first",
|
||||||
|
"selected_pr_number": 286,
|
||||||
|
"api_pr_numbers": [291, 286],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
self.assertTrue(result["proven"])
|
||||||
|
|
||||||
|
def test_earlier_actionable_pr_without_skip_reason_fails(self):
|
||||||
|
result = assess_queue_ordering_proof(
|
||||||
|
report_text=(
|
||||||
|
"Queue ordering policy: oldest PR number first\n"
|
||||||
|
"Selected the next eligible PR: PR #300."
|
||||||
|
),
|
||||||
|
queue_ordering={
|
||||||
|
"policy": "oldest PR number first",
|
||||||
|
"selected_pr_number": 300,
|
||||||
|
"api_pr_numbers": [300, 286],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
self.assertFalse(result["proven"])
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
|
||||||
|
def test_priority_override_requires_explanation(self):
|
||||||
|
result = assess_queue_ordering_proof(
|
||||||
|
report_text="Selected the next eligible PR: PR #300.",
|
||||||
|
queue_ordering={
|
||||||
|
"policy": "priority label",
|
||||||
|
"selected_pr_number": 300,
|
||||||
|
"api_pr_numbers": [286, 300],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
self.assertFalse(result["proven"])
|
||||||
|
|
||||||
|
def test_priority_override_with_explanation_passes(self):
|
||||||
|
result = assess_queue_ordering_proof(
|
||||||
|
report_text=(
|
||||||
|
"Queue ordering policy: priority label\n"
|
||||||
|
"Priority override: security label on PR #300.\n"
|
||||||
|
"Selected the next eligible PR: PR #300."
|
||||||
|
),
|
||||||
|
queue_ordering={
|
||||||
|
"policy": "priority label",
|
||||||
|
"selected_pr_number": 300,
|
||||||
|
"api_pr_numbers": [286, 300],
|
||||||
|
"priority_override": "security label on PR #300",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
self.assertTrue(result["proven"])
|
||||||
|
|
||||||
|
def test_build_final_report_downgrades_missing_ordering_proof(self):
|
||||||
|
report = build_final_report(
|
||||||
|
checkout_proof=_good_checkout(),
|
||||||
|
inventory=_good_inventory(),
|
||||||
|
validation=_good_validation(),
|
||||||
|
contamination=_good_contamination(),
|
||||||
|
identity_eligible=True,
|
||||||
|
merge_performed=False,
|
||||||
|
issue_status_verified=True,
|
||||||
|
capability_evidence=_good_capability_evidence(),
|
||||||
|
sweep=_good_sweep(),
|
||||||
|
live_state=_good_live_state(),
|
||||||
|
role_boundary=_good_role_boundary(),
|
||||||
|
review_mutation=_good_review_mutation(),
|
||||||
|
controller_handoff=_good_handoff(),
|
||||||
|
capability_proof=_good_capability_proof(),
|
||||||
|
sweep_proof=_good_secret_sweep(),
|
||||||
|
worktree_proof={
|
||||||
|
"worktree_path": "/repo/branches/review-feat-issue-224",
|
||||||
|
"porcelain_status": "",
|
||||||
|
"pr_scope_files": ["docs/wiki/Repositories.md"],
|
||||||
|
"scratch_used": True,
|
||||||
|
},
|
||||||
|
report_text="Selected the next eligible PR: PR #286.",
|
||||||
|
)
|
||||||
|
self.assertNotEqual(report["grade"], "A")
|
||||||
|
self.assertFalse(report["queue_ordering_proven"])
|
||||||
|
|
||||||
class TestReviewerBaselineValidationProof(unittest.TestCase):
|
class TestReviewerBaselineValidationProof(unittest.TestCase):
|
||||||
"""Issue #325: baseline validation must use branches/ worktrees."""
|
"""Issue #325: baseline validation must use branches/ worktrees."""
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,165 @@
|
|||||||
|
"""Tests for workspace-vs-worktree mutation consistency verifier (#313).
|
||||||
|
|
||||||
|
Final reports must not claim ``Workspace mutations: none`` when worktree
|
||||||
|
mutations (reset --hard, checkout, clean, worktree add/remove, ...)
|
||||||
|
occurred, and must report observed worktree / git ref mutations under the
|
||||||
|
precise category fields.
|
||||||
|
"""
|
||||||
|
import sys
|
||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||||
|
|
||||||
|
from review_proofs import assess_workspace_mutation_consistency # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
class TestWorkspaceNoneContradiction(unittest.TestCase):
|
||||||
|
def test_reset_hard_with_workspace_none_fails(self):
|
||||||
|
report = (
|
||||||
|
"Controller Handoff\n"
|
||||||
|
"- Workspace mutations: none (no local file edits)\n"
|
||||||
|
)
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=["git reset --hard FETCH_HEAD"],
|
||||||
|
)
|
||||||
|
self.assertFalse(result["complete"])
|
||||||
|
self.assertTrue(result["downgraded"])
|
||||||
|
self.assertTrue(
|
||||||
|
any("workspace mutations" in r.lower() for r in result["reasons"])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_checkout_with_workspace_none_fails(self):
|
||||||
|
report = "- Workspace mutations: none\n"
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=["git checkout feat/some-branch"],
|
||||||
|
)
|
||||||
|
self.assertFalse(result["complete"])
|
||||||
|
|
||||||
|
def test_worktree_add_with_workspace_none_fails(self):
|
||||||
|
report = "- Workspace mutations: none\n"
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=["git worktree add /tmp/review-pr1 abc123"],
|
||||||
|
)
|
||||||
|
self.assertFalse(result["complete"])
|
||||||
|
|
||||||
|
def test_clean_with_workspace_none_fails(self):
|
||||||
|
report = "- Workspace mutations: none\n"
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=["git clean -fd"],
|
||||||
|
)
|
||||||
|
self.assertFalse(result["complete"])
|
||||||
|
|
||||||
|
def test_self_reported_worktree_mutation_contradicts_workspace_none(self):
|
||||||
|
# No observed command log; the report itself carries the
|
||||||
|
# contradiction (#313 observed behavior).
|
||||||
|
report = (
|
||||||
|
"- Worktree mutations: git reset --hard FETCH_HEAD\n"
|
||||||
|
"- Workspace mutations: none (no local file edits)\n"
|
||||||
|
)
|
||||||
|
result = assess_workspace_mutation_consistency(report)
|
||||||
|
self.assertFalse(result["complete"])
|
||||||
|
self.assertTrue(
|
||||||
|
any("workspace mutations" in r.lower() for r in result["reasons"])
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestPreciseCategoriesPass(unittest.TestCase):
|
||||||
|
def test_file_edits_none_with_worktree_mutation_listed_passes(self):
|
||||||
|
report = (
|
||||||
|
"- File edits by reviewer: none\n"
|
||||||
|
"- Worktree mutations: git reset --hard FETCH_HEAD\n"
|
||||||
|
"- Git ref mutations: git fetch prgs\n"
|
||||||
|
)
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=[
|
||||||
|
"git fetch prgs",
|
||||||
|
"git reset --hard FETCH_HEAD",
|
||||||
|
],
|
||||||
|
)
|
||||||
|
self.assertTrue(result["complete"])
|
||||||
|
self.assertFalse(result["downgraded"])
|
||||||
|
self.assertEqual(result["reasons"], [])
|
||||||
|
|
||||||
|
def test_noop_report_passes(self):
|
||||||
|
report = (
|
||||||
|
"- File edits by reviewer: none\n"
|
||||||
|
"- Worktree mutations: none\n"
|
||||||
|
"- Git ref mutations: none\n"
|
||||||
|
)
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report, observed_commands=[]
|
||||||
|
)
|
||||||
|
self.assertTrue(result["complete"])
|
||||||
|
self.assertEqual(result["reasons"], [])
|
||||||
|
|
||||||
|
def test_fetch_only_with_ref_mutation_reported_passes(self):
|
||||||
|
# Fetch is a git ref mutation, not a worktree mutation; a
|
||||||
|
# workspace-none claim is not contradicted by fetch alone.
|
||||||
|
report = (
|
||||||
|
"- Workspace mutations: none\n"
|
||||||
|
"- Git ref mutations: git fetch prgs master\n"
|
||||||
|
)
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=["git fetch prgs master"],
|
||||||
|
)
|
||||||
|
self.assertTrue(result["complete"])
|
||||||
|
|
||||||
|
|
||||||
|
class TestObservedMutationsMustBeReported(unittest.TestCase):
|
||||||
|
def test_reset_without_worktree_mutation_field_fails(self):
|
||||||
|
report = "- File edits by reviewer: none\n"
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=["git reset --hard FETCH_HEAD"],
|
||||||
|
)
|
||||||
|
self.assertFalse(result["complete"])
|
||||||
|
self.assertTrue(
|
||||||
|
any("worktree mutation" in r.lower() for r in result["reasons"])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_reset_with_worktree_mutations_none_fails(self):
|
||||||
|
report = (
|
||||||
|
"- File edits by reviewer: none\n"
|
||||||
|
"- Worktree mutations: none\n"
|
||||||
|
)
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=["git reset --hard FETCH_HEAD"],
|
||||||
|
)
|
||||||
|
self.assertFalse(result["complete"])
|
||||||
|
|
||||||
|
def test_fetch_without_ref_mutation_field_fails(self):
|
||||||
|
report = (
|
||||||
|
"- File edits by reviewer: none\n"
|
||||||
|
"- Worktree mutations: none\n"
|
||||||
|
)
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=["git fetch prgs master"],
|
||||||
|
)
|
||||||
|
self.assertFalse(result["complete"])
|
||||||
|
self.assertTrue(
|
||||||
|
any("git ref mutation" in r.lower() for r in result["reasons"])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_fetch_with_ref_mutations_none_fails(self):
|
||||||
|
report = (
|
||||||
|
"- Worktree mutations: none\n"
|
||||||
|
"- Git ref mutations: none\n"
|
||||||
|
)
|
||||||
|
result = assess_workspace_mutation_consistency(
|
||||||
|
report,
|
||||||
|
observed_commands=["git fetch prgs master"],
|
||||||
|
)
|
||||||
|
self.assertFalse(result["complete"])
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
Reference in New Issue
Block a user