From 3c1027258b66a29e5c2fbb91369c8ae988c0018a Mon Sep 17 00:00:00 2001 From: OpenCode Date: Fri, 10 Jul 2026 16:53:36 +0700 Subject: [PATCH] feat: Phase Z Z2 batch explanation report -- BXR- schema per-candidate selection reason (winner_exploit/uncertainty_probe/etc.) with safety clearance fraction; makes multi-batch selection auditable (#xxxx) --- docs/research/NEXT_100_PR_MAP.md | 1 + .../evidence/batch_explanation_report.py | 280 +++++++++++ .../evidence/test_batch_explanation_report.py | 440 ++++++++++++++++++ tests/test_test_count_regression.py | 2 +- 4 files changed, 722 insertions(+), 1 deletion(-) create mode 100644 src/openamp_foundry/evidence/batch_explanation_report.py create mode 100644 tests/evidence/test_batch_explanation_report.py diff --git a/docs/research/NEXT_100_PR_MAP.md b/docs/research/NEXT_100_PR_MAP.md index 4ba35928..a7844306 100644 --- a/docs/research/NEXT_100_PR_MAP.md +++ b/docs/research/NEXT_100_PR_MAP.md @@ -313,3 +313,4 @@ Make per-family performance gaps visible and machine-checkable, so the pipeline | PR | Task | Why it matters | Review class | |---:|---|---|---| | Z1 | Add family blindness challenge harness schema (FBH-). | Per-family AUROC + panel representation check; flags when weak AMP classes (AUROC<0.55) are excluded from selected panel; prevents aggregate-metric hiding of family blind spots. | C | +| Z2 | Add batch explanation report schema (BXR-). | Per-candidate selection reason tracking (winner_exploit/uncertainty_probe/diversity_anchor/etc.); safety_cleared flag per candidate; verdict (explained/partially_explained/unexplained) based on safety clearance fraction; makes multi-batch selection auditable. | C | diff --git a/src/openamp_foundry/evidence/batch_explanation_report.py b/src/openamp_foundry/evidence/batch_explanation_report.py new file mode 100644 index 00000000..fcf56bf5 --- /dev/null +++ b/src/openamp_foundry/evidence/batch_explanation_report.py @@ -0,0 +1,280 @@ +"""BXR- batch explanation report schema. + +Per-candidate selection reason tracking for multi-batch pipelines. +Documents the *why* behind each candidate: exploitation, exploration, +safety retest, controls, etc. Makes batch-2 selection auditable. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +VALID_BXR_VERDICTS: frozenset[str] = frozenset({ + "explained", + "partially_explained", + "unexplained", +}) + +VALID_SELECTION_REASON_CATEGORIES: frozenset[str] = frozenset({ + "winner_exploit", + "uncertainty_probe", + "diversity_anchor", + "safety_retest", + "negative_control", + "calibration_check", +}) + +VALID_CONFIDENCE_LEVELS: frozenset[str] = frozenset({ + "high", + "moderate", + "low", + "not_assessed", +}) + +EXPLAINED_FRACTION_THRESHOLD: float = 0.80 +PARTIAL_FRACTION_THRESHOLD: float = 0.50 + + +@dataclass +class CandidateExplanationEntry: + candidate_id: str + selection_reason: str + confidence_level: str + predicted_score: float + uncertainty_score: float + safety_cleared: bool + reason_notes: str + + +@dataclass +class BatchExplanationReport: + bxr_id: str + pipeline_version: str + batch_id: str + candidate_explanations: list[CandidateExplanationEntry] + n_candidates: int + n_explained: int + n_safety_cleared: int + exploit_fraction: float + probe_fraction: float + explanation_fraction: float + verdict: str + dry_lab_only: bool + limitations: list[str] + created_at: str + + +def _compute_verdict( + n_candidates: int, + explanation_fraction: float, +) -> str: + if n_candidates == 0: + return "unexplained" + if explanation_fraction >= EXPLAINED_FRACTION_THRESHOLD: + return "explained" + if explanation_fraction >= PARTIAL_FRACTION_THRESHOLD: + return "partially_explained" + return "unexplained" + + +def validate_batch_explanation_report(bxr: BatchExplanationReport) -> None: + if not bxr.bxr_id.startswith("BXR-"): + raise ValueError(f"bxr_id must start with 'BXR-': {bxr.bxr_id!r}") + if not bxr.pipeline_version: + raise ValueError("pipeline_version must be non-empty") + if not bxr.batch_id: + raise ValueError("batch_id must be non-empty") + + for entry in bxr.candidate_explanations: + if not entry.candidate_id: + raise ValueError("candidate_id must be non-empty") + if entry.selection_reason not in VALID_SELECTION_REASON_CATEGORIES: + raise ValueError( + f"selection_reason {entry.selection_reason!r} " + f"not in VALID_SELECTION_REASON_CATEGORIES" + ) + if entry.confidence_level not in VALID_CONFIDENCE_LEVELS: + raise ValueError( + f"confidence_level {entry.confidence_level!r} " + f"not in VALID_CONFIDENCE_LEVELS" + ) + if not (-1.0 <= entry.predicted_score <= 1.0): + raise ValueError( + f"predicted_score must be in [-1.0, 1.0]: {entry.predicted_score}" + ) + if not (-1.0 <= entry.uncertainty_score <= 1.0): + raise ValueError( + f"uncertainty_score must be in [-1.0, 1.0]: {entry.uncertainty_score}" + ) + if len(entry.reason_notes) > 200: + raise ValueError( + f"reason_notes must be <= 200 chars: " + f"got {len(entry.reason_notes)}" + ) + + if bxr.n_candidates != len(bxr.candidate_explanations): + raise ValueError( + f"n_candidates {bxr.n_candidates} != " + f"len(candidate_explanations) {len(bxr.candidate_explanations)}" + ) + + expected_n_safety = sum(1 for e in bxr.candidate_explanations if e.safety_cleared) + if bxr.n_safety_cleared != expected_n_safety: + raise ValueError( + f"n_safety_cleared {bxr.n_safety_cleared} != computed {expected_n_safety}" + ) + + if bxr.n_explained != bxr.n_safety_cleared: + raise ValueError( + f"n_explained {bxr.n_explained} must equal n_safety_cleared {bxr.n_safety_cleared}" + ) + + if bxr.n_candidates == 0: + expected_explanation_fraction = 0.0 + else: + expected_explanation_fraction = round( + bxr.n_explained / bxr.n_candidates, 6 + ) + if abs(bxr.explanation_fraction - expected_explanation_fraction) > 0.01: + raise ValueError( + f"explanation_fraction {bxr.explanation_fraction} does not match " + f"computed {expected_explanation_fraction}" + ) + + expected_exploit = sum( + 1 for e in bxr.candidate_explanations + if e.selection_reason == "winner_exploit" + ) + expected_exploit_fraction = round( + expected_exploit / bxr.n_candidates, 6 + ) if bxr.n_candidates > 0 else 0.0 + if abs(bxr.exploit_fraction - expected_exploit_fraction) > 0.01: + raise ValueError( + f"exploit_fraction {bxr.exploit_fraction} != computed {expected_exploit_fraction}" + ) + + expected_probe = sum( + 1 for e in bxr.candidate_explanations + if e.selection_reason == "uncertainty_probe" + ) + expected_probe_fraction = round( + expected_probe / bxr.n_candidates, 6 + ) if bxr.n_candidates > 0 else 0.0 + if abs(bxr.probe_fraction - expected_probe_fraction) > 0.01: + raise ValueError( + f"probe_fraction {bxr.probe_fraction} != computed {expected_probe_fraction}" + ) + + if bxr.verdict not in VALID_BXR_VERDICTS: + raise ValueError(f"verdict {bxr.verdict!r} not in VALID_BXR_VERDICTS") + + expected_verdict = _compute_verdict(bxr.n_candidates, bxr.explanation_fraction) + if bxr.verdict != expected_verdict: + raise ValueError( + f"verdict {bxr.verdict!r} does not match computed verdict " + f"{expected_verdict!r}" + ) + + if not bxr.dry_lab_only: + raise ValueError("dry_lab_only must be True") + if not bxr.limitations: + raise ValueError("limitations must be non-empty") + if not bxr.created_at: + raise ValueError("created_at must be non-empty") + + +def build_batch_explanation_report( + *, + bxr_id: str, + pipeline_version: str, + batch_id: str, + candidate_explanations: list[dict | CandidateExplanationEntry], + limitations: list[str], + created_at: str, +) -> BatchExplanationReport: + """Build a BatchExplanationReport. + + candidate_explanations: list of dicts with keys: + candidate_id (str), selection_reason (str), + confidence_level (str), predicted_score (float), + uncertainty_score (float), safety_cleared (bool), + reason_notes (str, optional, default "") + """ + entries = [] + for item in candidate_explanations: + if isinstance(item, CandidateExplanationEntry): + entries.append(item) + else: + d = item + entries.append( + CandidateExplanationEntry( + candidate_id=d["candidate_id"], + selection_reason=d["selection_reason"], + confidence_level=d["confidence_level"], + predicted_score=float(d["predicted_score"]), + uncertainty_score=float(d["uncertainty_score"]), + safety_cleared=bool(d["safety_cleared"]), + reason_notes=d.get("reason_notes", ""), + ) + ) + + n_candidates = len(entries) + n_safety_cleared = sum(1 for e in entries if e.safety_cleared) + n_explained = n_safety_cleared + explanation_fraction = round(n_explained / n_candidates, 6) if n_candidates > 0 else 0.0 + exploit_fraction = round( + sum(1 for e in entries if e.selection_reason == "winner_exploit") / n_candidates, 6 + ) if n_candidates > 0 else 0.0 + probe_fraction = round( + sum(1 for e in entries if e.selection_reason == "uncertainty_probe") / n_candidates, 6 + ) if n_candidates > 0 else 0.0 + verdict = _compute_verdict(n_candidates, explanation_fraction) + + bxr = BatchExplanationReport( + bxr_id=bxr_id, + pipeline_version=pipeline_version, + batch_id=batch_id, + candidate_explanations=entries, + n_candidates=n_candidates, + n_explained=n_explained, + n_safety_cleared=n_safety_cleared, + exploit_fraction=exploit_fraction, + probe_fraction=probe_fraction, + explanation_fraction=explanation_fraction, + verdict=verdict, + dry_lab_only=True, + limitations=limitations, + created_at=created_at, + ) + validate_batch_explanation_report(bxr) + return bxr + + +def format_batch_explanation_report(bxr: BatchExplanationReport) -> str: + lines = [ + f"Batch Explanation Report — {bxr.bxr_id}", + f"Pipeline: {bxr.pipeline_version} | Batch: {bxr.batch_id}", + f"Verdict: {bxr.verdict}", + f"Candidates: {bxr.n_candidates} total, " + f"{bxr.n_safety_cleared} safety-cleared " + f"({bxr.explanation_fraction:.1%} explained)", + f"Exploit fraction: {bxr.exploit_fraction:.1%} | " + f"Probe fraction: {bxr.probe_fraction:.1%}", + ] + if bxr.candidate_explanations: + lines.append("Candidate explanations:") + for entry in bxr.candidate_explanations: + safety_flag = "CLEARED" if entry.safety_cleared else "NOT_CLEARED" + lines.append( + f" {entry.candidate_id}: {entry.selection_reason} " + f"(confidence={entry.confidence_level}, " + f"score={entry.predicted_score:.3f}, " + f"uncertainty={entry.uncertainty_score:.3f}, " + f"safety={safety_flag})" + ) + if entry.reason_notes: + lines.append(f" {entry.reason_notes}") + lines.append(f"Created: {bxr.created_at}") + lines.append(f"Limitations: {'; '.join(bxr.limitations)}") + lines.append(f"dry_lab_only: {bxr.dry_lab_only}") + return "\n".join(lines) diff --git a/tests/evidence/test_batch_explanation_report.py b/tests/evidence/test_batch_explanation_report.py new file mode 100644 index 00000000..40b57c5b --- /dev/null +++ b/tests/evidence/test_batch_explanation_report.py @@ -0,0 +1,440 @@ +"""Tests for BXR- batch explanation report schema.""" + +import pytest +from openamp_foundry.evidence.batch_explanation_report import ( + BatchExplanationReport, + CandidateExplanationEntry, + VALID_BXR_VERDICTS, + VALID_SELECTION_REASON_CATEGORIES, + VALID_CONFIDENCE_LEVELS, + EXPLAINED_FRACTION_THRESHOLD, + PARTIAL_FRACTION_THRESHOLD, + build_batch_explanation_report, + format_batch_explanation_report, + validate_batch_explanation_report, +) + +# --------------------------------------------------------------------------- +# Test data +# --------------------------------------------------------------------------- + +_ALL_EXPLAINED = [ + { + "candidate_id": "CAND-001", + "selection_reason": "winner_exploit", + "confidence_level": "high", + "predicted_score": 0.85, + "uncertainty_score": 0.10, + "safety_cleared": True, + "reason_notes": "Strong predicted activity and low uncertainty", + }, + { + "candidate_id": "CAND-002", + "selection_reason": "uncertainty_probe", + "confidence_level": "low", + "predicted_score": 0.45, + "uncertainty_score": 0.80, + "safety_cleared": True, + "reason_notes": "High uncertainty; exploration pick", + }, + { + "candidate_id": "CAND-003", + "selection_reason": "diversity_anchor", + "confidence_level": "moderate", + "predicted_score": 0.60, + "uncertainty_score": 0.35, + "safety_cleared": True, + "reason_notes": "", + }, +] + +_PARTIALLY_EXPLAINED = [ + { + "candidate_id": "CAND-001", + "selection_reason": "winner_exploit", + "confidence_level": "high", + "predicted_score": 0.82, + "uncertainty_score": 0.15, + "safety_cleared": True, + "reason_notes": "", + }, + { + "candidate_id": "CAND-002", + "selection_reason": "uncertainty_probe", + "confidence_level": "low", + "predicted_score": 0.40, + "uncertainty_score": 0.75, + "safety_cleared": False, + "reason_notes": "Safety gate not yet run", + }, + { + "candidate_id": "CAND-003", + "selection_reason": "safety_retest", + "confidence_level": "moderate", + "predicted_score": 0.70, + "uncertainty_score": 0.25, + "safety_cleared": True, + "reason_notes": "", + }, + { + "candidate_id": "CAND-004", + "selection_reason": "winner_exploit", + "confidence_level": "high", + "predicted_score": 0.90, + "uncertainty_score": 0.05, + "safety_cleared": False, + "reason_notes": "Awaiting safety review", + }, +] + +_ALL_EXPLOIT = [ + { + "candidate_id": "CAND-001", + "selection_reason": "winner_exploit", + "confidence_level": "high", + "predicted_score": 0.90, + "uncertainty_score": 0.05, + "safety_cleared": True, + "reason_notes": "", + }, +] + +_ALL_PROBE = [ + { + "candidate_id": "CAND-001", + "selection_reason": "uncertainty_probe", + "confidence_level": "low", + "predicted_score": 0.30, + "uncertainty_score": 0.85, + "safety_cleared": True, + "reason_notes": "", + }, +] + + +def _build(**kwargs): + defaults = dict( + bxr_id="BXR-001", + pipeline_version="v2.1", + batch_id="BSP-001", + candidate_explanations=_ALL_EXPLAINED, + limitations=["dry-lab only", "predicted scores are estimates"], + created_at="2026-07-10", + ) + defaults.update(kwargs) + return build_batch_explanation_report(**defaults) + + +# --------------------------------------------------------------------------- +# Section 1: Constants +# --------------------------------------------------------------------------- + + +def test_valid_bxr_verdicts_is_frozenset(): + assert isinstance(VALID_BXR_VERDICTS, frozenset) + + +def test_valid_bxr_verdicts_contains_explained(): + assert "explained" in VALID_BXR_VERDICTS + + +def test_valid_bxr_verdicts_contains_partially_explained(): + assert "partially_explained" in VALID_BXR_VERDICTS + + +def test_valid_bxr_verdicts_contains_unexplained(): + assert "unexplained" in VALID_BXR_VERDICTS + + +def test_valid_selection_reason_categories_is_frozenset(): + assert isinstance(VALID_SELECTION_REASON_CATEGORIES, frozenset) + + +def test_valid_selection_reason_categories_contains_winner_exploit(): + assert "winner_exploit" in VALID_SELECTION_REASON_CATEGORIES + + +def test_valid_selection_reason_categories_contains_uncertainty_probe(): + assert "uncertainty_probe" in VALID_SELECTION_REASON_CATEGORIES + + +def test_valid_selection_reason_categories_contains_diversity_anchor(): + assert "diversity_anchor" in VALID_SELECTION_REASON_CATEGORIES + + +def test_valid_selection_reason_categories_contains_safety_retest(): + assert "safety_retest" in VALID_SELECTION_REASON_CATEGORIES + + +def test_valid_selection_reason_categories_contains_negative_control(): + assert "negative_control" in VALID_SELECTION_REASON_CATEGORIES + + +def test_explained_fraction_threshold(): + assert EXPLAINED_FRACTION_THRESHOLD == 0.80 + + +def test_partial_fraction_threshold(): + assert PARTIAL_FRACTION_THRESHOLD == 0.50 + + +# --------------------------------------------------------------------------- +# Section 2: build – happy paths +# --------------------------------------------------------------------------- + + +def test_build_returns_batch_explanation_report(): + assert isinstance(_build(), BatchExplanationReport) + + +def test_build_bxr_id_stored(): + assert _build().bxr_id == "BXR-001" + + +def test_build_pipeline_version_stored(): + assert _build().pipeline_version == "v2.1" + + +def test_build_batch_id_stored(): + assert _build().batch_id == "BSP-001" + + +def test_build_dry_lab_only_true(): + assert _build().dry_lab_only is True + + +def test_build_n_candidates_auto_computed(): + r = _build() + assert r.n_candidates == 3 + + +def test_build_n_safety_cleared_auto_computed(): + r = _build() + assert r.n_safety_cleared == 3 + + +def test_build_n_explained_equals_safety_cleared(): + r = _build() + assert r.n_explained == r.n_safety_cleared + + +def test_build_explained_verdict_when_all_safety_cleared(): + r = _build() + assert r.verdict == "explained" + + +def test_build_explanation_fraction_one_when_all_cleared(): + r = _build() + assert r.explanation_fraction == 1.0 + + +def test_build_partially_explained_verdict(): + r = _build(candidate_explanations=_PARTIALLY_EXPLAINED) + assert r.verdict == "partially_explained" + + +def test_build_partially_explained_n_safety_cleared(): + r = _build(candidate_explanations=_PARTIALLY_EXPLAINED) + assert r.n_safety_cleared == 2 + + +def test_build_partially_explained_explanation_fraction(): + r = _build(candidate_explanations=_PARTIALLY_EXPLAINED) + assert abs(r.explanation_fraction - 0.50) < 1e-4 + + +def test_build_exploit_fraction_all_exploit(): + r = _build(candidate_explanations=_ALL_EXPLOIT) + assert r.exploit_fraction == 1.0 + + +def test_build_probe_fraction_all_probe(): + r = _build(candidate_explanations=_ALL_PROBE) + assert r.probe_fraction == 1.0 + + +def test_build_unexplained_verdict_when_empty(): + r = _build(candidate_explanations=[]) + assert r.verdict == "unexplained" + + +def test_build_empty_n_candidates_zero(): + r = _build(candidate_explanations=[]) + assert r.n_candidates == 0 + + +def test_build_empty_fractions_zero(): + r = _build(candidate_explanations=[]) + assert r.explanation_fraction == 0.0 + assert r.exploit_fraction == 0.0 + assert r.probe_fraction == 0.0 + + +# --------------------------------------------------------------------------- +# Section 3: validate – rejection cases +# --------------------------------------------------------------------------- + + +def test_validate_rejects_bad_bxr_id_prefix(): + with pytest.raises(ValueError, match="BXR-"): + _build(bxr_id="BAD-001") + + +def test_validate_rejects_empty_pipeline_version(): + with pytest.raises(ValueError): + _build(pipeline_version="") + + +def test_validate_rejects_empty_batch_id(): + with pytest.raises(ValueError): + _build(batch_id="") + + +def test_validate_rejects_invalid_selection_reason(): + entries = [{"candidate_id": "X", "selection_reason": "unknown_reason", + "confidence_level": "high", "predicted_score": 0.5, + "uncertainty_score": 0.1, "safety_cleared": True}] + with pytest.raises(ValueError, match="VALID_SELECTION_REASON_CATEGORIES"): + _build(candidate_explanations=entries) + + +def test_validate_rejects_invalid_confidence_level(): + entries = [{"candidate_id": "X", "selection_reason": "winner_exploit", + "confidence_level": "unknown_conf", "predicted_score": 0.5, + "uncertainty_score": 0.1, "safety_cleared": True}] + with pytest.raises(ValueError, match="VALID_CONFIDENCE_LEVELS"): + _build(candidate_explanations=entries) + + +def test_validate_rejects_predicted_score_below_negative_one(): + entries = [{"candidate_id": "X", "selection_reason": "winner_exploit", + "confidence_level": "high", "predicted_score": -1.5, + "uncertainty_score": 0.1, "safety_cleared": True}] + with pytest.raises(ValueError, match="predicted_score"): + _build(candidate_explanations=entries) + + +def test_validate_rejects_predicted_score_above_one(): + entries = [{"candidate_id": "X", "selection_reason": "winner_exploit", + "confidence_level": "high", "predicted_score": 1.5, + "uncertainty_score": 0.1, "safety_cleared": True}] + with pytest.raises(ValueError, match="predicted_score"): + _build(candidate_explanations=entries) + + +def test_validate_rejects_uncertainty_score_below_negative_one(): + entries = [{"candidate_id": "X", "selection_reason": "winner_exploit", + "confidence_level": "high", "predicted_score": 0.5, + "uncertainty_score": -1.5, "safety_cleared": True}] + with pytest.raises(ValueError, match="uncertainty_score"): + _build(candidate_explanations=entries) + + +def test_validate_rejects_uncertainty_score_above_one(): + entries = [{"candidate_id": "X", "selection_reason": "winner_exploit", + "confidence_level": "high", "predicted_score": 0.5, + "uncertainty_score": 1.5, "safety_cleared": True}] + with pytest.raises(ValueError, match="uncertainty_score"): + _build(candidate_explanations=entries) + + +def test_validate_rejects_reason_notes_too_long(): + entries = [{"candidate_id": "X", "selection_reason": "winner_exploit", + "confidence_level": "high", "predicted_score": 0.5, + "uncertainty_score": 0.1, "safety_cleared": True, + "reason_notes": "x" * 201}] + with pytest.raises(ValueError, match="reason_notes"): + _build(candidate_explanations=entries) + + +def test_validate_rejects_empty_candidate_id(): + entries = [{"candidate_id": "", "selection_reason": "winner_exploit", + "confidence_level": "high", "predicted_score": 0.5, + "uncertainty_score": 0.1, "safety_cleared": True}] + with pytest.raises(ValueError, match="candidate_id"): + _build(candidate_explanations=entries) + + +def test_validate_rejects_n_candidates_mismatch(): + r = _build() + r.n_candidates = 999 + with pytest.raises(ValueError, match="n_candidates"): + validate_batch_explanation_report(r) + + +def test_validate_rejects_n_safety_cleared_mismatch(): + r = _build() + r.n_safety_cleared = 999 + with pytest.raises(ValueError, match="n_safety_cleared"): + validate_batch_explanation_report(r) + + +def test_validate_rejects_n_explained_not_equal_safety(): + r = _build() + r.n_explained = 0 + with pytest.raises(ValueError, match="n_explained"): + validate_batch_explanation_report(r) + + +def test_validate_rejects_explanation_fraction_mismatch(): + r = _build() + r.explanation_fraction = 0.99 + with pytest.raises(ValueError, match="explanation_fraction"): + validate_batch_explanation_report(r) + + +def test_validate_rejects_dry_lab_only_false(): + r = _build() + r.dry_lab_only = False + with pytest.raises(ValueError, match="dry_lab_only"): + validate_batch_explanation_report(r) + + +def test_validate_rejects_empty_limitations(): + with pytest.raises(ValueError, match="limitations"): + _build(limitations=[]) + + +def test_validate_rejects_empty_created_at(): + with pytest.raises(ValueError): + _build(created_at="") + + +# --------------------------------------------------------------------------- +# Section 4: format +# --------------------------------------------------------------------------- + + +def test_format_contains_bxr_id(): + assert "BXR-001" in format_batch_explanation_report(_build()) + + +def test_format_contains_batch_id(): + assert "BSP-001" in format_batch_explanation_report(_build()) + + +def test_format_contains_verdict(): + assert "explained" in format_batch_explanation_report(_build()) + + +def test_format_contains_candidate_id(): + text = format_batch_explanation_report(_build()) + assert "CAND-001" in text + + +def test_format_contains_selection_reason(): + text = format_batch_explanation_report(_build()) + assert "winner_exploit" in text + + +def test_format_contains_safety_flag(): + text = format_batch_explanation_report(_build()) + assert "CLEARED" in text + + +def test_format_contains_limitations(): + assert "dry-lab only" in format_batch_explanation_report(_build()) + + +def test_format_is_string(): + assert isinstance(format_batch_explanation_report(_build()), str) diff --git a/tests/test_test_count_regression.py b/tests/test_test_count_regression.py index fe06f834..01929473 100644 --- a/tests/test_test_count_regression.py +++ b/tests/test_test_count_regression.py @@ -4,7 +4,7 @@ import sys import math -BASELINE = 11299 +BASELINE = 11355 def test_test_count_regression():