diff --git a/docs/research/NEXT_100_PR_MAP.md b/docs/research/NEXT_100_PR_MAP.md index 49c6fb51..4ba35928 100644 --- a/docs/research/NEXT_100_PR_MAP.md +++ b/docs/research/NEXT_100_PR_MAP.md @@ -305,3 +305,11 @@ Track and publish structured comparisons between pipeline selections and cheap b | Y3 | Add selection diversity audit schema (SDA-) (complete). — src/openamp_foundry/evidence/selection_diversity_audit.py: VALID_SDA_VERDICTS (4: diverse_panel/moderately_diverse/proximity_driven/insufficient_data), VALID_DIVERSITY_METRICS (4), DIVERSE_PANEL_THRESHOLD=0.10, PROXIMITY_DRIVEN_THRESHOLD=-0.05, MIN_PANEL_SIZE=3; diversity_delta auto-computed; verdict: diverse_panel (delta>=0.10), proximity_driven (delta<=-0.05); dry_lab_only=True; 46 tests. | Tracks sequence diversity of selected panel vs random draw; detects proximity-driven selection masquerading as discovery; required before any novelty claim. | C | | Y4 | Add pipeline maturity certificate schema (PMC-) (complete). — src/openamp_foundry/evidence/pipeline_maturity_certificate.py: VALID_PMC_GRADES (A-D), VALID_PMC_VERDICTS (4: pipeline_validated/pipeline_provisional/pipeline_unvalidated/insufficient_evidence), REQUIRED_PMC_COMPONENTS=(CBR,FIA,SDA); PMCComponentCheck helper; grade A (all 3 superior), B (2), C (1), D (0/none assessed); contributes_to_grade auto-derived from verdict; dry_lab_only=True; 54 tests. | Aggregates CBR/FIA/SDA results into A/B/C/D maturity grade; anchors pre-registration; prevents retroactive interpretation. | C | | Y5 | Add Phase Y accountability gate (YAG-) (complete). — src/openamp_foundry/evidence/phase_y_accountability_gate.py: REQUIRED_Y_COMPONENTS=(CBR,FIA,SDA,PMC), VALID_YAG_VERDICTS (3: accountability_verified/accountability_partial/accountability_not_established); YComponentCheck helper; verdict: accountability_verified (all 4), accountability_partial (2-3), accountability_not_established (0-1); artifact_id prefix-validated per component; dry_lab_only=True; 54 tests. Closes Phase Y. | Top-level gate asserting CBR+FIA+SDA+PMC all present; closes Phase Y; no external pilot claim is credible without passing this gate. | C | + +## Phase Z — Per-family benchmark accountability + +Make per-family performance gaps visible and machine-checkable, so the pipeline cannot hide weak AMP class coverage behind aggregate metrics. + +| PR | Task | Why it matters | Review class | +|---:|---|---|---| +| Z1 | Add family blindness challenge harness schema (FBH-). | Per-family AUROC + panel representation check; flags when weak AMP classes (AUROC<0.55) are excluded from selected panel; prevents aggregate-metric hiding of family blind spots. | C | diff --git a/src/openamp_foundry/evidence/family_blindness_challenge.py b/src/openamp_foundry/evidence/family_blindness_challenge.py new file mode 100644 index 00000000..b40841bb --- /dev/null +++ b/src/openamp_foundry/evidence/family_blindness_challenge.py @@ -0,0 +1,219 @@ +"""FBH- family blindness challenge harness schema. + +Per-family benchmark performance record: flags when the pipeline +underperforms on specific AMP families. Prevents aggregate-metric +hiding of family-blind spots. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +VALID_FBH_VERDICTS: frozenset[str] = frozenset({ + "all_families_represented", + "weak_family_excluded", + "insufficient_data", +}) + +VALID_AUROC_GRADES: frozenset[str] = frozenset({ + "strong", + "adequate", + "weak", + "not_evaluated", +}) + +WEAK_FAMILY_AUROC_THRESHOLD: float = 0.55 +WEAK_FAMILY_PANEL_FLOOR: float = 0.10 + + +@dataclass +class FamilyPerformanceEntry: + family_name: str + auroc: float + n_candidates: int + panel_count: int + panel_fraction: float + auroc_grade: str + is_weak_class: bool + + +@dataclass +class FamilyBlingnessChallenge: + fbh_id: str + pipeline_version: str + family_entries: list[FamilyPerformanceEntry] + n_families_total: int + n_weak_families: int + n_excluded_weak_families: int + excluded_weak_family_names: list[str] + verdict: str + dry_lab_only: bool + limitations: list[str] + created_at: str + + +def _compute_auroc_grade(auroc: float) -> str: + if auroc == -1.0: + return "not_evaluated" + if auroc >= 0.70: + return "strong" + if auroc >= 0.55: + return "adequate" + return "weak" + + +def _compute_is_weak_class(auroc: float, n_candidates: int) -> bool: + return auroc < WEAK_FAMILY_AUROC_THRESHOLD and n_candidates >= 3 + + +def _compute_verdict(family_entries: list[FamilyPerformanceEntry]) -> str: + if len(family_entries) == 0: + return "insufficient_data" + for entry in family_entries: + if entry.is_weak_class and entry.panel_fraction < WEAK_FAMILY_PANEL_FLOOR: + return "weak_family_excluded" + return "all_families_represented" + + +def validate_family_blindness_challenge(fbh: FamilyBlingnessChallenge) -> None: + if not fbh.fbh_id.startswith("FBH-"): + raise ValueError(f"fbh_id must start with 'FBH-': {fbh.fbh_id!r}") + if not fbh.pipeline_version: + raise ValueError("pipeline_version must be non-empty") + for entry in fbh.family_entries: + if not (-1.0 <= entry.auroc <= 1.0): + raise ValueError(f"auroc must be in [-1.0, 1.0]: {entry.auroc}") + if entry.n_candidates < 1: + raise ValueError(f"n_candidates must be >= 1: {entry.n_candidates}") + if entry.panel_count < 0: + raise ValueError(f"panel_count must be >= 0: {entry.panel_count}") + if entry.panel_count > entry.n_candidates: + raise ValueError( + f"panel_count ({entry.panel_count}) cannot exceed " + f"n_candidates ({entry.n_candidates})" + ) + if fbh.n_families_total != len(fbh.family_entries): + raise ValueError( + f"n_families_total {fbh.n_families_total} != " + f"len(family_entries) {len(fbh.family_entries)}" + ) + expected_n_weak = sum(1 for e in fbh.family_entries if e.is_weak_class) + if fbh.n_weak_families != expected_n_weak: + raise ValueError( + f"n_weak_families {fbh.n_weak_families} != computed {expected_n_weak}" + ) + expected_excluded = [ + e.family_name + for e in fbh.family_entries + if e.is_weak_class and e.panel_fraction < WEAK_FAMILY_PANEL_FLOOR + ] + if fbh.n_excluded_weak_families != len(expected_excluded): + raise ValueError( + f"n_excluded_weak_families {fbh.n_excluded_weak_families} != " + f"computed {len(expected_excluded)}" + ) + if fbh.excluded_weak_family_names != expected_excluded: + raise ValueError("excluded_weak_family_names mismatch") + if fbh.verdict not in VALID_FBH_VERDICTS: + raise ValueError(f"verdict {fbh.verdict!r} not in VALID_FBH_VERDICTS") + if fbh.verdict == "weak_family_excluded" and fbh.n_excluded_weak_families < 1: + raise ValueError( + "verdict weak_family_excluded but n_excluded_weak_families < 1" + ) + if fbh.verdict == "all_families_represented" and fbh.n_excluded_weak_families != 0: + raise ValueError( + "verdict all_families_represented but n_excluded_weak_families != 0" + ) + if not fbh.dry_lab_only: + raise ValueError("dry_lab_only must be True") + if not fbh.limitations: + raise ValueError("limitations must be non-empty") + if not fbh.created_at: + raise ValueError("created_at must be non-empty") + + +def build_family_blindness_challenge( + *, + fbh_id: str, + pipeline_version: str, + family_entry_dicts: list[dict], + limitations: list[str], + created_at: str, +) -> FamilyBlingnessChallenge: + """Build a FamilyBlingnessChallenge. + + family_entry_dicts: list of dicts with keys: + family_name (str), auroc (float), + n_candidates (int), panel_count (int) + """ + family_entries = [] + for d in family_entry_dicts: + auroc = float(d["auroc"]) + n_candidates = int(d["n_candidates"]) + panel_count = int(d["panel_count"]) + panel_fraction = round(panel_count / n_candidates, 6) if n_candidates > 0 else 0.0 + auroc_grade = _compute_auroc_grade(auroc) + is_weak_class = _compute_is_weak_class(auroc, n_candidates) + family_entries.append( + FamilyPerformanceEntry( + family_name=d["family_name"], + auroc=auroc, + n_candidates=n_candidates, + panel_count=panel_count, + panel_fraction=panel_fraction, + auroc_grade=auroc_grade, + is_weak_class=is_weak_class, + ) + ) + n_families_total = len(family_entries) + n_weak_families = sum(1 for e in family_entries if e.is_weak_class) + excluded_names = [ + e.family_name + for e in family_entries + if e.is_weak_class and e.panel_fraction < WEAK_FAMILY_PANEL_FLOOR + ] + n_excluded_weak_families = len(excluded_names) + verdict = _compute_verdict(family_entries) + fbh = FamilyBlingnessChallenge( + fbh_id=fbh_id, + pipeline_version=pipeline_version, + family_entries=family_entries, + n_families_total=n_families_total, + n_weak_families=n_weak_families, + n_excluded_weak_families=n_excluded_weak_families, + excluded_weak_family_names=excluded_names, + verdict=verdict, + dry_lab_only=True, + limitations=limitations, + created_at=created_at, + ) + validate_family_blindness_challenge(fbh) + return fbh + + +def format_family_blindness_challenge(fbh: FamilyBlingnessChallenge) -> str: + lines = [ + f"Family Blindness Challenge — {fbh.fbh_id}", + f"Pipeline: {fbh.pipeline_version}", + f"Verdict: {fbh.verdict}", + f"Families: {fbh.n_families_total} total, {fbh.n_weak_families} weak, " + f"{fbh.n_excluded_weak_families} excluded", + ] + if fbh.excluded_weak_family_names: + lines.append( + "Excluded weak families: " + f"{', '.join(fbh.excluded_weak_family_names)}" + ) + if fbh.family_entries: + lines.append("Per-family performance:") + for entry in fbh.family_entries: + lines.append( + f" {entry.family_name}: AUROC={entry.auroc:.3f} " + f"(grade={entry.auroc_grade}), " + f"panel={entry.panel_count}/{entry.n_candidates} " + f"({entry.panel_fraction:.1%}), weak={entry.is_weak_class}" + ) + lines.append(f"Created: {fbh.created_at}") + lines.append(f"Limitations: {'; '.join(fbh.limitations)}") + lines.append(f"dry_lab_only: {fbh.dry_lab_only}") + return "\n".join(lines) diff --git a/tests/evidence/test_family_blindness_challenge.py b/tests/evidence/test_family_blindness_challenge.py new file mode 100644 index 00000000..c99d99e5 --- /dev/null +++ b/tests/evidence/test_family_blindness_challenge.py @@ -0,0 +1,367 @@ +"""Tests for FBH- family blindness challenge harness schema.""" + +import pytest +from openamp_foundry.evidence.family_blindness_challenge import ( + FamilyBlingnessChallenge, + FamilyPerformanceEntry, + VALID_FBH_VERDICTS, + VALID_AUROC_GRADES, + WEAK_FAMILY_AUROC_THRESHOLD, + WEAK_FAMILY_PANEL_FLOOR, + build_family_blindness_challenge, + format_family_blindness_challenge, + validate_family_blindness_challenge, +) + +# --------------------------------------------------------------------------- +# Test data +# --------------------------------------------------------------------------- + +_ALL_REPRESENTED = [ + {"family_name": "defensin", "auroc": 0.85, "n_candidates": 100, "panel_count": 30}, + {"family_name": "cathelicidin", "auroc": 0.72, "n_candidates": 50, "panel_count": 15}, +] + +_WEAK_EXCLUDED = [ + {"family_name": "defensin", "auroc": 0.85, "n_candidates": 100, "panel_count": 30}, + {"family_name": "cathelicidin", "auroc": 0.72, "n_candidates": 50, "panel_count": 15}, + {"family_name": "bacteriocin", "auroc": 0.45, "n_candidates": 20, "panel_count": 1}, +] + +_WEAK_NOT_EXCLUDED = [ + {"family_name": "defensin", "auroc": 0.85, "n_candidates": 100, "panel_count": 30}, + {"family_name": "cathelicidin", "auroc": 0.72, "n_candidates": 50, "panel_count": 15}, + {"family_name": "bacteriocin", "auroc": 0.50, "n_candidates": 10, "panel_count": 3}, +] + + +def _build(**kwargs): + defaults = dict( + fbh_id="FBH-001", + pipeline_version="v1.0", + family_entry_dicts=_ALL_REPRESENTED, + limitations=["dry-lab only", "AUROC estimates are approximate"], + created_at="2026-07-10", + ) + defaults.update(kwargs) + return build_family_blindness_challenge(**defaults) + + +# --------------------------------------------------------------------------- +# Section 1: Constants +# --------------------------------------------------------------------------- + + +def test_valid_fbh_verdicts_is_frozenset(): + assert isinstance(VALID_FBH_VERDICTS, frozenset) + + +def test_valid_fbh_verdicts_contains_all_families_represented(): + assert "all_families_represented" in VALID_FBH_VERDICTS + + +def test_valid_fbh_verdicts_contains_weak_family_excluded(): + assert "weak_family_excluded" in VALID_FBH_VERDICTS + + +def test_valid_fbh_verdicts_contains_insufficient_data(): + assert "insufficient_data" in VALID_FBH_VERDICTS + + +def test_valid_auroc_grades_is_frozenset(): + assert isinstance(VALID_AUROC_GRADES, frozenset) + + +def test_valid_auroc_grades_contains_strong(): + assert "strong" in VALID_AUROC_GRADES + + +def test_valid_auroc_grades_contains_adequate(): + assert "adequate" in VALID_AUROC_GRADES + + +def test_valid_auroc_grades_contains_weak(): + assert "weak" in VALID_AUROC_GRADES + + +def test_valid_auroc_grades_contains_not_evaluated(): + assert "not_evaluated" in VALID_AUROC_GRADES + + +def test_weak_family_auroc_threshold(): + assert WEAK_FAMILY_AUROC_THRESHOLD == 0.55 + + +def test_weak_family_panel_floor(): + assert WEAK_FAMILY_PANEL_FLOOR == 0.10 + + +# --------------------------------------------------------------------------- +# Section 2: build – happy paths +# --------------------------------------------------------------------------- + + +def test_build_returns_fbh(): + assert isinstance(_build(), FamilyBlingnessChallenge) + + +def test_build_fbh_id_stored(): + assert _build().fbh_id == "FBH-001" + + +def test_build_pipeline_version_stored(): + assert _build().pipeline_version == "v1.0" + + +def test_build_dry_lab_only_true(): + assert _build().dry_lab_only is True + + +def test_build_all_families_represented_verdict(): + r = _build(family_entry_dicts=_ALL_REPRESENTED) + assert r.verdict == "all_families_represented" + + +def test_build_n_families_total_auto_computed(): + r = _build(family_entry_dicts=_ALL_REPRESENTED) + assert r.n_families_total == 2 + + +def test_build_n_weak_families_zero_when_all_strong(): + r = _build(family_entry_dicts=_ALL_REPRESENTED) + assert r.n_weak_families == 0 + + +def test_build_weak_family_excluded_verdict(): + r = _build(family_entry_dicts=_WEAK_EXCLUDED) + assert r.verdict == "weak_family_excluded" + + +def test_build_weak_family_excluded_counts_and_names(): + r = _build(family_entry_dicts=_WEAK_EXCLUDED) + assert r.n_weak_families == 1 + assert r.n_excluded_weak_families == 1 + assert r.excluded_weak_family_names == ["bacteriocin"] + + +def test_build_insufficient_data_verdict(): + r = _build(family_entry_dicts=[]) + assert r.verdict == "insufficient_data" + + +def test_build_panel_fraction_auto_computed(): + r = _build(family_entry_dicts=_ALL_REPRESENTED) + assert abs(r.family_entries[0].panel_fraction - 0.30) < 1e-4 + assert abs(r.family_entries[1].panel_fraction - 0.30) < 1e-4 + + +def test_build_auroc_grade_strong(): + r = _build(family_entry_dicts=_ALL_REPRESENTED) + assert r.family_entries[0].auroc_grade == "strong" + assert r.family_entries[1].auroc_grade == "strong" + + +def test_build_auroc_grade_adequate(): + entries = [ + {"family_name": "test", "auroc": 0.60, "n_candidates": 10, "panel_count": 3}, + ] + r = _build(family_entry_dicts=entries) + assert r.family_entries[0].auroc_grade == "adequate" + + +def test_build_auroc_grade_weak(): + entries = [ + {"family_name": "test", "auroc": 0.45, "n_candidates": 10, "panel_count": 3}, + ] + r = _build(family_entry_dicts=entries) + assert r.family_entries[0].auroc_grade == "weak" + + +def test_build_auroc_grade_not_evaluated(): + entries = [ + {"family_name": "test", "auroc": -1.0, "n_candidates": 10, "panel_count": 0}, + ] + r = _build(family_entry_dicts=entries) + assert r.family_entries[0].auroc_grade == "not_evaluated" + + +def test_build_is_weak_class_true(): + entries = [ + {"family_name": "test", "auroc": 0.50, "n_candidates": 5, "panel_count": 1}, + ] + r = _build(family_entry_dicts=entries) + assert r.family_entries[0].is_weak_class is True + + +def test_build_is_weak_class_false_not_enough_candidates(): + entries = [ + {"family_name": "test", "auroc": 0.50, "n_candidates": 2, "panel_count": 0}, + ] + r = _build(family_entry_dicts=entries) + assert r.family_entries[0].is_weak_class is False + + +def test_build_limitations_and_created_at_stored(): + r = _build() + assert "dry-lab only" in r.limitations + assert r.created_at == "2026-07-10" + + +# --------------------------------------------------------------------------- +# Section 3: validate – rejection cases +# --------------------------------------------------------------------------- + + +def test_validate_rejects_bad_fbh_id_prefix(): + with pytest.raises(ValueError, match="FBH-"): + _build(fbh_id="BAD-001") + + +def test_validate_rejects_empty_pipeline_version(): + with pytest.raises(ValueError): + _build(pipeline_version="") + + +def test_validate_rejects_auroc_below_negative_one(): + entries = [ + {"family_name": "test", "auroc": -1.5, "n_candidates": 10, "panel_count": 3}, + ] + with pytest.raises(ValueError, match="auroc"): + _build(family_entry_dicts=entries) + + +def test_validate_rejects_auroc_above_one(): + entries = [ + {"family_name": "test", "auroc": 1.5, "n_candidates": 10, "panel_count": 3}, + ] + with pytest.raises(ValueError, match="auroc"): + _build(family_entry_dicts=entries) + + +def test_validate_rejects_n_candidates_zero(): + entries = [ + {"family_name": "test", "auroc": 0.5, "n_candidates": 0, "panel_count": 0}, + ] + with pytest.raises(ValueError, match="n_candidates"): + _build(family_entry_dicts=entries) + + +def test_validate_rejects_panel_count_exceeds_n_candidates(): + entries = [ + {"family_name": "test", "auroc": 0.5, "n_candidates": 5, "panel_count": 10}, + ] + with pytest.raises(ValueError, match="panel_count"): + _build(family_entry_dicts=entries) + + +def test_validate_rejects_negative_panel_count(): + entries = [ + {"family_name": "test", "auroc": 0.5, "n_candidates": 5, "panel_count": -1}, + ] + with pytest.raises(ValueError, match="panel_count"): + _build(family_entry_dicts=entries) + + +def test_validate_rejects_n_families_total_mismatch(): + r = _build() + r.n_families_total = 999 + with pytest.raises(ValueError, match="n_families_total"): + validate_family_blindness_challenge(r) + + +def test_validate_rejects_n_weak_families_mismatch(): + r = _build(family_entry_dicts=_WEAK_EXCLUDED) + r.n_weak_families = 999 + with pytest.raises(ValueError, match="n_weak_families"): + validate_family_blindness_challenge(r) + + +def test_validate_rejects_n_excluded_weak_families_mismatch(): + r = _build(family_entry_dicts=_WEAK_EXCLUDED) + r.n_excluded_weak_families = 999 + with pytest.raises(ValueError, match="n_excluded_weak_families"): + validate_family_blindness_challenge(r) + + +def test_validate_rejects_excluded_names_length_mismatch(): + r = _build(family_entry_dicts=_WEAK_EXCLUDED) + r.excluded_weak_family_names = [] + with pytest.raises(ValueError, match="excluded_weak_family_names"): + validate_family_blindness_challenge(r) + + +def test_validate_rejects_invalid_verdict(): + r = _build() + r.verdict = "UNKNOWN" + with pytest.raises(ValueError, match="verdict"): + validate_family_blindness_challenge(r) + + +def test_validate_rejects_weak_excluded_without_excluded(): + r = _build(family_entry_dicts=_ALL_REPRESENTED) + r.verdict = "weak_family_excluded" + with pytest.raises(ValueError, match="weak_family_excluded"): + validate_family_blindness_challenge(r) + + +def test_validate_rejects_all_represented_with_excluded(): + r = _build(family_entry_dicts=_WEAK_EXCLUDED) + r.verdict = "all_families_represented" + with pytest.raises(ValueError, match="all_families_represented"): + validate_family_blindness_challenge(r) + + +def test_validate_rejects_dry_lab_only_false(): + r = _build() + r.dry_lab_only = False + with pytest.raises(ValueError, match="dry_lab_only"): + validate_family_blindness_challenge(r) + + +def test_validate_rejects_empty_limitations(): + with pytest.raises(ValueError, match="limitations"): + _build(limitations=[]) + + +def test_validate_rejects_empty_created_at(): + with pytest.raises(ValueError): + _build(created_at="") + + +# --------------------------------------------------------------------------- +# Section 4: format +# --------------------------------------------------------------------------- + + +def test_format_contains_fbh_id(): + assert "FBH-001" in format_family_blindness_challenge(_build()) + + +def test_format_contains_pipeline_version(): + assert "v1.0" in format_family_blindness_challenge(_build()) + + +def test_format_contains_verdict(): + assert "all_families_represented" in format_family_blindness_challenge(_build()) + + +def test_format_contains_weak_family_names(): + r = _build(family_entry_dicts=_WEAK_EXCLUDED) + assert "bacteriocin" in format_family_blindness_challenge(r) + + +def test_format_contains_limitations(): + assert "dry-lab only" in format_family_blindness_challenge(_build()) + + +def test_format_contains_dry_lab_only(): + assert "dry_lab_only: True" in format_family_blindness_challenge(_build()) + + +def test_format_contains_n_families(): + text = format_family_blindness_challenge(_build()) + assert "2 total" in text + + +def test_format_is_string(): + assert isinstance(format_family_blindness_challenge(_build()), str) diff --git a/tests/test_test_count_regression.py b/tests/test_test_count_regression.py index 5b75ba7a..fe06f834 100644 --- a/tests/test_test_count_regression.py +++ b/tests/test_test_count_regression.py @@ -4,7 +4,7 @@ import sys import math -BASELINE = 11245 +BASELINE = 11299 def test_test_count_regression():