diff --git a/docs/research/NEXT_100_PR_MAP.md b/docs/research/NEXT_100_PR_MAP.md index 17fefeb5..d690f00d 100644 --- a/docs/research/NEXT_100_PR_MAP.md +++ b/docs/research/NEXT_100_PR_MAP.md @@ -281,3 +281,15 @@ Make it machine-verifiable that the pipeline produces novel candidates that beat | W3 | Add similarity challenge harness schema (SCH-) (complete). — src/openamp_foundry/evidence/similarity_challenge_harness.py: VALID_SCH_VERDICTS (4: selection_adds_value/marginal_improvement/proximity_driven/challenge_not_run), VALID_SIMILARITY_METRICS (4), SELECTION_VALUE_GAP_THRESHOLD=0.10, MARGINAL_IMPROVEMENT_LOWER=0.03; SimilarityGroupStats helper; similarity_gap auto-computed; dry_lab_only=True enforced; 60 tests in tests/evidence/test_similarity_challenge_harness.py. | Documents whether pipeline-selected candidates are systematically more similar to known AMPs than random selection from the sequence space; flags selection bias from similarity clustering; prevents "novel panel" claim when selection is proximity-driven. | C | | W4 | Add benchmark challenge registry schema (BCR-) (complete). — src/openamp_foundry/evidence/benchmark_challenge_registry.py: REQUIRED_CHALLENGE_TYPES (3: NCH/CMC/SCH), VALID_CHALLENGE_VERDICTS (4: pass/marginal/fail/not_run), VALID_BCR_HARDNESS_GRADES (A-D); ChallengeEntry helper; verdict mapping (novel_batch→pass, mixed_novelty→marginal, gap_meaningful→pass, etc.); grade A=all pass, B=all pass+marginal, C=some pass/marginal, D=all fail/not_run; dry_lab_only=True; 64 tests in tests/evidence/test_benchmark_challenge_registry.py. | Machine-readable registry of which benchmark challenges (NCH/CMC/SCH) have been run and passed for a given pipeline version; aggregates challenge verdicts; overall hardness grade (A: all passed, B: most passed, C: some passed, D: none passed). | C | | W5 | Add Phase W benchmark gate (WBG-) (complete). — src/openamp_foundry/evidence/phase_w_benchmark_gate.py: REQUIRED_W_COMPONENTS (4: NCH/CMC/SCH/BCR), VALID_WBG_VERDICTS (3: hardened/partially_hardened/not_hardened), HARDENED_REQUIRED_PRESENT=4, PARTIALLY_HARDENED_MIN_PRESENT=2; WComponentCheck helper; prefix validation for all artifact IDs; verdict: hardened=all 4 present, partially_hardened=2-3, not_hardened=0-1; dry_lab_only=True; 61 tests. Closes Phase W. | Top-level gate asserting NCH + CMC + SCH + BCR all present; overall verdict: hardened/partially_hardened/not_hardened; closes Phase W; no batch-level performance claim is credible without passing this gate. | C | + +## Phase X — Multi-batch learning loop + +Track whether the pipeline actually improves across batches by capturing per-batch quality snapshots and aggregating them into calibration progress records. Every improvement claim must reference machine-verifiable learning records. + +| ID | Task | Why it matters | Priority | +|----|------|----------------|----------| +| X1 | Add multi-batch learning record schema (MBL-) (complete). — src/openamp_foundry/evidence/multi_batch_learning_record.py: VALID_MBL_QUALITY_GRADES (5: A-D/N/A), VALID_BATCH_LEARNING_STATUSES (3), GRADE_A_HIT_RATE=0.40; hit_rate auto-computed from n_confirmed_hits/n_candidates_tested; grade auto-derived; no_wet_lab_data forces N/A grade; whr_ids list; dry_lab_only=True; 61 tests in tests/evidence/test_multi_batch_learning_record.py. | Per-batch snapshot of prediction quality (hit rate, AUROC, n_confirmed_hits) after wet-lab feedback; enables cross-batch comparison of whether the pipeline is learning; feeds into calibration improvement tracker. | C | +| X2 | Add calibration improvement tracker schema (CIT-). | Aggregates MBL records across batches; computes hit-rate trend direction (improving/stable/degrading/insufficient_data); minimum 2 batches required; flags when calibration is not producing measurable improvement. | C | +| X3 | Add learning progress report schema (LPR-). | Human-readable summary of what the pipeline has learned from all batches to date; references CIT- for trend data; includes which candidate features proved predictive vs not; links to calibration decision logs. | C | +| X4 | Add recalibration confidence certificate schema (RCC-). | Asserts with what confidence the current calibration weights are reliable based on cohort size, quality, and consistency across batches; A/B/C/D grade; prevents overconfident calibration claims. | C | +| X5 | Add Phase X learning gate (XLG-). | Top-level gate asserting MBL + CIT + LPR + RCC all present; overall verdict: learning_verified/learning_in_progress/learning_not_started; closes Phase X; no calibration improvement claim is credible without passing this gate. | C | diff --git a/src/openamp_foundry/evidence/multi_batch_learning_record.py b/src/openamp_foundry/evidence/multi_batch_learning_record.py new file mode 100644 index 00000000..d837ba43 --- /dev/null +++ b/src/openamp_foundry/evidence/multi_batch_learning_record.py @@ -0,0 +1,173 @@ +"""MBL- multi-batch learning record schema. + +Per-batch snapshot of prediction quality after wet-lab feedback. Captures +hit rate, AUROC, and n_confirmed for each batch, enabling cross-batch +comparison of whether the pipeline is improving. Feeds into the calibration +improvement tracker (CIT-). +""" + +from __future__ import annotations + +from dataclasses import dataclass + +VALID_MBL_QUALITY_GRADES: frozenset[str] = frozenset({ + "A", + "B", + "C", + "D", + "N/A", +}) + +VALID_BATCH_LEARNING_STATUSES: frozenset[str] = frozenset({ + "data_complete", + "data_partial", + "no_wet_lab_data", +}) + +MIN_AUROC: float = 0.0 +MAX_AUROC: float = 1.0 +MIN_HIT_RATE: float = 0.0 +MAX_HIT_RATE: float = 1.0 + +GRADE_A_HIT_RATE: float = 0.40 +GRADE_B_HIT_RATE: float = 0.25 +GRADE_C_HIT_RATE: float = 0.10 + + +@dataclass +class MultiBatchLearningRecord: + mbl_id: str + batch_id: str + pipeline_version: str + n_candidates_tested: int + n_confirmed_hits: int + hit_rate: float + auroc: float + quality_grade: str + batch_learning_status: str + whr_ids: list[str] + pcu_id: str + dry_lab_only: bool + limitations: list[str] + created_at: str + + +def validate_multi_batch_learning_record(mbl: MultiBatchLearningRecord) -> None: + if not mbl.mbl_id.startswith("MBL-"): + raise ValueError(f"mbl_id must start with 'MBL-': {mbl.mbl_id!r}") + if not mbl.batch_id: + raise ValueError("batch_id must be non-empty") + if not mbl.pipeline_version: + raise ValueError("pipeline_version must be non-empty") + if mbl.n_candidates_tested < 0: + raise ValueError("n_candidates_tested must be non-negative") + if mbl.n_confirmed_hits < 0: + raise ValueError("n_confirmed_hits must be non-negative") + if mbl.n_confirmed_hits > mbl.n_candidates_tested: + raise ValueError("n_confirmed_hits cannot exceed n_candidates_tested") + if mbl.batch_learning_status != "no_wet_lab_data": + if not (MIN_AUROC <= mbl.auroc <= MAX_AUROC): + raise ValueError(f"auroc must be in [0, 1]: {mbl.auroc}") + if not (MIN_HIT_RATE <= mbl.hit_rate <= MAX_HIT_RATE): + raise ValueError(f"hit_rate must be in [0, 1]: {mbl.hit_rate}") + if mbl.n_candidates_tested > 0: + expected_hr = round(mbl.n_confirmed_hits / mbl.n_candidates_tested, 6) + if abs(mbl.hit_rate - expected_hr) > 1e-4: + raise ValueError( + f"hit_rate {mbl.hit_rate} does not match computed " + f"{expected_hr}" + ) + if mbl.quality_grade not in VALID_MBL_QUALITY_GRADES: + raise ValueError( + f"quality_grade {mbl.quality_grade!r} not in VALID_MBL_QUALITY_GRADES" + ) + if mbl.batch_learning_status not in VALID_BATCH_LEARNING_STATUSES: + raise ValueError( + f"batch_learning_status {mbl.batch_learning_status!r} not in " + f"VALID_BATCH_LEARNING_STATUSES" + ) + if mbl.batch_learning_status == "no_wet_lab_data" and mbl.quality_grade != "N/A": + raise ValueError( + "quality_grade must be 'N/A' when batch_learning_status='no_wet_lab_data'" + ) + if not mbl.dry_lab_only: + raise ValueError("dry_lab_only must be True") + if not mbl.limitations: + raise ValueError("limitations must be non-empty") + if not mbl.created_at: + raise ValueError("created_at must be non-empty") + + +def _compute_quality_grade( + hit_rate: float, + n_tested: int, + status: str, +) -> str: + if status == "no_wet_lab_data" or n_tested == 0: + return "N/A" + if hit_rate >= GRADE_A_HIT_RATE: + return "A" + if hit_rate >= GRADE_B_HIT_RATE: + return "B" + if hit_rate >= GRADE_C_HIT_RATE: + return "C" + return "D" + + +def build_multi_batch_learning_record( + *, + mbl_id: str, + batch_id: str, + pipeline_version: str, + n_candidates_tested: int, + n_confirmed_hits: int, + auroc: float, + batch_learning_status: str, + whr_ids: list[str], + pcu_id: str = "", + limitations: list[str], + created_at: str, +) -> MultiBatchLearningRecord: + if n_candidates_tested > 0 and batch_learning_status != "no_wet_lab_data": + hit_rate = round(n_confirmed_hits / n_candidates_tested, 6) + else: + hit_rate = 0.0 + grade = _compute_quality_grade(hit_rate, n_candidates_tested, batch_learning_status) + mbl = MultiBatchLearningRecord( + mbl_id=mbl_id, + batch_id=batch_id, + pipeline_version=pipeline_version, + n_candidates_tested=n_candidates_tested, + n_confirmed_hits=n_confirmed_hits, + hit_rate=hit_rate, + auroc=auroc, + quality_grade=grade, + batch_learning_status=batch_learning_status, + whr_ids=list(whr_ids), + pcu_id=pcu_id, + dry_lab_only=True, + limitations=limitations, + created_at=created_at, + ) + validate_multi_batch_learning_record(mbl) + return mbl + + +def format_multi_batch_learning_record(mbl: MultiBatchLearningRecord) -> str: + lines = [ + f"Multi-Batch Learning Record — {mbl.mbl_id}", + f"Batch: {mbl.batch_id} | Pipeline: {mbl.pipeline_version}", + f"Status: {mbl.batch_learning_status} | Grade: {mbl.quality_grade}", + f"Candidates tested: {mbl.n_candidates_tested} | " + f"Confirmed hits: {mbl.n_confirmed_hits} | " + f"Hit rate: {mbl.hit_rate:.1%}", + f"AUROC: {mbl.auroc:.4f}", + ] + if mbl.whr_ids: + lines.append(f"WHR IDs: {', '.join(mbl.whr_ids)}") + if mbl.pcu_id: + lines.append(f"PCU: {mbl.pcu_id}") + lines.append(f"Created: {mbl.created_at}") + lines.append(f"Limitations: {'; '.join(mbl.limitations)}") + lines.append(f"dry_lab_only: {mbl.dry_lab_only}") + return "\n".join(lines) diff --git a/tests/evidence/test_multi_batch_learning_record.py b/tests/evidence/test_multi_batch_learning_record.py new file mode 100644 index 00000000..207c3291 --- /dev/null +++ b/tests/evidence/test_multi_batch_learning_record.py @@ -0,0 +1,341 @@ +"""Tests for MBL- multi-batch learning record schema.""" + +import pytest +from openamp_foundry.evidence.multi_batch_learning_record import ( + MultiBatchLearningRecord, + VALID_MBL_QUALITY_GRADES, + VALID_BATCH_LEARNING_STATUSES, + GRADE_A_HIT_RATE, + GRADE_B_HIT_RATE, + GRADE_C_HIT_RATE, + build_multi_batch_learning_record, + format_multi_batch_learning_record, + validate_multi_batch_learning_record, +) + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def _build(**kwargs): + defaults = dict( + mbl_id="MBL-001", + batch_id="BATCH-01", + pipeline_version="v1.0", + n_candidates_tested=20, + n_confirmed_hits=8, + auroc=0.78, + batch_learning_status="data_complete", + whr_ids=["WHR-001", "WHR-002"], + pcu_id="PCU-001", + limitations=["dry-lab only"], + created_at="2026-07-10", + ) + defaults.update(kwargs) + return build_multi_batch_learning_record(**defaults) + + +# --------------------------------------------------------------------------- +# 1. Constants +# --------------------------------------------------------------------------- + + +def test_valid_mbl_quality_grades_is_frozenset(): + assert isinstance(VALID_MBL_QUALITY_GRADES, frozenset) + + +def test_valid_mbl_quality_grades_contains_a(): + assert "A" in VALID_MBL_QUALITY_GRADES + + +def test_valid_mbl_quality_grades_contains_b(): + assert "B" in VALID_MBL_QUALITY_GRADES + + +def test_valid_mbl_quality_grades_contains_c(): + assert "C" in VALID_MBL_QUALITY_GRADES + + +def test_valid_mbl_quality_grades_contains_d(): + assert "D" in VALID_MBL_QUALITY_GRADES + + +def test_valid_mbl_quality_grades_contains_na(): + assert "N/A" in VALID_MBL_QUALITY_GRADES + + +def test_valid_batch_learning_statuses_is_frozenset(): + assert isinstance(VALID_BATCH_LEARNING_STATUSES, frozenset) + + +def test_valid_batch_learning_statuses_contains_data_complete(): + assert "data_complete" in VALID_BATCH_LEARNING_STATUSES + + +def test_valid_batch_learning_statuses_contains_data_partial(): + assert "data_partial" in VALID_BATCH_LEARNING_STATUSES + + +def test_valid_batch_learning_statuses_contains_no_wet_lab_data(): + assert "no_wet_lab_data" in VALID_BATCH_LEARNING_STATUSES + + +def test_grade_a_hit_rate(): + assert GRADE_A_HIT_RATE == 0.40 + + +def test_grade_b_hit_rate(): + assert GRADE_B_HIT_RATE == 0.25 + + +def test_grade_c_hit_rate(): + assert GRADE_C_HIT_RATE == 0.10 + + +# --------------------------------------------------------------------------- +# 2. build – happy paths +# --------------------------------------------------------------------------- + + +def test_build_returns_multi_batch_learning_record(): + assert isinstance(_build(), MultiBatchLearningRecord) + + +def test_build_mbl_id_stored(): + assert _build().mbl_id == "MBL-001" + + +def test_build_batch_id_stored(): + assert _build().batch_id == "BATCH-01" + + +def test_build_pipeline_version_stored(): + assert _build().pipeline_version == "v1.0" + + +def test_build_dry_lab_only_true(): + assert _build().dry_lab_only is True + + +def test_build_n_candidates_tested_stored(): + assert _build().n_candidates_tested == 20 + + +def test_build_n_confirmed_hits_stored(): + assert _build().n_confirmed_hits == 8 + + +def test_build_hit_rate_computed(): + r = _build() + assert abs(r.hit_rate - 8 / 20) < 1e-4 + + +def test_build_auroc_stored(): + assert abs(_build().auroc - 0.78) < 1e-9 + + +def test_build_grade_a_when_hit_rate_high(): + r = _build(n_candidates_tested=10, n_confirmed_hits=5) + assert r.quality_grade == "A" + + +def test_build_grade_b_when_hit_rate_medium(): + r = _build(n_candidates_tested=10, n_confirmed_hits=3) + assert r.quality_grade == "B" + + +def test_build_grade_c_when_hit_rate_low(): + r = _build(n_candidates_tested=10, n_confirmed_hits=1) + assert r.quality_grade == "C" + + +def test_build_grade_d_when_hit_rate_zero(): + r = _build(n_candidates_tested=10, n_confirmed_hits=0) + assert r.quality_grade == "D" + + +def test_build_grade_na_when_no_wet_lab_data(): + r = _build(n_candidates_tested=0, n_confirmed_hits=0, auroc=0.0, batch_learning_status="no_wet_lab_data") + assert r.quality_grade == "N/A" + + +def test_build_status_data_complete_stored(): + assert _build().batch_learning_status == "data_complete" + + +def test_build_status_data_partial(): + r = _build(batch_learning_status="data_partial") + assert r.batch_learning_status == "data_partial" + + +def test_build_no_wet_lab_data_hit_rate_zero(): + r = _build(n_candidates_tested=0, n_confirmed_hits=0, auroc=0.0, batch_learning_status="no_wet_lab_data") + assert r.hit_rate == 0.0 + + +def test_build_whr_ids_stored(): + assert _build().whr_ids == ["WHR-001", "WHR-002"] + + +def test_build_pcu_id_stored(): + assert _build().pcu_id == "PCU-001" + + +def test_build_pcu_id_defaults_empty(): + r = _build(pcu_id="") + assert r.pcu_id == "" + + +def test_build_whr_ids_empty_list(): + r = _build(whr_ids=[]) + assert r.whr_ids == [] + + +def test_build_limitations_stored(): + assert _build().limitations == ["dry-lab only"] + + +def test_build_created_at_stored(): + assert _build().created_at == "2026-07-10" + + +def test_build_grade_a_at_boundary(): + r = _build(n_candidates_tested=10, n_confirmed_hits=4) + assert r.quality_grade == "A" + + +def test_build_grade_b_at_boundary(): + r = _build(n_candidates_tested=20, n_confirmed_hits=5) + assert r.quality_grade == "B" + + +# --------------------------------------------------------------------------- +# 3. validate – rejection cases +# --------------------------------------------------------------------------- + + +def test_validate_rejects_bad_mbl_id_prefix(): + with pytest.raises(ValueError, match="MBL-"): + _build(mbl_id="BAD-001") + + +def test_validate_rejects_empty_batch_id(): + with pytest.raises(ValueError): + _build(batch_id="") + + +def test_validate_rejects_empty_pipeline_version(): + with pytest.raises(ValueError): + _build(pipeline_version="") + + +def test_validate_rejects_negative_n_candidates(): + mbl = _build() + mbl.n_candidates_tested = -1 + with pytest.raises(ValueError, match="n_candidates_tested"): + validate_multi_batch_learning_record(mbl) + + +def test_validate_rejects_negative_n_hits(): + mbl = _build() + mbl.n_confirmed_hits = -1 + with pytest.raises(ValueError, match="n_confirmed_hits"): + validate_multi_batch_learning_record(mbl) + + +def test_validate_rejects_hits_exceeding_tested(): + with pytest.raises(ValueError, match="n_confirmed_hits"): + _build(n_candidates_tested=5, n_confirmed_hits=6) + + +def test_validate_rejects_invalid_quality_grade(): + mbl = _build() + mbl.quality_grade = "X" + with pytest.raises(ValueError, match="quality_grade"): + validate_multi_batch_learning_record(mbl) + + +def test_validate_rejects_invalid_status(): + mbl = _build() + mbl.batch_learning_status = "UNKNOWN" + with pytest.raises(ValueError, match="batch_learning_status"): + validate_multi_batch_learning_record(mbl) + + +def test_validate_rejects_grade_not_na_when_no_data(): + mbl = _build(n_candidates_tested=0, n_confirmed_hits=0, auroc=0.0, batch_learning_status="no_wet_lab_data") + mbl.quality_grade = "A" + with pytest.raises(ValueError, match="N/A"): + validate_multi_batch_learning_record(mbl) + + +def test_validate_rejects_hit_rate_mismatch(): + mbl = _build() + mbl.hit_rate = 0.99 + with pytest.raises(ValueError, match="hit_rate"): + validate_multi_batch_learning_record(mbl) + + +def test_validate_rejects_dry_lab_only_false(): + mbl = _build() + mbl.dry_lab_only = False + with pytest.raises(ValueError, match="dry_lab_only"): + validate_multi_batch_learning_record(mbl) + + +def test_validate_rejects_empty_limitations(): + with pytest.raises(ValueError, match="limitations"): + _build(limitations=[]) + + +def test_validate_rejects_empty_created_at(): + with pytest.raises(ValueError): + _build(created_at="") + + +def test_validate_rejects_auroc_above_one(): + with pytest.raises(ValueError, match="auroc"): + _build(auroc=1.01) + + +def test_validate_rejects_auroc_below_zero(): + with pytest.raises(ValueError, match="auroc"): + _build(auroc=-0.01) + + +# --------------------------------------------------------------------------- +# 4. format +# --------------------------------------------------------------------------- + + +def test_format_contains_mbl_id(): + assert "MBL-001" in format_multi_batch_learning_record(_build()) + + +def test_format_contains_batch_id(): + assert "BATCH-01" in format_multi_batch_learning_record(_build()) + + +def test_format_contains_status(): + assert "data_complete" in format_multi_batch_learning_record(_build()) + + +def test_format_contains_grade(): + assert _build().quality_grade in format_multi_batch_learning_record(_build()) + + +def test_format_contains_n_confirmed_hits(): + assert "8" in format_multi_batch_learning_record(_build()) + + +def test_format_contains_limitations(): + assert "dry-lab only" in format_multi_batch_learning_record(_build()) + + +def test_format_contains_dry_lab_only(): + assert "dry_lab_only: True" in format_multi_batch_learning_record(_build()) + + +def test_format_is_string(): + assert isinstance(format_multi_batch_learning_record(_build()), str)