diff --git a/docs/research/NEXT_100_PR_MAP.md b/docs/research/NEXT_100_PR_MAP.md index 700ecc6c..7c135fd7 100644 --- a/docs/research/NEXT_100_PR_MAP.md +++ b/docs/research/NEXT_100_PR_MAP.md @@ -277,7 +277,7 @@ Make it machine-verifiable that the pipeline produces novel candidates that beat | ID | Task | Why it matters | Priority | |----|------|----------------|----------| | W1 | Add novelty challenge harness schema (NCH-) (complete). — src/openamp_foundry/evidence/novelty_challenge_harness.py: VALID_NCH_VERDICTS (4: novel_batch/mixed_novelty/near_neighbor_dominated/challenge_not_run), VALID_REFERENCE_DATABASES (6), NEAR_NEIGHBOR_IDENTITY_THRESHOLD=0.80, NOVEL_BATCH_CEILING=0.20, NEAR_NEIGHBOR_DOMINATED_FLOOR=0.60; NCHCandidateResult helper; build() auto-computes is_near_neighbor from identity vs threshold, fraction, verdict; dry_lab_only=True enforced; 63 tests in tests/evidence/test_novelty_challenge_harness.py. | Batch-level novelty challenge: documents the fraction of top candidates with ≥80% sequence identity to a known AMP in a reference database (APD3/DRAMP/etc.); blocks novel_batch claim when near-neighbor fraction exceeds 20%; prevents pipeline from advancing near-copies of known AMPs under a novelty label. | C | -| W2 | Add charge-matched challenge schema (CMC-). | Formally documents the charge-matched challenge: compares pipeline AUROC vs a charge-only baseline on the same candidate set; verdict controlled vocabulary (gap_meaningful/gap_marginal/gap_absent/not_run); blocks performance claims when the charge-only baseline explains the gap. | C | +| W2 | Add charge-matched challenge schema (CMC-) (complete). — src/openamp_foundry/evidence/charge_matched_challenge.py: VALID_CMC_VERDICTS (4: gap_meaningful/gap_marginal/gap_absent/challenge_not_run), VALID_CHARGE_BASELINE_METHODS (4), MEANINGFUL_GAP_THRESHOLD=0.05, MARGINAL_GAP_LOWER=0.02; auroc_gap auto-computed; verdict auto-derived; dry_lab_only=True enforced; 60 tests in tests/evidence/test_charge_matched_challenge.py. | Formally documents the charge-matched challenge: compares pipeline AUROC vs a charge-only baseline on the same candidate set; verdict controlled vocabulary (gap_meaningful/gap_marginal/gap_absent/not_run); blocks performance claims when the charge-only baseline explains the gap. | C | | W3 | Add similarity challenge harness schema (SCH-). | Documents whether pipeline-selected candidates are systematically more similar to known AMPs than random selection from the sequence space; flags selection bias from similarity clustering; prevents "novel panel" claim when selection is proximity-driven. | C | | W4 | Add benchmark challenge registry schema (BCR-). | Machine-readable registry of which benchmark challenges (NCH/CMC/SCH) have been run and passed for a given pipeline version; aggregates challenge verdicts; overall hardness grade (A: all passed, B: most passed, C: some passed, D: none passed). | C | | W5 | Add Phase W benchmark gate (WBG-). | Top-level gate asserting NCH + CMC + SCH + BCR all present; overall verdict: hardened/partially_hardened/not_hardened; closes Phase W; no batch-level performance claim is credible without passing this gate. | C | diff --git a/src/openamp_foundry/evidence/charge_matched_challenge.py b/src/openamp_foundry/evidence/charge_matched_challenge.py new file mode 100644 index 00000000..7461c302 --- /dev/null +++ b/src/openamp_foundry/evidence/charge_matched_challenge.py @@ -0,0 +1,154 @@ +"""CMC- charge-matched challenge schema. + +Documents the charge-matched challenge run for a batch: compares pipeline +AUROC versus a charge-only baseline on the same candidate set with the same +charge distribution. A meaningful gap is required to claim the pipeline adds +value beyond simply preferring cationic sequences. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +VALID_CMC_VERDICTS: frozenset[str] = frozenset({ + "gap_meaningful", + "gap_marginal", + "gap_absent", + "challenge_not_run", +}) + +VALID_CHARGE_BASELINE_METHODS: frozenset[str] = frozenset({ + "charge_only_rank", + "charge_length_rank", + "charge_hydrophobicity_rank", + "logistic_charge_only", +}) + +MEANINGFUL_GAP_THRESHOLD: float = 0.05 +MARGINAL_GAP_LOWER: float = 0.02 + +MIN_AUROC: float = 0.0 +MAX_AUROC: float = 1.0 + + +@dataclass +class ChargeMatchedChallenge: + cmc_id: str + batch_id: str + pipeline_version: str + baseline_method: str + pipeline_auroc: float + baseline_auroc: float + auroc_gap: float + n_candidates: int + mean_charge_pipeline: float + mean_charge_baseline: float + charge_distribution_matched: bool + cmc_verdict: str + dry_lab_only: bool + limitations: list[str] + created_at: str + + +def validate_charge_matched_challenge(cmc: ChargeMatchedChallenge) -> None: + if not cmc.cmc_id.startswith("CMC-"): + raise ValueError(f"cmc_id must start with 'CMC-': {cmc.cmc_id!r}") + if not cmc.batch_id: + raise ValueError("batch_id must be non-empty") + if not cmc.pipeline_version: + raise ValueError("pipeline_version must be non-empty") + if cmc.baseline_method not in VALID_CHARGE_BASELINE_METHODS: + raise ValueError( + f"baseline_method {cmc.baseline_method!r} not in VALID_CHARGE_BASELINE_METHODS" + ) + if not (MIN_AUROC <= cmc.pipeline_auroc <= MAX_AUROC): + raise ValueError( + f"pipeline_auroc must be in [0, 1]: {cmc.pipeline_auroc}" + ) + if not (MIN_AUROC <= cmc.baseline_auroc <= MAX_AUROC): + raise ValueError( + f"baseline_auroc must be in [0, 1]: {cmc.baseline_auroc}" + ) + expected_gap = round(cmc.pipeline_auroc - cmc.baseline_auroc, 6) + if abs(cmc.auroc_gap - expected_gap) > 1e-4: + raise ValueError( + f"auroc_gap {cmc.auroc_gap} does not match computed " + f"{expected_gap} (pipeline_auroc - baseline_auroc)" + ) + if cmc.n_candidates < 0: + raise ValueError("n_candidates must be non-negative") + if cmc.cmc_verdict not in VALID_CMC_VERDICTS: + raise ValueError( + f"cmc_verdict {cmc.cmc_verdict!r} not in VALID_CMC_VERDICTS" + ) + if not cmc.dry_lab_only: + raise ValueError("dry_lab_only must be True") + if not cmc.limitations: + raise ValueError("limitations must be non-empty") + if not cmc.created_at: + raise ValueError("created_at must be non-empty") + + +def _compute_verdict(n_candidates: int, auroc_gap: float) -> str: + if n_candidates == 0: + return "challenge_not_run" + if auroc_gap >= MEANINGFUL_GAP_THRESHOLD: + return "gap_meaningful" + if auroc_gap >= MARGINAL_GAP_LOWER: + return "gap_marginal" + return "gap_absent" + + +def build_charge_matched_challenge( + *, + cmc_id: str, + batch_id: str, + pipeline_version: str, + baseline_method: str, + pipeline_auroc: float, + baseline_auroc: float, + n_candidates: int, + mean_charge_pipeline: float, + mean_charge_baseline: float, + charge_distribution_matched: bool, + limitations: list[str], + created_at: str, +) -> ChargeMatchedChallenge: + auroc_gap = round(pipeline_auroc - baseline_auroc, 6) + verdict = _compute_verdict(n_candidates, auroc_gap) + cmc = ChargeMatchedChallenge( + cmc_id=cmc_id, + batch_id=batch_id, + pipeline_version=pipeline_version, + baseline_method=baseline_method, + pipeline_auroc=pipeline_auroc, + baseline_auroc=baseline_auroc, + auroc_gap=auroc_gap, + n_candidates=n_candidates, + mean_charge_pipeline=mean_charge_pipeline, + mean_charge_baseline=mean_charge_baseline, + charge_distribution_matched=charge_distribution_matched, + cmc_verdict=verdict, + dry_lab_only=True, + limitations=limitations, + created_at=created_at, + ) + validate_charge_matched_challenge(cmc) + return cmc + + +def format_charge_matched_challenge(cmc: ChargeMatchedChallenge) -> str: + lines = [ + f"Charge-Matched Challenge — {cmc.cmc_id}", + f"Batch: {cmc.batch_id} | Pipeline: {cmc.pipeline_version}", + f"Baseline method: {cmc.baseline_method}", + f"Verdict: {cmc.cmc_verdict}", + f"Pipeline AUROC: {cmc.pipeline_auroc:.4f} | Baseline AUROC: {cmc.baseline_auroc:.4f} | Gap: {cmc.auroc_gap:+.4f}", + f"N candidates: {cmc.n_candidates}", + f"Mean charge — pipeline: {cmc.mean_charge_pipeline:.2f} | baseline: {cmc.mean_charge_baseline:.2f}", + f"Charge distribution matched: {cmc.charge_distribution_matched}", + f"Created: {cmc.created_at}", + f"Limitations: {'; '.join(cmc.limitations)}", + f"dry_lab_only: {cmc.dry_lab_only}", + ] + return "\n".join(lines) diff --git a/tests/evidence/test_charge_matched_challenge.py b/tests/evidence/test_charge_matched_challenge.py new file mode 100644 index 00000000..31d57dc9 --- /dev/null +++ b/tests/evidence/test_charge_matched_challenge.py @@ -0,0 +1,331 @@ +"""Tests for CMC- charge-matched challenge schema.""" + +import pytest +from openamp_foundry.evidence.charge_matched_challenge import ( + ChargeMatchedChallenge, + VALID_CMC_VERDICTS, + VALID_CHARGE_BASELINE_METHODS, + MEANINGFUL_GAP_THRESHOLD, + MARGINAL_GAP_LOWER, + build_charge_matched_challenge, + format_charge_matched_challenge, + validate_charge_matched_challenge, +) + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def _build(**kwargs): + defaults = dict( + cmc_id="CMC-001", + batch_id="BATCH-01", + pipeline_version="v1.0", + baseline_method="charge_only_rank", + pipeline_auroc=0.82, + baseline_auroc=0.71, + n_candidates=100, + mean_charge_pipeline=4.2, + mean_charge_baseline=4.1, + charge_distribution_matched=True, + limitations=["dry-lab only", "AUROC is approximate"], + created_at="2026-07-10", + ) + defaults.update(kwargs) + return build_charge_matched_challenge(**defaults) + + +# --------------------------------------------------------------------------- +# 1. Constants +# --------------------------------------------------------------------------- + + +def test_valid_cmc_verdicts_is_frozenset(): + assert isinstance(VALID_CMC_VERDICTS, frozenset) + + +def test_valid_cmc_verdicts_contains_gap_meaningful(): + assert "gap_meaningful" in VALID_CMC_VERDICTS + + +def test_valid_cmc_verdicts_contains_gap_marginal(): + assert "gap_marginal" in VALID_CMC_VERDICTS + + +def test_valid_cmc_verdicts_contains_gap_absent(): + assert "gap_absent" in VALID_CMC_VERDICTS + + +def test_valid_cmc_verdicts_contains_challenge_not_run(): + assert "challenge_not_run" in VALID_CMC_VERDICTS + + +def test_valid_charge_baseline_methods_is_frozenset(): + assert isinstance(VALID_CHARGE_BASELINE_METHODS, frozenset) + + +def test_valid_charge_baseline_methods_contains_charge_only(): + assert "charge_only_rank" in VALID_CHARGE_BASELINE_METHODS + + +def test_valid_charge_baseline_methods_contains_charge_length(): + assert "charge_length_rank" in VALID_CHARGE_BASELINE_METHODS + + +def test_valid_charge_baseline_methods_contains_logistic(): + assert "logistic_charge_only" in VALID_CHARGE_BASELINE_METHODS + + +def test_meaningful_gap_threshold(): + assert MEANINGFUL_GAP_THRESHOLD == 0.05 + + +def test_marginal_gap_lower(): + assert MARGINAL_GAP_LOWER == 0.02 + + +# --------------------------------------------------------------------------- +# 2. build – happy paths +# --------------------------------------------------------------------------- + + +def test_build_returns_charge_matched_challenge(): + assert isinstance(_build(), ChargeMatchedChallenge) + + +def test_build_cmc_id_stored(): + assert _build().cmc_id == "CMC-001" + + +def test_build_batch_id_stored(): + assert _build().batch_id == "BATCH-01" + + +def test_build_pipeline_version_stored(): + assert _build().pipeline_version == "v1.0" + + +def test_build_dry_lab_only_true(): + assert _build().dry_lab_only is True + + +def test_build_baseline_method_stored(): + assert _build().baseline_method == "charge_only_rank" + + +def test_build_pipeline_auroc_stored(): + assert abs(_build().pipeline_auroc - 0.82) < 1e-9 + + +def test_build_baseline_auroc_stored(): + assert abs(_build().baseline_auroc - 0.71) < 1e-9 + + +def test_build_auroc_gap_computed(): + r = _build() + assert abs(r.auroc_gap - (0.82 - 0.71)) < 1e-4 + + +def test_build_gap_meaningful_verdict(): + r = _build(pipeline_auroc=0.82, baseline_auroc=0.71) + assert r.cmc_verdict == "gap_meaningful" + + +def test_build_gap_marginal_verdict(): + r = _build(pipeline_auroc=0.73, baseline_auroc=0.71) + assert r.cmc_verdict == "gap_marginal" + + +def test_build_gap_absent_verdict(): + r = _build(pipeline_auroc=0.71, baseline_auroc=0.71) + assert r.cmc_verdict == "gap_absent" + + +def test_build_challenge_not_run_when_n_candidates_zero(): + r = _build(n_candidates=0) + assert r.cmc_verdict == "challenge_not_run" + + +def test_build_gap_absent_when_baseline_higher(): + r = _build(pipeline_auroc=0.70, baseline_auroc=0.75) + assert r.cmc_verdict == "gap_absent" + assert r.auroc_gap < 0 + + +def test_build_n_candidates_stored(): + assert _build().n_candidates == 100 + + +def test_build_mean_charge_pipeline_stored(): + assert abs(_build().mean_charge_pipeline - 4.2) < 1e-9 + + +def test_build_mean_charge_baseline_stored(): + assert abs(_build().mean_charge_baseline - 4.1) < 1e-9 + + +def test_build_charge_distribution_matched_stored(): + assert _build().charge_distribution_matched is True + + +def test_build_charge_distribution_not_matched(): + r = _build(charge_distribution_matched=False) + assert r.charge_distribution_matched is False + + +def test_build_charge_length_rank_method(): + r = _build(baseline_method="charge_length_rank") + assert r.baseline_method == "charge_length_rank" + + +def test_build_charge_hydrophobicity_method(): + r = _build(baseline_method="charge_hydrophobicity_rank") + assert r.baseline_method == "charge_hydrophobicity_rank" + + +def test_build_logistic_charge_only_method(): + r = _build(baseline_method="logistic_charge_only") + assert r.baseline_method == "logistic_charge_only" + + +def test_build_limitations_stored(): + assert "dry-lab only" in _build().limitations + + +def test_build_created_at_stored(): + assert _build().created_at == "2026-07-10" + + +def test_build_auroc_gap_boundary_meaningful(): + r = _build(pipeline_auroc=0.76, baseline_auroc=0.71) + assert r.cmc_verdict == "gap_meaningful" + + +def test_build_auroc_gap_boundary_marginal_at_lower(): + r = _build(pipeline_auroc=0.73, baseline_auroc=0.71) + assert r.cmc_verdict == "gap_marginal" + + +# --------------------------------------------------------------------------- +# 3. validate – rejection cases +# --------------------------------------------------------------------------- + + +def test_validate_rejects_bad_cmc_id_prefix(): + with pytest.raises(ValueError, match="CMC-"): + _build(cmc_id="BAD-001") + + +def test_validate_rejects_empty_batch_id(): + with pytest.raises(ValueError): + _build(batch_id="") + + +def test_validate_rejects_empty_pipeline_version(): + with pytest.raises(ValueError): + _build(pipeline_version="") + + +def test_validate_rejects_invalid_baseline_method(): + with pytest.raises(ValueError, match="VALID_CHARGE_BASELINE_METHODS"): + _build(baseline_method="UNKNOWN_METHOD") + + +def test_validate_rejects_pipeline_auroc_below_zero(): + with pytest.raises(ValueError, match="pipeline_auroc"): + _build(pipeline_auroc=-0.01) + + +def test_validate_rejects_pipeline_auroc_above_one(): + with pytest.raises(ValueError, match="pipeline_auroc"): + _build(pipeline_auroc=1.01) + + +def test_validate_rejects_baseline_auroc_below_zero(): + with pytest.raises(ValueError, match="baseline_auroc"): + _build(baseline_auroc=-0.01) + + +def test_validate_rejects_baseline_auroc_above_one(): + with pytest.raises(ValueError, match="baseline_auroc"): + _build(baseline_auroc=1.01) + + +def test_validate_rejects_auroc_gap_mismatch(): + cmc = _build() + cmc.auroc_gap = 0.99 + with pytest.raises(ValueError, match="auroc_gap"): + validate_charge_matched_challenge(cmc) + + +def test_validate_rejects_negative_n_candidates(): + with pytest.raises(ValueError, match="n_candidates"): + cmc = _build() + cmc.n_candidates = -1 + validate_charge_matched_challenge(cmc) + + +def test_validate_rejects_invalid_verdict(): + cmc = _build() + cmc.cmc_verdict = "UNKNOWN" + with pytest.raises(ValueError, match="cmc_verdict"): + validate_charge_matched_challenge(cmc) + + +def test_validate_rejects_dry_lab_only_false(): + cmc = _build() + cmc.dry_lab_only = False + with pytest.raises(ValueError, match="dry_lab_only"): + validate_charge_matched_challenge(cmc) + + +def test_validate_rejects_empty_limitations(): + with pytest.raises(ValueError, match="limitations"): + _build(limitations=[]) + + +def test_validate_rejects_empty_created_at(): + with pytest.raises(ValueError): + _build(created_at="") + + +# --------------------------------------------------------------------------- +# 4. format +# --------------------------------------------------------------------------- + + +def test_format_contains_cmc_id(): + assert "CMC-001" in format_charge_matched_challenge(_build()) + + +def test_format_contains_batch_id(): + assert "BATCH-01" in format_charge_matched_challenge(_build()) + + +def test_format_contains_baseline_method(): + assert "charge_only_rank" in format_charge_matched_challenge(_build()) + + +def test_format_contains_verdict(): + assert "gap_meaningful" in format_charge_matched_challenge(_build()) + + +def test_format_contains_pipeline_auroc(): + assert "0.8200" in format_charge_matched_challenge(_build()) + + +def test_format_contains_baseline_auroc(): + assert "0.7100" in format_charge_matched_challenge(_build()) + + +def test_format_contains_limitations(): + assert "dry-lab only" in format_charge_matched_challenge(_build()) + + +def test_format_contains_dry_lab_only(): + assert "dry_lab_only: True" in format_charge_matched_challenge(_build()) + + +def test_format_is_string(): + assert isinstance(format_charge_matched_challenge(_build()), str)