|
| 1 | +"""CMC- charge-matched challenge schema. |
| 2 | +
|
| 3 | +Documents the charge-matched challenge run for a batch: compares pipeline |
| 4 | +AUROC versus a charge-only baseline on the same candidate set with the same |
| 5 | +charge distribution. A meaningful gap is required to claim the pipeline adds |
| 6 | +value beyond simply preferring cationic sequences. |
| 7 | +""" |
| 8 | + |
| 9 | +from __future__ import annotations |
| 10 | + |
| 11 | +from dataclasses import dataclass |
| 12 | + |
| 13 | +VALID_CMC_VERDICTS: frozenset[str] = frozenset({ |
| 14 | + "gap_meaningful", |
| 15 | + "gap_marginal", |
| 16 | + "gap_absent", |
| 17 | + "challenge_not_run", |
| 18 | +}) |
| 19 | + |
| 20 | +VALID_CHARGE_BASELINE_METHODS: frozenset[str] = frozenset({ |
| 21 | + "charge_only_rank", |
| 22 | + "charge_length_rank", |
| 23 | + "charge_hydrophobicity_rank", |
| 24 | + "logistic_charge_only", |
| 25 | +}) |
| 26 | + |
| 27 | +MEANINGFUL_GAP_THRESHOLD: float = 0.05 |
| 28 | +MARGINAL_GAP_LOWER: float = 0.02 |
| 29 | + |
| 30 | +MIN_AUROC: float = 0.0 |
| 31 | +MAX_AUROC: float = 1.0 |
| 32 | + |
| 33 | + |
| 34 | +@dataclass |
| 35 | +class ChargeMatchedChallenge: |
| 36 | + cmc_id: str |
| 37 | + batch_id: str |
| 38 | + pipeline_version: str |
| 39 | + baseline_method: str |
| 40 | + pipeline_auroc: float |
| 41 | + baseline_auroc: float |
| 42 | + auroc_gap: float |
| 43 | + n_candidates: int |
| 44 | + mean_charge_pipeline: float |
| 45 | + mean_charge_baseline: float |
| 46 | + charge_distribution_matched: bool |
| 47 | + cmc_verdict: str |
| 48 | + dry_lab_only: bool |
| 49 | + limitations: list[str] |
| 50 | + created_at: str |
| 51 | + |
| 52 | + |
| 53 | +def validate_charge_matched_challenge(cmc: ChargeMatchedChallenge) -> None: |
| 54 | + if not cmc.cmc_id.startswith("CMC-"): |
| 55 | + raise ValueError(f"cmc_id must start with 'CMC-': {cmc.cmc_id!r}") |
| 56 | + if not cmc.batch_id: |
| 57 | + raise ValueError("batch_id must be non-empty") |
| 58 | + if not cmc.pipeline_version: |
| 59 | + raise ValueError("pipeline_version must be non-empty") |
| 60 | + if cmc.baseline_method not in VALID_CHARGE_BASELINE_METHODS: |
| 61 | + raise ValueError( |
| 62 | + f"baseline_method {cmc.baseline_method!r} not in VALID_CHARGE_BASELINE_METHODS" |
| 63 | + ) |
| 64 | + if not (MIN_AUROC <= cmc.pipeline_auroc <= MAX_AUROC): |
| 65 | + raise ValueError( |
| 66 | + f"pipeline_auroc must be in [0, 1]: {cmc.pipeline_auroc}" |
| 67 | + ) |
| 68 | + if not (MIN_AUROC <= cmc.baseline_auroc <= MAX_AUROC): |
| 69 | + raise ValueError( |
| 70 | + f"baseline_auroc must be in [0, 1]: {cmc.baseline_auroc}" |
| 71 | + ) |
| 72 | + expected_gap = round(cmc.pipeline_auroc - cmc.baseline_auroc, 6) |
| 73 | + if abs(cmc.auroc_gap - expected_gap) > 1e-4: |
| 74 | + raise ValueError( |
| 75 | + f"auroc_gap {cmc.auroc_gap} does not match computed " |
| 76 | + f"{expected_gap} (pipeline_auroc - baseline_auroc)" |
| 77 | + ) |
| 78 | + if cmc.n_candidates < 0: |
| 79 | + raise ValueError("n_candidates must be non-negative") |
| 80 | + if cmc.cmc_verdict not in VALID_CMC_VERDICTS: |
| 81 | + raise ValueError( |
| 82 | + f"cmc_verdict {cmc.cmc_verdict!r} not in VALID_CMC_VERDICTS" |
| 83 | + ) |
| 84 | + if not cmc.dry_lab_only: |
| 85 | + raise ValueError("dry_lab_only must be True") |
| 86 | + if not cmc.limitations: |
| 87 | + raise ValueError("limitations must be non-empty") |
| 88 | + if not cmc.created_at: |
| 89 | + raise ValueError("created_at must be non-empty") |
| 90 | + |
| 91 | + |
| 92 | +def _compute_verdict(n_candidates: int, auroc_gap: float) -> str: |
| 93 | + if n_candidates == 0: |
| 94 | + return "challenge_not_run" |
| 95 | + if auroc_gap >= MEANINGFUL_GAP_THRESHOLD: |
| 96 | + return "gap_meaningful" |
| 97 | + if auroc_gap >= MARGINAL_GAP_LOWER: |
| 98 | + return "gap_marginal" |
| 99 | + return "gap_absent" |
| 100 | + |
| 101 | + |
| 102 | +def build_charge_matched_challenge( |
| 103 | + *, |
| 104 | + cmc_id: str, |
| 105 | + batch_id: str, |
| 106 | + pipeline_version: str, |
| 107 | + baseline_method: str, |
| 108 | + pipeline_auroc: float, |
| 109 | + baseline_auroc: float, |
| 110 | + n_candidates: int, |
| 111 | + mean_charge_pipeline: float, |
| 112 | + mean_charge_baseline: float, |
| 113 | + charge_distribution_matched: bool, |
| 114 | + limitations: list[str], |
| 115 | + created_at: str, |
| 116 | +) -> ChargeMatchedChallenge: |
| 117 | + auroc_gap = round(pipeline_auroc - baseline_auroc, 6) |
| 118 | + verdict = _compute_verdict(n_candidates, auroc_gap) |
| 119 | + cmc = ChargeMatchedChallenge( |
| 120 | + cmc_id=cmc_id, |
| 121 | + batch_id=batch_id, |
| 122 | + pipeline_version=pipeline_version, |
| 123 | + baseline_method=baseline_method, |
| 124 | + pipeline_auroc=pipeline_auroc, |
| 125 | + baseline_auroc=baseline_auroc, |
| 126 | + auroc_gap=auroc_gap, |
| 127 | + n_candidates=n_candidates, |
| 128 | + mean_charge_pipeline=mean_charge_pipeline, |
| 129 | + mean_charge_baseline=mean_charge_baseline, |
| 130 | + charge_distribution_matched=charge_distribution_matched, |
| 131 | + cmc_verdict=verdict, |
| 132 | + dry_lab_only=True, |
| 133 | + limitations=limitations, |
| 134 | + created_at=created_at, |
| 135 | + ) |
| 136 | + validate_charge_matched_challenge(cmc) |
| 137 | + return cmc |
| 138 | + |
| 139 | + |
| 140 | +def format_charge_matched_challenge(cmc: ChargeMatchedChallenge) -> str: |
| 141 | + lines = [ |
| 142 | + f"Charge-Matched Challenge — {cmc.cmc_id}", |
| 143 | + f"Batch: {cmc.batch_id} | Pipeline: {cmc.pipeline_version}", |
| 144 | + f"Baseline method: {cmc.baseline_method}", |
| 145 | + f"Verdict: {cmc.cmc_verdict}", |
| 146 | + f"Pipeline AUROC: {cmc.pipeline_auroc:.4f} | Baseline AUROC: {cmc.baseline_auroc:.4f} | Gap: {cmc.auroc_gap:+.4f}", |
| 147 | + f"N candidates: {cmc.n_candidates}", |
| 148 | + f"Mean charge — pipeline: {cmc.mean_charge_pipeline:.2f} | baseline: {cmc.mean_charge_baseline:.2f}", |
| 149 | + f"Charge distribution matched: {cmc.charge_distribution_matched}", |
| 150 | + f"Created: {cmc.created_at}", |
| 151 | + f"Limitations: {'; '.join(cmc.limitations)}", |
| 152 | + f"dry_lab_only: {cmc.dry_lab_only}", |
| 153 | + ] |
| 154 | + return "\n".join(lines) |
0 commit comments