Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion docs/research/NEXT_100_PR_MAP.md
Original file line number Diff line number Diff line change
Expand Up @@ -278,6 +278,6 @@ Make it machine-verifiable that the pipeline produces novel candidates that beat
|----|------|----------------|----------|
| W1 | Add novelty challenge harness schema (NCH-) (complete). — src/openamp_foundry/evidence/novelty_challenge_harness.py: VALID_NCH_VERDICTS (4: novel_batch/mixed_novelty/near_neighbor_dominated/challenge_not_run), VALID_REFERENCE_DATABASES (6), NEAR_NEIGHBOR_IDENTITY_THRESHOLD=0.80, NOVEL_BATCH_CEILING=0.20, NEAR_NEIGHBOR_DOMINATED_FLOOR=0.60; NCHCandidateResult helper; build() auto-computes is_near_neighbor from identity vs threshold, fraction, verdict; dry_lab_only=True enforced; 63 tests in tests/evidence/test_novelty_challenge_harness.py. | Batch-level novelty challenge: documents the fraction of top candidates with ≥80% sequence identity to a known AMP in a reference database (APD3/DRAMP/etc.); blocks novel_batch claim when near-neighbor fraction exceeds 20%; prevents pipeline from advancing near-copies of known AMPs under a novelty label. | C |
| W2 | Add charge-matched challenge schema (CMC-) (complete). — src/openamp_foundry/evidence/charge_matched_challenge.py: VALID_CMC_VERDICTS (4: gap_meaningful/gap_marginal/gap_absent/challenge_not_run), VALID_CHARGE_BASELINE_METHODS (4), MEANINGFUL_GAP_THRESHOLD=0.05, MARGINAL_GAP_LOWER=0.02; auroc_gap auto-computed; verdict auto-derived; dry_lab_only=True enforced; 60 tests in tests/evidence/test_charge_matched_challenge.py. | Formally documents the charge-matched challenge: compares pipeline AUROC vs a charge-only baseline on the same candidate set; verdict controlled vocabulary (gap_meaningful/gap_marginal/gap_absent/not_run); blocks performance claims when the charge-only baseline explains the gap. | C |
| W3 | Add similarity challenge harness schema (SCH-). | Documents whether pipeline-selected candidates are systematically more similar to known AMPs than random selection from the sequence space; flags selection bias from similarity clustering; prevents "novel panel" claim when selection is proximity-driven. | C |
| W3 | Add similarity challenge harness schema (SCH-) (complete). — src/openamp_foundry/evidence/similarity_challenge_harness.py: VALID_SCH_VERDICTS (4: selection_adds_value/marginal_improvement/proximity_driven/challenge_not_run), VALID_SIMILARITY_METRICS (4), SELECTION_VALUE_GAP_THRESHOLD=0.10, MARGINAL_IMPROVEMENT_LOWER=0.03; SimilarityGroupStats helper; similarity_gap auto-computed; dry_lab_only=True enforced; 60 tests in tests/evidence/test_similarity_challenge_harness.py. | Documents whether pipeline-selected candidates are systematically more similar to known AMPs than random selection from the sequence space; flags selection bias from similarity clustering; prevents "novel panel" claim when selection is proximity-driven. | C |
| W4 | Add benchmark challenge registry schema (BCR-). | Machine-readable registry of which benchmark challenges (NCH/CMC/SCH) have been run and passed for a given pipeline version; aggregates challenge verdicts; overall hardness grade (A: all passed, B: most passed, C: some passed, D: none passed). | C |
| W5 | Add Phase W benchmark gate (WBG-). | Top-level gate asserting NCH + CMC + SCH + BCR all present; overall verdict: hardened/partially_hardened/not_hardened; closes Phase W; no batch-level performance claim is credible without passing this gate. | C |
167 changes: 167 additions & 0 deletions src/openamp_foundry/evidence/similarity_challenge_harness.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,167 @@
"""SCH- similarity challenge harness schema.

Documents whether pipeline-selected candidates are systematically more similar
to known AMPs than a random draw from the same sequence space. Flags selection
bias from proximity clustering: if the pipeline is essentially selecting
near-neighbors of known AMPs rather than exploring new space, novelty claims
are not credible.
"""

from __future__ import annotations

from dataclasses import dataclass

VALID_SCH_VERDICTS: frozenset[str] = frozenset({
"selection_adds_value",
"marginal_improvement",
"proximity_driven",
"challenge_not_run",
})

VALID_SIMILARITY_METRICS: frozenset[str] = frozenset({
"sequence_identity",
"blosum62_score",
"edit_distance",
"physicochemical_distance",
})

SELECTION_VALUE_GAP_THRESHOLD: float = 0.10
MARGINAL_IMPROVEMENT_LOWER: float = 0.03


@dataclass
class SimilarityGroupStats:
group_label: str
mean_similarity_to_known: float
n_sequences: int


@dataclass
class SimilarityChallengeHarness:
sch_id: str
batch_id: str
pipeline_version: str
similarity_metric: str
pipeline_group: SimilarityGroupStats
random_group: SimilarityGroupStats
similarity_gap: float
sch_verdict: str
dry_lab_only: bool
limitations: list[str]
created_at: str


def validate_similarity_challenge_harness(sch: SimilarityChallengeHarness) -> None:
if not sch.sch_id.startswith("SCH-"):
raise ValueError(f"sch_id must start with 'SCH-': {sch.sch_id!r}")
if not sch.batch_id:
raise ValueError("batch_id must be non-empty")
if not sch.pipeline_version:
raise ValueError("pipeline_version must be non-empty")
if sch.similarity_metric not in VALID_SIMILARITY_METRICS:
raise ValueError(
f"similarity_metric {sch.similarity_metric!r} not in VALID_SIMILARITY_METRICS"
)
for group in (sch.pipeline_group, sch.random_group):
if not (0.0 <= group.mean_similarity_to_known <= 1.0):
raise ValueError(
f"mean_similarity_to_known must be in [0, 1]: {group.mean_similarity_to_known}"
)
if group.n_sequences < 0:
raise ValueError(
f"n_sequences must be non-negative: {group.n_sequences}"
)
expected_gap = round(
sch.pipeline_group.mean_similarity_to_known
- sch.random_group.mean_similarity_to_known,
6,
)
if abs(sch.similarity_gap - expected_gap) > 1e-4:
raise ValueError(
f"similarity_gap {sch.similarity_gap} does not match computed "
f"{expected_gap}"
)
if sch.sch_verdict not in VALID_SCH_VERDICTS:
raise ValueError(
f"sch_verdict {sch.sch_verdict!r} not in VALID_SCH_VERDICTS"
)
if not sch.dry_lab_only:
raise ValueError("dry_lab_only must be True")
if not sch.limitations:
raise ValueError("limitations must be non-empty")
if not sch.created_at:
raise ValueError("created_at must be non-empty")


def _compute_verdict(
n_pipeline: int,
n_random: int,
similarity_gap: float,
) -> str:
if n_pipeline == 0 or n_random == 0:
return "challenge_not_run"
if similarity_gap >= SELECTION_VALUE_GAP_THRESHOLD:
return "selection_adds_value"
if similarity_gap >= MARGINAL_IMPROVEMENT_LOWER:
return "marginal_improvement"
return "proximity_driven"


def build_similarity_challenge_harness(
*,
sch_id: str,
batch_id: str,
pipeline_version: str,
similarity_metric: str,
pipeline_mean_similarity: float,
pipeline_n_sequences: int,
random_mean_similarity: float,
random_n_sequences: int,
limitations: list[str],
created_at: str,
) -> SimilarityChallengeHarness:
pipeline_group = SimilarityGroupStats(
group_label="pipeline_selected",
mean_similarity_to_known=pipeline_mean_similarity,
n_sequences=pipeline_n_sequences,
)
random_group = SimilarityGroupStats(
group_label="random_draw",
mean_similarity_to_known=random_mean_similarity,
n_sequences=random_n_sequences,
)
gap = round(pipeline_mean_similarity - random_mean_similarity, 6)
verdict = _compute_verdict(pipeline_n_sequences, random_n_sequences, gap)
sch = SimilarityChallengeHarness(
sch_id=sch_id,
batch_id=batch_id,
pipeline_version=pipeline_version,
similarity_metric=similarity_metric,
pipeline_group=pipeline_group,
random_group=random_group,
similarity_gap=gap,
sch_verdict=verdict,
dry_lab_only=True,
limitations=limitations,
created_at=created_at,
)
validate_similarity_challenge_harness(sch)
return sch


def format_similarity_challenge_harness(sch: SimilarityChallengeHarness) -> str:
lines = [
f"Similarity Challenge Harness — {sch.sch_id}",
f"Batch: {sch.batch_id} | Pipeline: {sch.pipeline_version}",
f"Similarity metric: {sch.similarity_metric}",
f"Verdict: {sch.sch_verdict}",
f"Pipeline selected: mean={sch.pipeline_group.mean_similarity_to_known:.4f} "
f"(n={sch.pipeline_group.n_sequences})",
f"Random draw: mean={sch.random_group.mean_similarity_to_known:.4f} "
f"(n={sch.random_group.n_sequences})",
f"Gap (pipeline - random): {sch.similarity_gap:+.4f}",
f"Created: {sch.created_at}",
f"Limitations: {'; '.join(sch.limitations)}",
f"dry_lab_only: {sch.dry_lab_only}",
]
return "\n".join(lines)
Loading
Loading