diff --git a/examples/benchmark/robustness_positives.csv b/examples/benchmark/robustness_positives.csv new file mode 100644 index 00000000..39ec9072 --- /dev/null +++ b/examples/benchmark/robustness_positives.csv @@ -0,0 +1,4 @@ +id,sequence,source +ROB-POS-001,KWKLFKKIGAVLKVL,robustness +ROB-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,robustness +ROB-POS-003,RRWQWRMKKLG,robustness diff --git a/examples/negative/poly_cationic.csv b/examples/negative/poly_cationic.csv new file mode 100644 index 00000000..5d58cab3 --- /dev/null +++ b/examples/negative/poly_cationic.csv @@ -0,0 +1,6 @@ +id,sequence,source +POLYK-001,KKKKKKKKKKKKK,poly_cationic +POLYR-001,RRRRRRRRRRRRR,poly_cationic +POLYKR-001,KRKRKRKRKRKRK,poly_cationic +POLYKR-002,KKRRKKKRRKKR,poly_cationic +POLYKR-003,RRKKRRKKRRKK,poly_cationic diff --git a/examples/negative/poly_hydrophobic.csv b/examples/negative/poly_hydrophobic.csv new file mode 100644 index 00000000..0dd84b58 --- /dev/null +++ b/examples/negative/poly_hydrophobic.csv @@ -0,0 +1,6 @@ +id,sequence,source +POLYH-001,LLLLLLLLLLLLL,poly_hydrophobic +POLYH-002,IIIIIIIIIIIII,poly_hydrophobic +POLYH-003,VVVVVVVVVVVVV,poly_hydrophobic +POLYH-004,LLIIVVLLIIVVL,poly_hydrophobic +POLYH-005,LLLLVVIIILLLL,poly_hydrophobic diff --git a/src/openamp_foundry/benchmark/evaluate.py b/src/openamp_foundry/benchmark/evaluate.py index 34c5ae0b..a2f3f42a 100644 --- a/src/openamp_foundry/benchmark/evaluate.py +++ b/src/openamp_foundry/benchmark/evaluate.py @@ -6,3 +6,93 @@ def top_k_ids(scored: list[ScoredCandidate], k: int) -> set[str]: ranked = sorted(scored, key=lambda x: x.scores.get("ensemble", 0.0), reverse=True) return {item.candidate.candidate_id for item in ranked[:k]} + + +def recall_at_k( + scored: list[ScoredCandidate], + positive_ids: set[str], + k: int, +) -> float: + """Fraction of known positives recovered in the top-k ranked candidates. + + Recall@k = |positives in top-k| / |total positives| + """ + if not positive_ids: + return 0.0 + top = top_k_ids(scored, k) + recovered = len(top & positive_ids) + return round(recovered / len(positive_ids), 4) + + +def random_recall_at_k(n_candidates: int, n_positives: int, k: int) -> float: + """Expected recall@k for a random ranker (sampling without replacement).""" + if n_positives == 0 or n_candidates == 0: + return 0.0 + expected_hits = min(k, n_positives) * min(k, n_candidates) / max(n_candidates, 1) + return round(min(1.0, expected_hits / n_positives), 4) + + +def enrichment_factor( + scored: list[ScoredCandidate], + positive_ids: set[str], + k: int, +) -> float: + """Enrichment factor at k: recall@k / random_recall@k. + + EF > 1.0 means the pipeline outperforms random selection. + EF = 1.0 is random. + EF < 1.0 is worse than random. + """ + n = len(scored) + n_pos = len(positive_ids) + rc = recall_at_k(scored, positive_ids, k) + random_rc = random_recall_at_k(n, n_pos, k) + if random_rc == 0.0: + return 0.0 + return round(rc / random_rc, 4) + + +def benchmark_summary( + scored: list[ScoredCandidate], + positive_ids: set[str], + ks: list[int] | None = None, +) -> dict: + """Produce a benchmark summary comparing pipeline vs random ranker. + + All results are computational only — they do not prove biological activity. + """ + if ks is None: + n = len(scored) + ks = sorted({max(1, n // 10), max(1, n // 5), max(1, n // 2), n}) + + results = [] + for k in ks: + rc = recall_at_k(scored, positive_ids, k) + rrc = random_recall_at_k(len(scored), len(positive_ids), k) + ef = enrichment_factor(scored, positive_ids, k) + results.append( + { + "k": k, + "recall_at_k": rc, + "random_recall_at_k": rrc, + "enrichment_factor": ef, + } + ) + + any_enrichment = any(r["enrichment_factor"] > 1.0 for r in results) + verdict = ( + "pipeline outperforms random" if any_enrichment else "pipeline does not outperform random" + ) + + return { + "disclaimer": ( + "These are retrospective benchmark results on demo data. " + "They do not prove biological efficacy. " + "They only measure whether the pipeline recovers known positives " + "better than a random ranker would." + ), + "n_candidates": len(scored), + "n_positives": len(positive_ids), + "results": results, + "verdict": verdict, + } diff --git a/src/openamp_foundry/pipeline.py b/src/openamp_foundry/pipeline.py index 9a0d0d14..ebd6a6bd 100644 --- a/src/openamp_foundry/pipeline.py +++ b/src/openamp_foundry/pipeline.py @@ -1,6 +1,5 @@ from __future__ import annotations -import hashlib import uuid from datetime import datetime, timezone from pathlib import Path diff --git a/tests/test_cli.py b/tests/test_cli.py index 02e1f390..06a689f8 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -3,7 +3,6 @@ import json -import pytest from openamp_foundry.cli import main @@ -45,7 +44,6 @@ def test_validate_command_success(tmp_path): "--out", out, "--cert-dir", certs, ]) - import os cert_files = list((tmp_path / "certs").glob("*.json")) assert cert_files, "No certificates were generated" ret = main([ diff --git a/tests/test_negative_robustness.py b/tests/test_negative_robustness.py new file mode 100644 index 00000000..3a82e987 --- /dev/null +++ b/tests/test_negative_robustness.py @@ -0,0 +1,218 @@ +"""Negative-set robustness tests — Phase 2 requirement. + +AGENTS.md: "Negative-set robustness — Performance remains meaningful across multiple +negative datasets." + +These tests verify that the pipeline enriches AMP-like positives over negatives +regardless of the type of negative control used: + + A. All-repeat / degenerate negatives (easy) — already tested elsewhere + B. Poly-cationic negatives (hard: high charge) — new + C. Poly-hydrophobic negatives (hard: high hydro) — new + +A pipeline that only works on "easy" negatives (all-A, all-G) cannot be trusted. +A pipeline that works across all three negative types is more credible. + +All scores are computational proxies. No biological activity is implied. +""" +from __future__ import annotations + +import csv +from pathlib import Path + +from openamp_foundry.benchmark.evaluate import enrichment_factor, recall_at_k +from openamp_foundry.features.physchem import compute_features +from openamp_foundry.pipeline import score_candidates +from openamp_foundry.scoring.activity import activity_likeness_score +from openamp_foundry.scoring.safety import safety_score + + +POSITIVES_CSV = "examples/benchmark/robustness_positives.csv" +NEGATIVE_A_CSV = "examples/negative/demo_negative_peptides.csv" +NEGATIVE_B_CSV = "examples/negative/poly_cationic.csv" +NEGATIVE_C_CSV = "examples/negative/poly_hydrophobic.csv" + +POSITIVE_IDS = {"ROB-POS-001", "ROB-POS-002", "ROB-POS-003"} + + +def _merge_csv_files(pos_csv: str, neg_csv: str, tmp_path: Path) -> Path: + """Combine a positive and negative CSV into a single candidate file.""" + out = tmp_path / "merged.csv" + rows = [] + for path in [pos_csv, neg_csv]: + with open(path) as f: + reader = csv.DictReader(f) + rows.extend(list(reader)) + with out.open("w", newline="") as f: + writer = csv.DictWriter(f, fieldnames=["id", "sequence", "source"]) + writer.writeheader() + writer.writerows(rows) + return out + + +class TestNegativeSetProperties: + """Verify the negative datasets have the expected physicochemical profiles.""" + + def test_poly_cationic_has_high_charge_low_hydrophobicity(self): + features = compute_features("KKKKKKKKKKKKK") + assert features["charge_density"] > 0.8, "Poly-K should have high charge density" + assert features["hydrophobic_fraction"] == 0.0, "Poly-K should have zero hydrophobic fraction" + + def test_poly_hydrophobic_has_high_hydro_zero_charge(self): + features = compute_features("LLLLLLLLLLLLL") + assert features["hydrophobic_fraction"] == 1.0, "Poly-L should have 100% hydrophobic fraction" + assert features["charge_density"] == 0.0, "Poly-L should have zero charge density" + + def test_poly_cationic_activity_score_below_real_amps(self): + amp_feat = compute_features("KWKLFKKIGAVLKVL") + polyk_feat = compute_features("KKKKKKKKKKKKK") + amp_score = activity_likeness_score(amp_feat) + polyk_score = activity_likeness_score(polyk_feat) + assert amp_score > polyk_score, ( + f"Real AMP activity ({amp_score:.3f}) should exceed poly-K ({polyk_score:.3f}): " + "poly-K has high charge but no hydrophobic face" + ) + + def test_poly_hydrophobic_activity_score_below_real_amps(self): + amp_feat = compute_features("KWKLFKKIGAVLKVL") + polyl_feat = compute_features("LLLLLLLLLLLLL") + amp_score = activity_likeness_score(amp_feat) + polyl_score = activity_likeness_score(polyl_feat) + assert amp_score > polyl_score, ( + f"Real AMP activity ({amp_score:.3f}) should exceed poly-L ({polyl_score:.3f}): " + "poly-L has no charge" + ) + + def test_poly_cationic_safety_penalized(self): + polyk_feat = compute_features("KKKKKKKKKKKKK") + polyk_safety = safety_score(polyk_feat) + # Very high charge density should trigger the safety risk penalty + assert polyk_safety < 0.7, ( + f"Poly-K safety={polyk_safety:.3f} should be <0.7 due to high charge density risk" + ) + + def test_poly_hydrophobic_safety_penalized(self): + polyl_feat = compute_features("LLLLLLLLLLLLL") + polyl_safety = safety_score(polyl_feat) + # Very high hydrophobic fraction triggers hemolysis proxy penalty + assert polyl_safety < 0.5, ( + f"Poly-L safety={polyl_safety:.3f} should be <0.5 due to extreme hydrophobicity" + ) + + def test_poly_repeat_long_run_detected(self): + polyk_feat = compute_features("KKKKKKKKKKKKK") + assert polyk_feat["longest_repeat_run"] == 13, "Poly-K should have repeat run of 13" + + def test_poly_cationic_file_exists(self): + assert Path(NEGATIVE_B_CSV).exists(), "Poly-cationic negative set CSV not found" + + def test_poly_hydrophobic_file_exists(self): + assert Path(NEGATIVE_C_CSV).exists(), "Poly-hydrophobic negative set CSV not found" + + +class TestEnrichmentAcrossNegativeSets: + """Verify EF > 1.0 when positives are mixed with each negative set.""" + + def test_enrichment_vs_degenerate_negatives(self, tmp_path): + """Baseline: positives should easily outrank all-repeat / degenerate negatives.""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_A_CSV, tmp_path) + scored, _ = score_candidates(merged) + k = len(POSITIVE_IDS) + ef = enrichment_factor(scored, POSITIVE_IDS, k=k) + assert ef > 1.0, f"EF={ef:.3f} should exceed 1.0 vs degenerate negatives" + + def test_enrichment_vs_poly_cationic_negatives(self, tmp_path): + """Harder test: positives outrank high-charge-only negatives (poly-K/R).""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_B_CSV, tmp_path) + scored, _ = score_candidates(merged) + k = len(POSITIVE_IDS) + ef = enrichment_factor(scored, POSITIVE_IDS, k=k) + assert ef > 1.0, ( + f"EF={ef:.3f} should exceed 1.0 vs poly-cationic negatives. " + "Pipeline must use BOTH charge and hydrophobicity, not just charge." + ) + + def test_enrichment_vs_poly_hydrophobic_negatives(self, tmp_path): + """Harder test: positives outrank hydrophobic-only negatives (poly-L/I/V).""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_C_CSV, tmp_path) + scored, _ = score_candidates(merged) + k = len(POSITIVE_IDS) + ef = enrichment_factor(scored, POSITIVE_IDS, k=k) + assert ef > 1.0, ( + f"EF={ef:.3f} should exceed 1.0 vs poly-hydrophobic negatives. " + "Pipeline must use BOTH charge and hydrophobicity, not just hydrophobicity." + ) + + def test_recall_at_k_vs_poly_cationic(self, tmp_path): + """recall@3 should be 1.0 (all 3 positives in top 3) vs poly-cationic negatives.""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_B_CSV, tmp_path) + scored, _ = score_candidates(merged) + rc = recall_at_k(scored, POSITIVE_IDS, k=3) + assert rc == 1.0, ( + f"recall@3={rc:.4f}: all 3 known positives should be in the top 3 " + "when mixed with 5 poly-cationic negatives" + ) + + def test_recall_at_k_vs_poly_hydrophobic(self, tmp_path): + """recall@3 should be 1.0 (all 3 positives in top 3) vs poly-hydrophobic negatives.""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_C_CSV, tmp_path) + scored, _ = score_candidates(merged) + rc = recall_at_k(scored, POSITIVE_IDS, k=3) + assert rc == 1.0, ( + f"recall@3={rc:.4f}: all 3 known positives should be in the top 3 " + "when mixed with 5 poly-hydrophobic negatives" + ) + + def test_ef_consistent_across_all_negative_sets(self, tmp_path): + """EF > 1.0 for all three negative set types — robustness check.""" + results = {} + for label, neg_csv in [ + ("degenerate", NEGATIVE_A_CSV), + ("poly_cationic", NEGATIVE_B_CSV), + ("poly_hydrophobic", NEGATIVE_C_CSV), + ]: + subdir = tmp_path / label + subdir.mkdir() + merged = _merge_csv_files(POSITIVES_CSV, neg_csv, subdir) + scored, _ = score_candidates(merged) + ef = enrichment_factor(scored, POSITIVE_IDS, k=len(POSITIVE_IDS)) + results[label] = ef + + failures = [ + f"{label} (EF={ef:.3f})" + for label, ef in results.items() + if ef <= 1.0 + ] + assert not failures, ( + f"EF ≤ 1.0 for negative set(s): {failures}. " + "Pipeline must beat random across all tested negative types." + ) + + def test_mean_positive_score_exceeds_mean_negative_across_all_sets(self, tmp_path): + """Mean ensemble score of positives > mean of negatives for all negative types.""" + for label, neg_csv in [ + ("degenerate", NEGATIVE_A_CSV), + ("poly_cationic", NEGATIVE_B_CSV), + ("poly_hydrophobic", NEGATIVE_C_CSV), + ]: + subdir = tmp_path / label + subdir.mkdir() + merged = _merge_csv_files(POSITIVES_CSV, neg_csv, subdir) + scored, _ = score_candidates(merged) + + pos_ensemble = [ + s.scores["ensemble"] + for s in scored + if s.candidate.candidate_id in POSITIVE_IDS + ] + neg_ensemble = [ + s.scores["ensemble"] + for s in scored + if s.candidate.candidate_id not in POSITIVE_IDS + ] + avg_pos = sum(pos_ensemble) / len(pos_ensemble) + avg_neg = sum(neg_ensemble) / len(neg_ensemble) + assert avg_pos > avg_neg, ( + f"[{label}] Mean positive ensemble ({avg_pos:.3f}) should exceed " + f"mean negative ensemble ({avg_neg:.3f})" + ) diff --git a/tests/test_pipeline_filters.py b/tests/test_pipeline_filters.py index a078063c..2a57b0d9 100644 --- a/tests/test_pipeline_filters.py +++ b/tests/test_pipeline_filters.py @@ -2,9 +2,7 @@ from __future__ import annotations import json -from pathlib import Path -import pytest from openamp_foundry.pipeline import _passes_length_filter, run_ranking_pipeline, score_candidates