Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions examples/benchmark/robustness_positives.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,4 @@
id,sequence,source
ROB-POS-001,KWKLFKKIGAVLKVL,robustness
ROB-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,robustness
ROB-POS-003,RRWQWRMKKLG,robustness
6 changes: 6 additions & 0 deletions examples/negative/poly_cationic.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
id,sequence,source
POLYK-001,KKKKKKKKKKKKK,poly_cationic
POLYR-001,RRRRRRRRRRRRR,poly_cationic
POLYKR-001,KRKRKRKRKRKRK,poly_cationic
POLYKR-002,KKRRKKKRRKKR,poly_cationic
POLYKR-003,RRKKRRKKRRKK,poly_cationic
6 changes: 6 additions & 0 deletions examples/negative/poly_hydrophobic.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
id,sequence,source
POLYH-001,LLLLLLLLLLLLL,poly_hydrophobic
POLYH-002,IIIIIIIIIIIII,poly_hydrophobic
POLYH-003,VVVVVVVVVVVVV,poly_hydrophobic
POLYH-004,LLIIVVLLIIVVL,poly_hydrophobic
POLYH-005,LLLLVVIIILLLL,poly_hydrophobic
90 changes: 90 additions & 0 deletions src/openamp_foundry/benchmark/evaluate.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,3 +6,93 @@
def top_k_ids(scored: list[ScoredCandidate], k: int) -> set[str]:
ranked = sorted(scored, key=lambda x: x.scores.get("ensemble", 0.0), reverse=True)
return {item.candidate.candidate_id for item in ranked[:k]}


def recall_at_k(
scored: list[ScoredCandidate],
positive_ids: set[str],
k: int,
) -> float:
"""Fraction of known positives recovered in the top-k ranked candidates.

Recall@k = |positives in top-k| / |total positives|
"""
if not positive_ids:
return 0.0
top = top_k_ids(scored, k)
recovered = len(top & positive_ids)
return round(recovered / len(positive_ids), 4)


def random_recall_at_k(n_candidates: int, n_positives: int, k: int) -> float:
"""Expected recall@k for a random ranker (sampling without replacement)."""
if n_positives == 0 or n_candidates == 0:
return 0.0
expected_hits = min(k, n_positives) * min(k, n_candidates) / max(n_candidates, 1)
return round(min(1.0, expected_hits / n_positives), 4)


def enrichment_factor(
scored: list[ScoredCandidate],
positive_ids: set[str],
k: int,
) -> float:
"""Enrichment factor at k: recall@k / random_recall@k.

EF > 1.0 means the pipeline outperforms random selection.
EF = 1.0 is random.
EF < 1.0 is worse than random.
"""
n = len(scored)
n_pos = len(positive_ids)
rc = recall_at_k(scored, positive_ids, k)
random_rc = random_recall_at_k(n, n_pos, k)
if random_rc == 0.0:
return 0.0
return round(rc / random_rc, 4)


def benchmark_summary(
scored: list[ScoredCandidate],
positive_ids: set[str],
ks: list[int] | None = None,
) -> dict:
"""Produce a benchmark summary comparing pipeline vs random ranker.

All results are computational only — they do not prove biological activity.
"""
if ks is None:
n = len(scored)
ks = sorted({max(1, n // 10), max(1, n // 5), max(1, n // 2), n})

results = []
for k in ks:
rc = recall_at_k(scored, positive_ids, k)
rrc = random_recall_at_k(len(scored), len(positive_ids), k)
ef = enrichment_factor(scored, positive_ids, k)
results.append(
{
"k": k,
"recall_at_k": rc,
"random_recall_at_k": rrc,
"enrichment_factor": ef,
}
)

any_enrichment = any(r["enrichment_factor"] > 1.0 for r in results)
verdict = (
"pipeline outperforms random" if any_enrichment else "pipeline does not outperform random"
)

return {
"disclaimer": (
"These are retrospective benchmark results on demo data. "
"They do not prove biological efficacy. "
"They only measure whether the pipeline recovers known positives "
"better than a random ranker would."
),
"n_candidates": len(scored),
"n_positives": len(positive_ids),
"results": results,
"verdict": verdict,
}
1 change: 0 additions & 1 deletion src/openamp_foundry/pipeline.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,5 @@
from __future__ import annotations

import hashlib
import uuid
from datetime import datetime, timezone
from pathlib import Path
Expand Down
2 changes: 0 additions & 2 deletions tests/test_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@

import json

import pytest

from openamp_foundry.cli import main

Expand Down Expand Up @@ -45,7 +44,6 @@ def test_validate_command_success(tmp_path):
"--out", out,
"--cert-dir", certs,
])
import os
cert_files = list((tmp_path / "certs").glob("*.json"))
assert cert_files, "No certificates were generated"
ret = main([
Expand Down
218 changes: 218 additions & 0 deletions tests/test_negative_robustness.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,218 @@
"""Negative-set robustness tests — Phase 2 requirement.

AGENTS.md: "Negative-set robustness — Performance remains meaningful across multiple
negative datasets."

These tests verify that the pipeline enriches AMP-like positives over negatives
regardless of the type of negative control used:

A. All-repeat / degenerate negatives (easy) — already tested elsewhere
B. Poly-cationic negatives (hard: high charge) — new
C. Poly-hydrophobic negatives (hard: high hydro) — new

A pipeline that only works on "easy" negatives (all-A, all-G) cannot be trusted.
A pipeline that works across all three negative types is more credible.

All scores are computational proxies. No biological activity is implied.
"""
from __future__ import annotations

import csv
from pathlib import Path

from openamp_foundry.benchmark.evaluate import enrichment_factor, recall_at_k
from openamp_foundry.features.physchem import compute_features
from openamp_foundry.pipeline import score_candidates
from openamp_foundry.scoring.activity import activity_likeness_score
from openamp_foundry.scoring.safety import safety_score


POSITIVES_CSV = "examples/benchmark/robustness_positives.csv"
NEGATIVE_A_CSV = "examples/negative/demo_negative_peptides.csv"
NEGATIVE_B_CSV = "examples/negative/poly_cationic.csv"
NEGATIVE_C_CSV = "examples/negative/poly_hydrophobic.csv"

POSITIVE_IDS = {"ROB-POS-001", "ROB-POS-002", "ROB-POS-003"}


def _merge_csv_files(pos_csv: str, neg_csv: str, tmp_path: Path) -> Path:
"""Combine a positive and negative CSV into a single candidate file."""
out = tmp_path / "merged.csv"
rows = []
for path in [pos_csv, neg_csv]:
with open(path) as f:
reader = csv.DictReader(f)
rows.extend(list(reader))
with out.open("w", newline="") as f:
writer = csv.DictWriter(f, fieldnames=["id", "sequence", "source"])
writer.writeheader()
writer.writerows(rows)
return out


class TestNegativeSetProperties:
"""Verify the negative datasets have the expected physicochemical profiles."""

def test_poly_cationic_has_high_charge_low_hydrophobicity(self):
features = compute_features("KKKKKKKKKKKKK")
assert features["charge_density"] > 0.8, "Poly-K should have high charge density"
assert features["hydrophobic_fraction"] == 0.0, "Poly-K should have zero hydrophobic fraction"

def test_poly_hydrophobic_has_high_hydro_zero_charge(self):
features = compute_features("LLLLLLLLLLLLL")
assert features["hydrophobic_fraction"] == 1.0, "Poly-L should have 100% hydrophobic fraction"
assert features["charge_density"] == 0.0, "Poly-L should have zero charge density"

def test_poly_cationic_activity_score_below_real_amps(self):
amp_feat = compute_features("KWKLFKKIGAVLKVL")
polyk_feat = compute_features("KKKKKKKKKKKKK")
amp_score = activity_likeness_score(amp_feat)
polyk_score = activity_likeness_score(polyk_feat)
assert amp_score > polyk_score, (
f"Real AMP activity ({amp_score:.3f}) should exceed poly-K ({polyk_score:.3f}): "
"poly-K has high charge but no hydrophobic face"
)

def test_poly_hydrophobic_activity_score_below_real_amps(self):
amp_feat = compute_features("KWKLFKKIGAVLKVL")
polyl_feat = compute_features("LLLLLLLLLLLLL")
amp_score = activity_likeness_score(amp_feat)
polyl_score = activity_likeness_score(polyl_feat)
assert amp_score > polyl_score, (
f"Real AMP activity ({amp_score:.3f}) should exceed poly-L ({polyl_score:.3f}): "
"poly-L has no charge"
)

def test_poly_cationic_safety_penalized(self):
polyk_feat = compute_features("KKKKKKKKKKKKK")
polyk_safety = safety_score(polyk_feat)
# Very high charge density should trigger the safety risk penalty
assert polyk_safety < 0.7, (
f"Poly-K safety={polyk_safety:.3f} should be <0.7 due to high charge density risk"
)

def test_poly_hydrophobic_safety_penalized(self):
polyl_feat = compute_features("LLLLLLLLLLLLL")
polyl_safety = safety_score(polyl_feat)
# Very high hydrophobic fraction triggers hemolysis proxy penalty
assert polyl_safety < 0.5, (
f"Poly-L safety={polyl_safety:.3f} should be <0.5 due to extreme hydrophobicity"
)

def test_poly_repeat_long_run_detected(self):
polyk_feat = compute_features("KKKKKKKKKKKKK")
assert polyk_feat["longest_repeat_run"] == 13, "Poly-K should have repeat run of 13"

def test_poly_cationic_file_exists(self):
assert Path(NEGATIVE_B_CSV).exists(), "Poly-cationic negative set CSV not found"

def test_poly_hydrophobic_file_exists(self):
assert Path(NEGATIVE_C_CSV).exists(), "Poly-hydrophobic negative set CSV not found"


class TestEnrichmentAcrossNegativeSets:
"""Verify EF > 1.0 when positives are mixed with each negative set."""

def test_enrichment_vs_degenerate_negatives(self, tmp_path):
"""Baseline: positives should easily outrank all-repeat / degenerate negatives."""
merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_A_CSV, tmp_path)
scored, _ = score_candidates(merged)
k = len(POSITIVE_IDS)
ef = enrichment_factor(scored, POSITIVE_IDS, k=k)
assert ef > 1.0, f"EF={ef:.3f} should exceed 1.0 vs degenerate negatives"

def test_enrichment_vs_poly_cationic_negatives(self, tmp_path):
"""Harder test: positives outrank high-charge-only negatives (poly-K/R)."""
merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_B_CSV, tmp_path)
scored, _ = score_candidates(merged)
k = len(POSITIVE_IDS)
ef = enrichment_factor(scored, POSITIVE_IDS, k=k)
assert ef > 1.0, (
f"EF={ef:.3f} should exceed 1.0 vs poly-cationic negatives. "
"Pipeline must use BOTH charge and hydrophobicity, not just charge."
)

def test_enrichment_vs_poly_hydrophobic_negatives(self, tmp_path):
"""Harder test: positives outrank hydrophobic-only negatives (poly-L/I/V)."""
merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_C_CSV, tmp_path)
scored, _ = score_candidates(merged)
k = len(POSITIVE_IDS)
ef = enrichment_factor(scored, POSITIVE_IDS, k=k)
assert ef > 1.0, (
f"EF={ef:.3f} should exceed 1.0 vs poly-hydrophobic negatives. "
"Pipeline must use BOTH charge and hydrophobicity, not just hydrophobicity."
)

def test_recall_at_k_vs_poly_cationic(self, tmp_path):
"""recall@3 should be 1.0 (all 3 positives in top 3) vs poly-cationic negatives."""
merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_B_CSV, tmp_path)
scored, _ = score_candidates(merged)
rc = recall_at_k(scored, POSITIVE_IDS, k=3)
assert rc == 1.0, (
f"recall@3={rc:.4f}: all 3 known positives should be in the top 3 "
"when mixed with 5 poly-cationic negatives"
)

def test_recall_at_k_vs_poly_hydrophobic(self, tmp_path):
"""recall@3 should be 1.0 (all 3 positives in top 3) vs poly-hydrophobic negatives."""
merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_C_CSV, tmp_path)
scored, _ = score_candidates(merged)
rc = recall_at_k(scored, POSITIVE_IDS, k=3)
assert rc == 1.0, (
f"recall@3={rc:.4f}: all 3 known positives should be in the top 3 "
"when mixed with 5 poly-hydrophobic negatives"
)

def test_ef_consistent_across_all_negative_sets(self, tmp_path):
"""EF > 1.0 for all three negative set types — robustness check."""
results = {}
for label, neg_csv in [
("degenerate", NEGATIVE_A_CSV),
("poly_cationic", NEGATIVE_B_CSV),
("poly_hydrophobic", NEGATIVE_C_CSV),
]:
subdir = tmp_path / label
subdir.mkdir()
merged = _merge_csv_files(POSITIVES_CSV, neg_csv, subdir)
scored, _ = score_candidates(merged)
ef = enrichment_factor(scored, POSITIVE_IDS, k=len(POSITIVE_IDS))
results[label] = ef

failures = [
f"{label} (EF={ef:.3f})"
for label, ef in results.items()
if ef <= 1.0
]
assert not failures, (
f"EF ≤ 1.0 for negative set(s): {failures}. "
"Pipeline must beat random across all tested negative types."
)

def test_mean_positive_score_exceeds_mean_negative_across_all_sets(self, tmp_path):
"""Mean ensemble score of positives > mean of negatives for all negative types."""
for label, neg_csv in [
("degenerate", NEGATIVE_A_CSV),
("poly_cationic", NEGATIVE_B_CSV),
("poly_hydrophobic", NEGATIVE_C_CSV),
]:
subdir = tmp_path / label
subdir.mkdir()
merged = _merge_csv_files(POSITIVES_CSV, neg_csv, subdir)
scored, _ = score_candidates(merged)

pos_ensemble = [
s.scores["ensemble"]
for s in scored
if s.candidate.candidate_id in POSITIVE_IDS
]
neg_ensemble = [
s.scores["ensemble"]
for s in scored
if s.candidate.candidate_id not in POSITIVE_IDS
]
avg_pos = sum(pos_ensemble) / len(pos_ensemble)
avg_neg = sum(neg_ensemble) / len(neg_ensemble)
assert avg_pos > avg_neg, (
f"[{label}] Mean positive ensemble ({avg_pos:.3f}) should exceed "
f"mean negative ensemble ({avg_neg:.3f})"
)
2 changes: 0 additions & 2 deletions tests/test_pipeline_filters.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,9 +2,7 @@
from __future__ import annotations

import json
from pathlib import Path

import pytest

from openamp_foundry.pipeline import _passes_length_filter, run_ranking_pipeline, score_candidates

Expand Down
Loading