Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 15 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
.PHONY: demo test lint clean bench-leakage
.PHONY: demo test lint clean bench-leakage bench-baseline bench-hidden-active

PYTHON := $(shell [ -f .venv/bin/python ] && echo .venv/bin/python || echo python3)
PYTEST := $(shell [ -f .venv/bin/pytest ] && echo .venv/bin/pytest || echo pytest)
Expand All @@ -25,5 +25,19 @@ bench-leakage:
--references examples/known_reference/demo_known_amps.csv \
--out outputs/leakage_report.json

bench-baseline:
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli bench baseline \
--candidates examples/sequences/demo_candidates.csv \
--references examples/known_reference/demo_known_amps.csv \
--positives examples/known_reference/demo_known_amps.csv \
--out outputs/bench_baseline_report.json

bench-hidden-active:
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli bench baseline \
--candidates examples/benchmark/mixed_candidates.csv \
--positives examples/benchmark/active_labels.csv \
--k 5 10 20 \
--out outputs/bench_hidden_active_report.json

clean:
rm -rf outputs/*.jsonl outputs/*.md outputs/*.json outputs/evidence .pytest_cache .ruff_cache
6 changes: 6 additions & 0 deletions examples/benchmark/active_labels.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
id,sequence,source
BM-POS-001,KWKLFKKIGAVLKVL,benchmark_positive
BM-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,benchmark_positive
BM-POS-003,GLFDIVKKVVGALGSL,benchmark_positive
BM-POS-004,RRWWLRVIAGLLKKVLS,benchmark_positive
BM-POS-005,KLLLKWLKKVLKA,benchmark_positive
21 changes: 21 additions & 0 deletions examples/benchmark/mixed_candidates.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
id,sequence,source
BM-POS-001,KWKLFKKIGAVLKVL,benchmark_positive
BM-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,benchmark_positive
BM-POS-003,GLFDIVKKVVGALGSL,benchmark_positive
BM-POS-004,RRWWLRVIAGLLKKVLS,benchmark_positive
BM-POS-005,KLLLKWLKKVLKA,benchmark_positive
BM-NEG-001,AAAAAAAAAAAA,benchmark_negative
BM-NEG-002,GGGGGGGGGGGG,benchmark_negative
BM-NEG-003,DEDEDEDEDEDE,benchmark_negative
BM-NEG-004,SSSSSSSSSSS,benchmark_negative
BM-NEG-005,EEEEEEEEEEEE,benchmark_negative
BM-NEG-006,PPPPPPPPPPPP,benchmark_negative
BM-NEG-007,NNNNNNNNNNN,benchmark_negative
BM-NEG-008,QQQQQQQQQQQQ,benchmark_negative
BM-NEG-009,TTTTTTTTTTTT,benchmark_negative
BM-NEG-010,AAGGAAGGAAGG,benchmark_negative
BM-NEG-011,SSTTSSTTSSTT,benchmark_negative
BM-NEG-012,EDEDEDEDEDED,benchmark_negative
BM-NEG-013,GGGAAAGGGAAA,benchmark_negative
BM-NEG-014,QQNNQQNNQQNN,benchmark_negative
BM-NEG-015,SSSSTTTTSSSS,benchmark_negative
95 changes: 95 additions & 0 deletions src/openamp_foundry/benchmark/evaluate.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,3 +6,98 @@
def top_k_ids(scored: list[ScoredCandidate], k: int) -> set[str]:
ranked = sorted(scored, key=lambda x: x.scores.get("ensemble", 0.0), reverse=True)
return {item.candidate.candidate_id for item in ranked[:k]}


def recall_at_k(
scored: list[ScoredCandidate],
positive_ids: set[str],
k: int,
) -> float:
"""Fraction of known positives recovered in the top-k ranked candidates.

Recall@k = |positives in top-k| / |total positives|

A random baseline would recover roughly k / |all candidates| of the positives.
The pipeline is meaningful if recall@k significantly exceeds the random baseline.
"""
if not positive_ids:
return 0.0
top = top_k_ids(scored, k)
recovered = len(top & positive_ids)
return round(recovered / len(positive_ids), 4)


def random_recall_at_k(n_candidates: int, n_positives: int, k: int) -> float:
"""Expected recall@k for a random ranker.

E[recall@k] = min(k, n_positives) / n_positives when sampling without replacement.
"""
if n_positives == 0 or n_candidates == 0:
return 0.0
expected_hits = min(k, n_positives) * min(k, n_candidates) / max(n_candidates, 1)
return round(min(1.0, expected_hits / n_positives), 4)


def enrichment_factor(
scored: list[ScoredCandidate],
positive_ids: set[str],
k: int,
) -> float:
"""Enrichment factor at k: recall@k relative to random baseline recall@k.

EF > 1.0 means the pipeline outperforms random.
EF = 1.0 is random.
EF < 1.0 is worse than random (anti-enrichment).
"""
n = len(scored)
n_pos = len(positive_ids)
rc = recall_at_k(scored, positive_ids, k)
random_rc = random_recall_at_k(n, n_pos, k)
if random_rc == 0.0:
return 0.0
return round(rc / random_rc, 4)


def benchmark_summary(
scored: list[ScoredCandidate],
positive_ids: set[str],
ks: list[int] | None = None,
) -> dict:
"""Produce a benchmark summary comparing pipeline vs random ranker.

Returns recall@k and enrichment factor at each k, plus a verdict.
All results are computational only — they do not prove biological activity.
"""
if ks is None:
n = len(scored)
ks = sorted({max(1, n // 10), max(1, n // 5), max(1, n // 2), n})

results = []
for k in ks:
rc = recall_at_k(scored, positive_ids, k)
rrc = random_recall_at_k(len(scored), len(positive_ids), k)
ef = enrichment_factor(scored, positive_ids, k)
results.append(
{
"k": k,
"recall_at_k": rc,
"random_recall_at_k": rrc,
"enrichment_factor": ef,
}
)

any_enrichment = any(r["enrichment_factor"] > 1.0 for r in results)
verdict = "pipeline outperforms random" if any_enrichment else "pipeline does not outperform random"

return {
"disclaimer": (
"These are retrospective benchmark results on demo data. "
"They do not prove biological efficacy. "
"They only measure whether the pipeline recovers known positives "
"better than a random ranker would."
),
"n_candidates": len(scored),
"n_positives": len(positive_ids),
"results": results,
"verdict": verdict,
}
42 changes: 41 additions & 1 deletion src/openamp_foundry/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,27 @@ def build_parser() -> argparse.ArgumentParser:
leakage.add_argument("--threshold", type=float, default=0.90)
leakage.add_argument("--out", required=False, help="Optional JSON output path.")

baseline = bench_sub.add_parser(
"baseline",
help="Evaluate pipeline recall vs random baseline on a labelled set.",
)
baseline.add_argument("--candidates", required=True, help="CSV of all candidates to score.")
baseline.add_argument("--references", required=False, help="Reference CSV for novelty scoring.")
baseline.add_argument(
"--positives",
required=True,
help="CSV of known-positive (active) peptides. IDs must match candidates CSV.",
)
baseline.add_argument(
"--k",
type=int,
nargs="+",
default=None,
help="Recall@k cutoffs (default: auto from dataset size).",
)
baseline.add_argument("--config", default="configs/pipeline.yaml")
baseline.add_argument("--out", required=False, help="Optional JSON output path.")

return parser


Expand Down Expand Up @@ -69,11 +90,12 @@ def main(argv: list[str] | None = None) -> int:


def _run_bench(args: argparse.Namespace) -> int:
from openamp_foundry.benchmark.leakage import find_near_duplicates
from openamp_foundry.data.loaders import load_candidates_csv
from openamp_foundry.utils.io import write_json

if args.bench_command == "leakage":
from openamp_foundry.benchmark.leakage import find_near_duplicates

candidates = load_candidates_csv(args.candidates)
references = load_candidates_csv(args.references)
hits = find_near_duplicates(candidates, references, threshold=args.threshold)
Expand All @@ -92,6 +114,24 @@ def _run_bench(args: argparse.Namespace) -> int:
print(json.dumps(result, indent=2))
return 0

if args.bench_command == "baseline":
from openamp_foundry.benchmark.evaluate import benchmark_summary
from openamp_foundry.pipeline import score_candidates

scored, _ = score_candidates(
candidate_path=args.candidates,
reference_path=args.references,
config_path=args.config,
)
positives = load_candidates_csv(args.positives)
positive_ids = {p.candidate_id for p in positives}
summary = benchmark_summary(scored, positive_ids, ks=args.k)
result = {"status": "ok", **summary}
if args.out:
write_json(args.out, result)
print(json.dumps(result, indent=2))
return 0

return 2


Expand Down
35 changes: 35 additions & 0 deletions src/openamp_foundry/features/physchem.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
from __future__ import annotations

import math
from collections import Counter

HYDROPHOBIC = set("AILMFWVY")
Expand All @@ -8,6 +9,15 @@
AROMATIC = set("FWY")
CYS = "C"

# Eisenberg consensus hydrophobicity scale (normalized, 0-centred removed, shifted to 0..1 range)
# Source: Eisenberg et al. (1984) J Mol Biol. Used for hydrophobic moment only.
_HYDROPHOBICITY: dict[str, float] = {
"A": 0.620, "R": -2.530, "N": -0.780, "D": -0.900, "C": 0.290,
"Q": -0.850, "E": -0.740, "G": 0.480, "H": -0.400, "I": 1.380,
"L": 1.060, "K": -1.500, "M": 0.640, "F": 1.190, "P": 0.120,
"S": -0.180, "T": -0.050, "W": 0.810, "Y": 0.260, "V": 1.080,
}


def net_charge_proxy(sequence: str) -> int:
return sum(1 for aa in sequence if aa in POSITIVE) - sum(1 for aa in sequence if aa in NEGATIVE)
Expand All @@ -33,6 +43,29 @@ def longest_repeat_run(sequence: str) -> int:
return best


def hydrophobic_moment(sequence: str, angle_deg: float = 100.0) -> float:
"""Compute mean hydrophobic moment for an assumed alpha-helical conformation.

Uses the Eisenberg (1984) consensus scale. Angle of 100° per residue is the
standard helical wheel projection used in AMP literature.

Returns a value in [0, ∞); typical AMPs have μH > 0.4. Not normalised to [0,1]
because the range is sequence-length-dependent — callers should normalise if needed.
"""
if not sequence:
return 0.0
angle_rad = math.radians(angle_deg)
sin_sum = 0.0
cos_sum = 0.0
for i, aa in enumerate(sequence):
h = _HYDROPHOBICITY.get(aa, 0.0)
theta = i * angle_rad
sin_sum += h * math.sin(theta)
cos_sum += h * math.cos(theta)
moment = math.sqrt(sin_sum ** 2 + cos_sum ** 2) / len(sequence)
return round(moment, 4)


def compute_features(sequence: str) -> dict[str, float | int | dict[str, int]]:
counts = Counter(sequence)
length = len(sequence)
Expand All @@ -43,6 +76,7 @@ def compute_features(sequence: str) -> dict[str, float | int | dict[str, int]]:
gly_fraction = counts.get("G", 0) / length if length else 0.0
pro_fraction = counts.get("P", 0) / length if length else 0.0
repeat_run = longest_repeat_run(sequence)
mu_h = hydrophobic_moment(sequence)
return {
"length": length,
"net_charge_proxy": charge,
Expand All @@ -53,5 +87,6 @@ def compute_features(sequence: str) -> dict[str, float | int | dict[str, int]]:
"glycine_fraction": round(gly_fraction, 4),
"proline_fraction": round(pro_fraction, 4),
"longest_repeat_run": repeat_run,
"hydrophobic_moment": mu_h,
"residue_counts": dict(sorted(counts.items())),
}
35 changes: 33 additions & 2 deletions src/openamp_foundry/scoring/activity.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,16 +6,47 @@ def clamp01(x: float) -> float:


def activity_likeness_score(features: dict) -> float:
"""Transparent baseline, not a validated biological predictor."""
"""Transparent baseline activity-likeness score.

This is NOT a validated biological predictor. It combines known physicochemical
correlates of AMP activity: length, cationic charge density, hydrophobic fraction,
aromatic content, and hydrophobic moment (amphipathicity). All weights and thresholds
are documented and fixed before any benchmark evaluation.

Literature basis:
- Charge: Zasloff (2002) Nature; most AMPs carry net charge +2 to +9
- Hydrophobicity: Hancock & Sahl (2006) Nat Biotechnol; ~30-50% hydrophobic
- Amphipathicity (μH): Eisenberg et al. (1984); helical amphipathicity correlates
with membrane disruption
- Length: Jenssen et al. (2006) Clin Microbiol Rev; typical 8-50 AA
"""
length = features["length"]
charge_density = features["charge_density"]
hydrophobic = features["hydrophobic_fraction"]
aromatic = features["aromatic_fraction"]
mu_h = features.get("hydrophobic_moment", 0.0)

# Length: peak at ~18 residues, broad tolerance
length_score = 1.0 - min(abs(length - 18) / 25, 1.0)

# Charge: AMPs typically have positive charge density 0.1–0.5
charge_score = clamp01((charge_density + 0.05) / 0.55)

# Hydrophobicity: 40-50% is a sweet spot for membrane interaction
hydro_score = 1.0 - min(abs(hydrophobic - 0.45) / 0.45, 1.0)

# Aromatic residues (F, W, Y) aid membrane insertion
aromatic_bonus = min(aromatic / 0.20, 1.0) * 0.10

score = 0.30 * length_score + 0.35 * charge_score + 0.25 * hydro_score + aromatic_bonus
# Amphipathicity: helical hydrophobic moment > 0.4 is associated with activity
# Typical range for AMPs: 0.3–0.8; scale to [0,1] over 0–0.8 range
amphipathicity_score = clamp01(mu_h / 0.8) * 0.15

score = (
0.28 * length_score
+ 0.32 * charge_score
+ 0.20 * hydro_score
+ aromatic_bonus
+ amphipathicity_score
)
return round(clamp01(score), 4)
Loading
Loading