Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 28 additions & 5 deletions Makefile
Original file line number Diff line number Diff line change
@@ -1,13 +1,36 @@
.PHONY: demo test lint clean
.PHONY: demo test lint clean bench-leakage bench-baseline

PYTHON := $(shell [ -f .venv/bin/python ] && echo .venv/bin/python || echo python3)
PYTEST := $(shell [ -f .venv/bin/pytest ] && echo .venv/bin/pytest || echo pytest)
RUFF := $(shell [ -f .venv/bin/ruff ] && echo .venv/bin/ruff || echo ruff)

demo:
PYTHONPATH=src python -m openamp_foundry.cli rank --candidates examples/sequences/demo_candidates.csv --references examples/known_reference/demo_known_amps.csv --out outputs/demo_ranked.jsonl --report outputs/demo_report.md --cert-dir outputs/evidence
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli rank \
--candidates examples/sequences/demo_candidates.csv \
--references examples/known_reference/demo_known_amps.csv \
--out outputs/demo_ranked.jsonl \
--report outputs/demo_report.md \
--cert-dir outputs/evidence \
--manifest outputs/run_manifest.json

test:
pytest -q
$(PYTEST) -q

lint:
ruff check src tests scripts
$(RUFF) check src tests scripts

bench-leakage:
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli bench leakage \
--candidates examples/sequences/demo_candidates.csv \
--references examples/known_reference/demo_known_amps.csv \
--out outputs/leakage_report.json

bench-baseline:
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli bench baseline \
--candidates examples/sequences/demo_candidates.csv \
--references examples/known_reference/demo_known_amps.csv \
--positives examples/known_reference/demo_known_amps.csv \
--out outputs/bench_baseline_report.json

clean:
rm -rf outputs/*.jsonl outputs/*.md outputs/evidence .pytest_cache .ruff_cache
rm -rf outputs/*.jsonl outputs/*.md outputs/*.json outputs/evidence .pytest_cache .ruff_cache
95 changes: 95 additions & 0 deletions src/openamp_foundry/benchmark/evaluate.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,3 +6,98 @@
def top_k_ids(scored: list[ScoredCandidate], k: int) -> set[str]:
ranked = sorted(scored, key=lambda x: x.scores.get("ensemble", 0.0), reverse=True)
return {item.candidate.candidate_id for item in ranked[:k]}


def recall_at_k(
scored: list[ScoredCandidate],
positive_ids: set[str],
k: int,
) -> float:
"""Fraction of known positives recovered in the top-k ranked candidates.

Recall@k = |positives in top-k| / |total positives|

A random baseline would recover roughly k / |all candidates| of the positives.
The pipeline is meaningful if recall@k significantly exceeds the random baseline.
"""
if not positive_ids:
return 0.0
top = top_k_ids(scored, k)
recovered = len(top & positive_ids)
return round(recovered / len(positive_ids), 4)


def random_recall_at_k(n_candidates: int, n_positives: int, k: int) -> float:
"""Expected recall@k for a random ranker.

E[recall@k] = min(k, n_positives) / n_positives when sampling without replacement.
"""
if n_positives == 0 or n_candidates == 0:
return 0.0
expected_hits = min(k, n_positives) * min(k, n_candidates) / max(n_candidates, 1)
return round(min(1.0, expected_hits / n_positives), 4)


def enrichment_factor(
scored: list[ScoredCandidate],
positive_ids: set[str],
k: int,
) -> float:
"""Enrichment factor at k: recall@k relative to random baseline recall@k.

EF > 1.0 means the pipeline outperforms random.
EF = 1.0 is random.
EF < 1.0 is worse than random (anti-enrichment).
"""
n = len(scored)
n_pos = len(positive_ids)
rc = recall_at_k(scored, positive_ids, k)
random_rc = random_recall_at_k(n, n_pos, k)
if random_rc == 0.0:
return 0.0
return round(rc / random_rc, 4)


def benchmark_summary(
scored: list[ScoredCandidate],
positive_ids: set[str],
ks: list[int] | None = None,
) -> dict:
"""Produce a benchmark summary comparing pipeline vs random ranker.

Returns recall@k and enrichment factor at each k, plus a verdict.
All results are computational only — they do not prove biological activity.
"""
if ks is None:
n = len(scored)
ks = sorted({max(1, n // 10), max(1, n // 5), max(1, n // 2), n})

results = []
for k in ks:
rc = recall_at_k(scored, positive_ids, k)
rrc = random_recall_at_k(len(scored), len(positive_ids), k)
ef = enrichment_factor(scored, positive_ids, k)
results.append(
{
"k": k,
"recall_at_k": rc,
"random_recall_at_k": rrc,
"enrichment_factor": ef,
}
)

any_enrichment = any(r["enrichment_factor"] > 1.0 for r in results)
verdict = "pipeline outperforms random" if any_enrichment else "pipeline does not outperform random"

return {
"disclaimer": (
"These are retrospective benchmark results on demo data. "
"They do not prove biological efficacy. "
"They only measure whether the pipeline recovers known positives "
"better than a random ranker would."
),
"n_candidates": len(scored),
"n_positives": len(positive_ids),
"results": results,
"verdict": verdict,
}
81 changes: 81 additions & 0 deletions src/openamp_foundry/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,12 +19,43 @@ def build_parser() -> argparse.ArgumentParser:
rank.add_argument("--out", required=True)
rank.add_argument("--report", required=False)
rank.add_argument("--cert-dir", required=False)
rank.add_argument("--manifest", required=False)
rank.add_argument("--config", default="configs/pipeline.yaml")

validate = sub.add_parser("validate", help="Validate a candidate certificate against JSON schema.")
validate.add_argument("--certificate", required=True)
validate.add_argument("--schema", required=True)

bench = sub.add_parser("bench", help="Run benchmark and leakage checks.")
bench_sub = bench.add_subparsers(dest="bench_command", required=True)

leakage = bench_sub.add_parser("leakage", help="Find near-duplicate candidates in references.")
leakage.add_argument("--candidates", required=True)
leakage.add_argument("--references", required=True)
leakage.add_argument("--threshold", type=float, default=0.90)
leakage.add_argument("--out", required=False, help="Optional JSON output path.")

baseline = bench_sub.add_parser(
"baseline",
help="Evaluate pipeline recall vs random baseline on a labelled set.",
)
baseline.add_argument("--candidates", required=True, help="CSV of all candidates to score.")
baseline.add_argument("--references", required=False, help="Reference CSV for novelty scoring.")
baseline.add_argument(
"--positives",
required=True,
help="CSV of known-positive (active) peptides. IDs must match candidates CSV.",
)
baseline.add_argument(
"--k",
type=int,
nargs="+",
default=None,
help="Recall@k cutoffs (default: auto from dataset size).",
)
baseline.add_argument("--config", default="configs/pipeline.yaml")
baseline.add_argument("--out", required=False, help="Optional JSON output path.")

return parser


Expand All @@ -40,6 +71,7 @@ def main(argv: list[str] | None = None) -> int:
report_path=args.report,
cert_dir=args.cert_dir,
config_path=args.config,
manifest_path=args.manifest,
)
print(json.dumps({"status": "ok", "out": args.out, "report": args.report}, indent=2))
return 0
Expand All @@ -50,9 +82,58 @@ def main(argv: list[str] | None = None) -> int:
print(json.dumps({"status": "valid", "certificate": args.certificate}, indent=2))
return 0

if args.command == "bench":
return _run_bench(args)

parser.error("unknown command")
return 2


def _run_bench(args: argparse.Namespace) -> int:
from openamp_foundry.data.loaders import load_candidates_csv
from openamp_foundry.utils.io import write_json

if args.bench_command == "leakage":
from openamp_foundry.benchmark.leakage import find_near_duplicates

candidates = load_candidates_csv(args.candidates)
references = load_candidates_csv(args.references)
hits = find_near_duplicates(candidates, references, threshold=args.threshold)
result = {
"status": "ok",
"threshold": args.threshold,
"near_duplicate_count": len(hits),
"near_duplicates": hits,
"warning": (
"Near-duplicates detected. If these candidates were used for training or "
"scoring baseline models, benchmark results may be inflated."
) if hits else None,
}
if args.out:
write_json(args.out, result)
print(json.dumps(result, indent=2))
return 0

if args.bench_command == "baseline":
from openamp_foundry.benchmark.evaluate import benchmark_summary
from openamp_foundry.pipeline import score_candidates

scored, _ = score_candidates(
candidate_path=args.candidates,
reference_path=args.references,
config_path=args.config,
)
positives = load_candidates_csv(args.positives)
positive_ids = {p.candidate_id for p in positives}
summary = benchmark_summary(scored, positive_ids, ks=args.k)
result = {"status": "ok", **summary}
if args.out:
write_json(args.out, result)
print(json.dumps(result, indent=2))
return 0

return 2


if __name__ == "__main__":
raise SystemExit(main())
35 changes: 35 additions & 0 deletions src/openamp_foundry/features/physchem.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
from __future__ import annotations

import math
from collections import Counter

HYDROPHOBIC = set("AILMFWVY")
Expand All @@ -8,6 +9,15 @@
AROMATIC = set("FWY")
CYS = "C"

# Eisenberg consensus hydrophobicity scale (normalized, 0-centred removed, shifted to 0..1 range)
# Source: Eisenberg et al. (1984) J Mol Biol. Used for hydrophobic moment only.
_HYDROPHOBICITY: dict[str, float] = {
"A": 0.620, "R": -2.530, "N": -0.780, "D": -0.900, "C": 0.290,
"Q": -0.850, "E": -0.740, "G": 0.480, "H": -0.400, "I": 1.380,
"L": 1.060, "K": -1.500, "M": 0.640, "F": 1.190, "P": 0.120,
"S": -0.180, "T": -0.050, "W": 0.810, "Y": 0.260, "V": 1.080,
}


def net_charge_proxy(sequence: str) -> int:
return sum(1 for aa in sequence if aa in POSITIVE) - sum(1 for aa in sequence if aa in NEGATIVE)
Expand All @@ -33,6 +43,29 @@ def longest_repeat_run(sequence: str) -> int:
return best


def hydrophobic_moment(sequence: str, angle_deg: float = 100.0) -> float:
"""Compute mean hydrophobic moment for an assumed alpha-helical conformation.

Uses the Eisenberg (1984) consensus scale. Angle of 100° per residue is the
standard helical wheel projection used in AMP literature.

Returns a value in [0, ∞); typical AMPs have μH > 0.4. Not normalised to [0,1]
because the range is sequence-length-dependent — callers should normalise if needed.
"""
if not sequence:
return 0.0
angle_rad = math.radians(angle_deg)
sin_sum = 0.0
cos_sum = 0.0
for i, aa in enumerate(sequence):
h = _HYDROPHOBICITY.get(aa, 0.0)
theta = i * angle_rad
sin_sum += h * math.sin(theta)
cos_sum += h * math.cos(theta)
moment = math.sqrt(sin_sum ** 2 + cos_sum ** 2) / len(sequence)
return round(moment, 4)


def compute_features(sequence: str) -> dict[str, float | int | dict[str, int]]:
counts = Counter(sequence)
length = len(sequence)
Expand All @@ -43,6 +76,7 @@ def compute_features(sequence: str) -> dict[str, float | int | dict[str, int]]:
gly_fraction = counts.get("G", 0) / length if length else 0.0
pro_fraction = counts.get("P", 0) / length if length else 0.0
repeat_run = longest_repeat_run(sequence)
mu_h = hydrophobic_moment(sequence)
return {
"length": length,
"net_charge_proxy": charge,
Expand All @@ -53,5 +87,6 @@ def compute_features(sequence: str) -> dict[str, float | int | dict[str, int]]:
"glycine_fraction": round(gly_fraction, 4),
"proline_fraction": round(pro_fraction, 4),
"longest_repeat_run": repeat_run,
"hydrophobic_moment": mu_h,
"residue_counts": dict(sorted(counts.items())),
}
Loading
Loading