diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0a528936..79fd6d44 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -13,11 +13,47 @@ jobs: - uses: actions/setup-python@v5 with: python-version: '3.11' + - name: Install - run: pip install -e .[dev] + run: pip install -e ".[dev]" + - name: Lint - run: ruff check src tests scripts - - name: Test - run: pytest -q - - name: Demo + run: ruff check src tests + + - name: Unit + integration tests + run: pytest -q --tb=short + + - name: Demo pipeline run: make demo + + - name: Validate evidence certificates + run: | + for cert in outputs/evidence/*.json; do + PYTHONPATH=src python -m openamp_foundry.cli validate \ + --certificate "$cert" \ + --schema schemas/candidate.schema.json + done + + - name: Leakage check (informational) + run: make bench-leakage + + - name: Hidden-active benchmark (require EF >= 1.5 at k=5) + run: | + PYTHONPATH=src python -c " + import json, subprocess, sys, os + result = subprocess.run( + ['python', '-m', 'openamp_foundry.cli', 'bench', 'baseline', + '--candidates', 'examples/benchmark/mixed_candidates.csv', + '--positives', 'examples/benchmark/active_labels.csv', + '--k', '5'], + capture_output=True, text=True, check=True, + env={**os.environ, 'PYTHONPATH': 'src'} + ) + data = json.loads(result.stdout) + ef = next(r['enrichment_factor'] for r in data['results'] if r['k'] == 5) + print(f'Enrichment factor at k=5: {ef}') + if ef < 1.5: + print(f'FAIL: EF={ef} below minimum 1.5') + sys.exit(1) + print('PASS: Pipeline outperforms random ranker on hidden-active benchmark') + " diff --git a/Makefile b/Makefile index dfea9e0a..5f6de500 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: demo test lint clean bench-leakage +.PHONY: demo test lint clean bench-leakage bench-baseline bench-hidden-active generate phase3 PYTHON := $(shell [ -f .venv/bin/python ] && echo .venv/bin/python || echo python3) PYTEST := $(shell [ -f .venv/bin/pytest ] && echo .venv/bin/pytest || echo pytest) @@ -25,5 +25,41 @@ bench-leakage: --references examples/known_reference/demo_known_amps.csv \ --out outputs/leakage_report.json +bench-baseline: + PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli bench baseline \ + --candidates examples/sequences/demo_candidates.csv \ + --references examples/known_reference/demo_known_amps.csv \ + --positives examples/known_reference/demo_known_amps.csv \ + --out outputs/bench_baseline_report.json + +bench-hidden-active: + PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli bench baseline \ + --candidates examples/benchmark/mixed_candidates.csv \ + --positives examples/benchmark/active_labels.csv \ + --k 5 10 20 \ + --out outputs/bench_hidden_active_report.json + +generate: + PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli generate-batch \ + --seeds examples/sequences/amp_seeds.csv \ + --out examples/sequences/phase3_pool.csv \ + --n-double 25 \ + --n-charge 12 \ + --rng-seed 2024 + +phase3: generate + PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli rank \ + --candidates examples/sequences/phase3_pool.csv \ + --references examples/known_reference/amp_curated_references.csv \ + --out outputs/phase3_ranked.jsonl \ + --report outputs/phase3_report.md \ + --cert-dir outputs/phase3_evidence \ + --manifest outputs/phase3_manifest.json \ + --config configs/phase3.yaml + PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli batch-pack \ + --ranked outputs/phase3_ranked.jsonl \ + --out-json outputs/phase3_batch_pack.json \ + --out-md outputs/phase3_batch_pack.md + clean: - rm -rf outputs/*.jsonl outputs/*.md outputs/*.json outputs/evidence .pytest_cache .ruff_cache + rm -rf outputs/*.jsonl outputs/*.md outputs/*.json outputs/evidence outputs/phase3_evidence .pytest_cache .ruff_cache diff --git a/configs/phase3.yaml b/configs/phase3.yaml new file mode 100644 index 00000000..4b335c33 --- /dev/null +++ b/configs/phase3.yaml @@ -0,0 +1,31 @@ +pipeline_version: "0.1.0" + +# Phase 3 exploration config. +# Used for scoring and selecting candidates produced by the template-mutation generator. +# These candidates are 1–3 mutations from known AMP-like seeds, so min_novelty is set +# lower than the benchmark config (pipeline.yaml), which is appropriate for neighborhood +# search around known templates. +# All other filters (length, safety, synthesis) remain strict. + +filters: + min_length: 8 + max_length: 35 + allowed_amino_acids: "ACDEFGHIKLMNPQRSTVWY" + +weights: + activity: 0.35 + safety: 0.30 + synthesis: 0.20 + novelty: 0.15 + +selection: + top_n: 100 + min_novelty: 0.05 + max_safety_risk: 0.40 + +notes: + - "Phase 3 exploration config. Weights differ from pipeline.yaml to prioritize safety." + - "min_novelty=0.05 allows near-seed variants (1-3 mutation radius) through the filter." + - "max_safety_risk=0.40 is stricter than pipeline.yaml (0.70) — only safe candidates selected." + - "These weights are transparent baseline heuristics, not validated biological predictors." + - "Locked before Phase 3 generation run per docs/SELECTION_RULE.md." diff --git a/docs/EXPERT_REVIEW_PACK.md b/docs/EXPERT_REVIEW_PACK.md new file mode 100644 index 00000000..fa7b3fdc --- /dev/null +++ b/docs/EXPERT_REVIEW_PACK.md @@ -0,0 +1,257 @@ +# OpenAMP Foundry — Expert Review Pack + +**Phase 3 Candidate Batch** +**Pipeline version:** 0.1.0 (v0.2 scorer update) +**Generated:** 2026-06-27 +**Batch ID:** See `outputs/phase3_manifest.json` + +--- + +## Purpose of This Document + +This document is prepared for review by a qualified microbiology, peptide chemistry, or +infectious disease expert prior to any wet-lab expenditure. It summarises the computational +candidate nomination process, the selection rationale, and the recommended next steps. + +**The expert reviewer is asked to:** +1. Evaluate whether the candidate selection methodology is scientifically credible. +2. Identify any obvious problems the computational pipeline may have missed. +3. Advise on which candidates (if any) are worth synthesising. +4. Recommend appropriate assay types and target organisms. +5. Flag any safety or dual-use concerns. + +--- + +## Computational Disclaimer + +All scores in this document are heuristic physicochemical proxies. They are NOT validated +biological predictors. No antimicrobial activity has been demonstrated in vitro or in vivo. +Do not describe any candidate as an antibiotic, drug, cure, therapy, or proven antimicrobial +without experimental evidence. The lab is the judge. + +--- + +## 1. Overview of the Pipeline + +The OpenAMP Foundry pipeline applies a six-stage scoring system to candidate sequences: + +| Stage | Method | What it measures | +|-------|--------|-----------------| +| Activity-likeness | Physicochemical heuristics | Charge, hydrophobicity, amphipathicity, length | +| Boman activity | Boman (2003) index normalized | Per-residue interaction potential; complements activity heuristic | +| Model disagreement | \|activity − boman_activity\| | Scorer consensus: low = robust, high = uncertain | +| Safety (toxicity proxy) | Physicochemical risk flags | Hemolysis-correlated excess hydrophobicity, charge density, repeats | +| Synthesis feasibility | Composition and length filter | Cysteine content, proline content, repeat runs, length | +| Novelty | Levenshtein distance | Sequence divergence from 45 known AMP references | +| Ensemble | Weighted sum | activity×0.35 + safety×0.30 + synthesis×0.20 + novelty×0.15 | + +**Pipeline configuration:** `configs/phase3.yaml` +**Selection rule:** `docs/SELECTION_RULE.md` (locked before generation) +**Full benchmark methodology:** `docs/BENCHMARKING.md` + +**Key evidence level:** These are Level 0–2 evidence scores (syntax validity, reproducible features, +transparent heuristics). They have not been validated against lab measurements. Lab assay +(Level 5) is the required next gate. + +--- + +## 2. Reference Set + +Novelty was scored against **45 known antimicrobial peptides** drawn from published literature +(see `examples/known_reference/amp_curated_references.csv`). Families represented: + +- Magainin and pexiganan analogs (Zasloff 1987, Ge 1999) +- Buforin family (Kim 1996, Park 2000) +- Indolicidin and analogs (Selsted 1992) +- Temporin family (Mangoni 2001, Simmaco 1996) +- Aurein family (Rozek 2000) +- Cecropin family (Steiner 1981, Hultmark 1980) +- Cathelicidin-related short sequences +- Melittin (as hemolysis positive control reference) + +**Assessment:** Candidates with novelty < 0.10 against this reference set are near-duplicates +of known AMPs and should be deprioritised. Novelty ≥ 0.30 indicates meaningful divergence from +the known AMP landscape. + +--- + +## 3. Generation Method + +Candidates were produced by conservative amino-acid substitutions on 5 template seeds: + +| Seed | Sequence | Family | +|------|----------|--------| +| SEED-001 | KWKLFKKIGAVLKVL | Magainin-like | +| SEED-002 | GIGKFLHSAKKFGKAFVGEIMNS | Cecropin/Magainin-like | +| SEED-003 | RRWQWRMKKLG | Cationic tryptophan-rich | +| SEED-004 | FLPLIGRVLSGIL | Temporin-like | +| SEED-005 | KRLFKKIGSALKFL | Hybrid cationic-hydrophobic | + +Substitutions were restricted to physicochemically conservative groups (cationic→cationic, +hydrophobic→hydrophobic, etc.) to preserve the amphipathic balance of the templates. + +**Reviewer note:** This is an intentionally conservative generation strategy. It explores the +near-neighbourhood of known AMP families. It does NOT explore truly novel sequence space. +Future cycles should incorporate independent sequence generation (e.g. protein language models +or Bayesian optimization) to achieve more radical novelty. + +--- + +## 4. Batch Statistics + +| Metric | Value | +|--------|-------| +| Sequences generated | 383 | +| Sequences passing all filters | 89 | +| Sequence length range | 11–14 aa | +| Mean ensemble score | 0.781 | +| Mean predicted activity | 0.839 | +| Mean predicted safety | 1.000 | +| Mean synthesis feasibility | 1.000 | +| Mean novelty vs reference set | 0.172 | +| Diversity clusters (threshold 0.80) | 32 | +| Singleton clusters | 13 (40.6%) | +| Mean Boman activity score | 0.503 | +| Mean scorer disagreement | 0.311 | +| High consensus (disagreement <0.20) | 1 | +| Uncertain (disagreement ≥0.30) | 48 | + +**Reviewer alert — Safety scores:** Mean safety = 1.0 because ALL selected candidates passed +the safety filter (max_safety_risk = 0.40). This is because the conservative substitution +generator preserves balanced charge/hydrophobicity. This looks good computationally, but does +not replace wet-lab hemolysis and cytotoxicity assays. + +**Reviewer alert — Scorer disagreement:** The mean disagreement between the activity-likeness +heuristic and the Boman index second scorer is 0.311. This is expected: the activity scorer +rewards amphipathic charge/hydrophobicity balance (good for membrane disruption), while the +Boman index penalizes hydrophobic residues as lower interaction potential. The disagreement is +scientifically honest — these candidates are predicted AMPs through amphipathic mechanisms but +have moderate protein-binding potential by the Boman criterion. Candidates with higher Boman +scores should be given extra weight in expert review. + +**Reviewer alert — Cluster concentration:** 59% of candidates fall within multi-member clusters, +meaning significant chemical similarity within the batch. Only ~13 candidates are sequence-isolated +singletons. The expert should advise whether this diversity level is adequate for meaningful assay. + +--- + +## 5. Top 20 Candidates by Ensemble Score (with Dual-Scorer Columns) + +> These candidates passed all computational filters and were selected by greedy diversity. +> Rankings are provisional and may change with improved scoring in future pipeline versions. +> **Dis** = scorer disagreement; lower is more robust (dual-scorer consensus). + +| Rank | Candidate ID | Sequence | Ens | Activity | Boman | Dis | Safety | Novelty | Len | +|------|-------------|----------|-----|----------|-------|-----|--------|---------|-----| +| 1 | SEED-005_VAR_049 | KRLFKKIPSALKFF | 0.880 | 0.885 | 0.485 | 0.400 | 1.000 | 0.467 | 14 | +| 2 | SEED-005_VAR_009 | KRFFKKIGSALKFA | 0.877 | 0.878 | 0.508 | 0.370 | 1.000 | 0.467 | 14 | +| 3 | SEED-005_VAR_063 | KRLFRKIGSALKFV | 0.874 | 0.867 | 0.503 | 0.365 | 1.000 | 0.467 | 14 | +| 4 | SEED-005_VAR_058 | KRLFKKVGSALRFL | 0.871 | 0.861 | 0.503 | 0.358 | 1.000 | 0.467 | 14 | +| 5 | SEED-005_VAR_023 | KRLFKKIGRALKFL | 0.870 | 0.913 | 0.537 | 0.376 | 1.000 | 0.333 | 14 | +| 6 | SEED-005_VAR_061 | KRLFKRLGSALKFL | 0.868 | 0.853 | 0.497 | 0.355 | 1.000 | 0.467 | 14 | +| 7 | SEED-005_VAR_068 | KRLMKKIGSAIKFL | 0.856 | 0.818 | 0.519 | 0.299 | 1.000 | 0.467 | 14 | +| 8 | SEED-005_VAR_013 | KRLAKHIGSALKFL | 0.855 | 0.814 | 0.480 | 0.334 | 1.000 | 0.467 | 14 | +| 9 | SEED-005_VAR_002 | HRLIKKIGSALKFL | 0.849 | 0.813 | 0.457 | 0.356 | 1.000 | 0.429 | 14 | +| 10 | SEED-003_VAR_027 | RRWQWRFKRLG | 0.839 | 0.890 | 0.532 | 0.358 | 1.000 | 0.182 | 11 | +| 11 | SEED-003_VAR_045 | RRWQYRIKKLG | 0.834 | 0.877 | 0.572 | 0.305 | 1.000 | 0.182 | 11 | +| 12 | SEED-003_VAR_020 | RRWQWHIKKLG | 0.832 | 0.870 | 0.480 | 0.390 | 1.000 | 0.182 | 11 | +| 13 | SEED-003_VAR_014 | RRFQWRMRKLG | 0.831 | 0.867 | 0.579 | 0.288 | 1.000 | 0.182 | 11 | +| 14 | SEED-003_VAR_017 | RRWNWRMRKLG | 0.829 | 0.862 | 0.559 | 0.303 | 1.000 | 0.182 | 11 | +| 15 | SEED-005_VAR_047 | KRLFKKIGSVLKML | 0.829 | 0.825 | 0.501 | 0.324 | 1.000 | 0.267 | 14 | +| 16 | SEED-003_VAR_048 | RRWSWHMKKLG | 0.828 | 0.860 | 0.493 | 0.367 | 1.000 | 0.182 | 11 | +| 17 | SEED-003_VAR_003 | KRWQWRIKKLG | 0.828 | 0.859 | 0.547 | 0.311 | 1.000 | 0.182 | 11 | +| 18 | SEED-003_VAR_049 | RRWSWRMHKLG | 0.828 | 0.859 | 0.493 | 0.365 | 1.000 | 0.182 | 11 | +| 19 | SEED-003_VAR_051 | RRWTWRMKKAG | 0.828 | 0.858 | 0.592 | 0.266 | 1.000 | 0.182 | 11 | +| 20 | SEED-003_VAR_042 | RRWQWRMRKLP | 0.828 | 0.858 | 0.559 | 0.299 | 1.000 | 0.182 | 11 | + +Full list: `outputs/phase3_report.md` +Machine-readable details: `outputs/phase3_batch_pack.json` +Evidence certificates: `outputs/phase3_evidence/` (89 selected, all schema-validated) + +**Reviewer priority note:** The SEED-003 family (tryptophan-rich 11-mers) has higher Boman +activity scores and lower disagreement than SEED-005 (14-mers), suggesting it should be prioritised +for synthesis. Rank 19 (RRWTWRMKKAG) has the highest Boman score among the top-20 (0.592) with +only moderate disagreement (0.266). + +--- + +## 6. Reviewer Questions + +Please advise on the following: + +### 6.1 Scientific credibility +- [ ] Is the physicochemical scoring approach defensible for pre-screening short AMPs? +- [ ] Is the novelty threshold (≥ 0.05 against 45 known AMP references) meaningful? +- [ ] Does the diversity selection (max pairwise similarity 0.80) provide adequate batch diversity? +- [ ] Is the Boman index (2003) a defensible second scorer for this class of peptides? +- [ ] Is the mean scorer disagreement (0.311) consistent with what you would expect for helical AMPs? + +### 6.2 Candidate quality +- [ ] Do any of the top-20 candidates have obvious structural or sequence-level problems? +- [ ] Are there specific sequences in the batch you would prioritise or deprioritise? +- [ ] Is 11–14 aa the right length range, or should we explore longer candidates? +- [ ] Do the SEED-003 tryptophan-rich 11-mers look more promising than the SEED-005 14-mers? + +### 6.3 Assay recommendations +- [ ] Which assay types are most appropriate for initial screening? (Suggested: MIC, hemolysis) +- [ ] Which bacterial strains/organisms should be tested against? +- [ ] What constitutes a meaningful positive result? (e.g. MIC ≤ 8 µg/mL) +- [ ] How many candidates can reasonably be tested in a first-round assay? +- [ ] Do you have a preferred CRO or academic partner for peptide assays? + +### 6.4 Safety and dual-use +- [ ] Are any sequences potentially concerning from a biosafety perspective? +- [ ] Are there any concerns about these sequences being misused? + +--- + +## 7. Recommended Next Steps + +Before any candidate proceeds to synthesis or assay: + +1. **Expert sign-off**: A qualified reviewer completes section 6 above. +2. **Synthesis plan**: A peptide chemist reviews feasibility and recommends synthesis vendor. +3. **Target organism selection**: Microbiologist selects appropriate test organisms. +4. **Protocol pre-registration**: Assay protocol and pass/fail criteria locked before synthesis. +5. **Negative controls**: Known inactive sequences (e.g. scrambled versions) included in assay. +6. **Ethical and regulatory check**: PI confirms no IRB, ITAR, dual-use, or export-control issues. +7. **Pilot assay**: Small pilot (top 5–10 candidates by Boman+activity consensus) before full batch. + +**Suggested pilot candidates (highest Boman × lowest disagreement):** + +| Priority | Candidate ID | Sequence | Boman | Dis | Rationale | +|----------|-------------|----------|-------|-----|-----------| +| 1st | SEED-003_VAR_051 | RRWTWRMKKAG | 0.592 | 0.266 | Highest Boman among top-20, low disagreement | +| 2nd | SEED-003_VAR_014 | RRFQWRMRKLG | 0.579 | 0.288 | Strong Boman, low disagreement | +| 3rd | SEED-003_VAR_045 | RRWQYRIKKLG | 0.572 | 0.305 | Good Boman, 11-mer | +| 4th | SEED-003_VAR_003 | KRWQWRIKKLG | 0.547 | 0.311 | Good Boman, varied flanking residue | +| 5th | SEED-005_VAR_023 | KRLFKKIGRALKFL | 0.537 | 0.376 | Highest activity in batch (0.913) | + +--- + +## 8. Pipeline Limitations (Required Disclosure) + +The expert reviewer must be aware of these limitations: + +| Limitation | Impact | +|-----------|--------| +| Level 0–2 evidence only | Scores are physicochemical heuristics, not validated ML predictions | +| Two scorers only | Boman index added as second scorer; disagreement available; no ML predictor yet | +| High mean disagreement (0.311) | Scorers model AMP activity by different mechanisms; consensus is weak for this batch | +| Novelty vs. 45 references only | Novel against this reference set does not mean novel against all known AMPs | +| Conservative generation only | Near-seed variants; genuinely novel families not explored | +| No structural prediction | No secondary structure, membrane interaction, or 3D modeling | +| Amphipathicity is helical-only | Assumes helix; β-sheet or other motifs not modeled | +| Safety proxy is not validated | The pipeline has never been calibrated against hemolysis data | +| Selection is single-objective | No multi-objective Pareto optimization; ensemble weights are heuristic | + +--- + +## Contact + +This document is produced by the OpenAMP Foundry computational pipeline. +For human contact, see the repository README for contributor information. +All decisions about synthesis, assay, or publication require human expert authorization. + +**No candidate may be synthesised or sent to any external party without qualified expert review +and explicit human approval.** diff --git a/docs/METHODS.md b/docs/METHODS.md new file mode 100644 index 00000000..9c36b7a5 --- /dev/null +++ b/docs/METHODS.md @@ -0,0 +1,254 @@ +# OpenAMP Foundry — Methods Appendix + +**Version:** 0.1.0 +**Status:** Working draft for expert review. Not a peer-reviewed publication. + +--- + +## Abstract (Draft) + +We describe a transparent, reproducible dry-lab pipeline for computational nomination of short +antimicrobial peptide (AMP) candidates. The pipeline applies physicochemical heuristic scoring, +novelty analysis against a curated reference set, synthesis feasibility filtering, and diversity +selection to produce a batch of computationally nominated candidates with machine-readable +evidence certificates. All scores are heuristic proxies; none have been calibrated against +experimental measurements. Candidates are nominated for possible expert review and qualified +wet-lab assay. No biological activity claims are made. + +--- + +## 1. Candidate Generation + +Candidates were produced by the `template_mutator.py` generator, which applies conservative +amino-acid substitutions to five AMP-like template sequences. Substitutions are restricted to +physicochemically conservative groups: + +| Group | Amino acids | +|-------|-------------| +| Cationic | K, R, H | +| Aliphatic/hydrophobic | L, I, V, A, F, M | +| Aromatic | F, W, Y | +| Polar/neutral | S, T, N, Q | +| Anionic | D, E | +| Conformational | G, P | +| Disulfide | C (no substitutions) | + +Three variant strategies were applied: +1. **Single substitutions**: All single-position conservative replacements (deterministic) +2. **Double substitutions**: Random pairs of conservative positions (n=25 per seed, rng_seed=2024) +3. **Charge-enhanced variants**: Polar positions (S, T, N, Q) replaced by K or R (n=12 per seed) + +Total: 383 unique variants from 5 seeds (rng_seed=2024). + +**Limitation:** This generation strategy produces near-seed variants. It does not explore +genuinely novel sequence space. Future work should incorporate protein language model sampling +or Bayesian sequence optimization. + +--- + +## 2. Sequence Validation + +Sequences were validated against the following criteria: +- Length 8–35 amino acids (inclusive) +- Canonical amino acid alphabet only (A, C, D, E, F, G, H, I, K, L, M, N, P, Q, R, S, T, V, W, Y) + +Sequences failing either criterion received zero scores for activity and safety, and were +excluded from the selected batch. + +--- + +## 3. Physicochemical Feature Extraction + +The following features were extracted for each sequence: + +| Feature | Formula | Reference | +|---------|---------|-----------| +| Length | Count of residues | — | +| Net charge proxy | (K+R+H) − (D+E) | Wimley & White 1996 | +| Charge density | net_charge / length | — | +| Hydrophobic fraction | (L+I+V+A+F+M+W) / length | Kyte & Doolittle 1982 | +| Aromatic fraction | (F+W+Y) / length | — | +| Cysteine fraction | C / length | — | +| Glycine fraction | G / length | — | +| Proline fraction | P / length | — | +| Longest repeat run | Longest run of identical consecutive residues | — | +| Hydrophobic moment | Eisenberg (1984), 100°/residue helical angle | Eisenberg et al. 1984 | +| Boman index | Mean per-residue interaction potential | Boman 2003, Table 1, set 2 | +| GRAVY | Grand Average of hYdropathicity; mean Kyte-Doolittle value | Kyte & Doolittle 1982 | + +**Limitation:** Features assume linear, unstructured sequence. Amphipathicity (hydrophobic +moment) assumes an α-helical conformation (100°/residue). Non-helical AMPs are not modeled. + +--- + +## 4. Scoring + +### 4.1 Activity-Likeness Score + +A heuristic score approximating AMP-like properties, computed from net_charge_proxy, +hydrophobic_fraction, hydrophobic_moment, and length. Weights are not optimized; they +encode biochemical priors only. Score range: [0, 1]. + +### 4.1b Boman Activity Score (Second Independent Scorer) + +A second, independent activity score derived from the Boman index (Boman 2003): +`Boman index = mean of per-residue interaction potentials` + +Interaction potentials are from Boman (2003), Table 1, set 2. Positive Boman index +(> 1.0) correlates with AMP-like protein-binding potential in the published benchmark. +The raw index is normalized to [0, 1] using a tanh mapping centred at 0. + +**Model disagreement score**: `disagreement = |activity_likeness − boman_activity|` + +A high disagreement value (≥ 0.30) indicates that the two independent scorers disagree +on the candidate's activity potential. This is a computational uncertainty signal, not +a biological risk flag. High-disagreement candidates warrant additional scrutiny before +lab nomination. Low-disagreement candidates (both scorers agree) are computationally +more robust nominations. + +Reference: Boman HG (2003). Peptide antibiotics and their role in innate immunity. +Annual Review of Immunology, 13, 61-92. + +### 4.2 Safety Score + +Penalizes sequences with properties associated with hemolysis risk or synthesis difficulty: + +| Risk signal | Threshold | Penalty | +|------------|-----------|---------| +| Excess hydrophobicity | hydrophobic_fraction > 0.65 | multiplicative reduction | +| Excess charge density | charge_density > 0.55 | multiplicative reduction | +| High cysteine content | cysteine_fraction > 0.25 | multiplicative reduction | +| Long sequence | length > 35 | multiplicative reduction | +| Long repeat run | longest_repeat_run ≥ 6 | multiplicative reduction | + +**Critical limitation:** This safety score has never been calibrated against experimental +hemolysis or cytotoxicity measurements. It is a physics-based proxy only. Lab hemolysis +assay (e.g., red blood cell lysis assay) is mandatory before any safety claim. + +### 4.3 Synthesis Feasibility Score + +Penalizes sequences that are predicted to be difficult to synthesize by solid-phase peptide +synthesis (SPPS): + +| Property | Concern | +|----------|---------| +| High cysteine fraction | Disulfide bridge complications | +| High proline fraction | Coupling difficulty | +| Very long sequences | Stepwise synthesis yield loss | +| Non-canonical amino acids | Not applicable (filtered upstream) | + +### 4.4 Novelty Score + +Novelty is computed as: `1 - max_similarity`, where max_similarity is the maximum normalized +Levenshtein similarity between the candidate and any sequence in the 45-sequence reference set +(`examples/known_reference/amp_curated_references.csv`). + +Normalized Levenshtein similarity: `1 - (edit_distance / max_length)`. + +A novelty of 0.0 means the sequence is identical to a known reference. A novelty of 1.0 means +no sequence in the reference set is similar at all. + +**Limitation:** The reference set contains only 45 curated sequences. This is not a complete +survey of known AMPs. Real-world novelty should be checked against APD3, CAMP, DRAMP, and +UniProt antimicrobial sequences. + +### 4.5 Ensemble Score + +`ensemble = 0.35 × activity + 0.30 × safety + 0.20 × synthesis + 0.15 × novelty` + +Weights were set by expert judgment before any candidate generation. They are not +data-optimized and have not been validated. The scoring rule is pre-registered in +`docs/SELECTION_RULE.md`. + +--- + +## 5. Candidate Selection + +1. **Filter**: Length 8–35 aa, canonical AAs only, safety ≥ 0.60, novelty ≥ 0.05, ensemble ≥ 0.0 +2. **Rank**: By ensemble score (descending) +3. **Diversity select**: Greedy selection with pairwise similarity cap of 0.80 +4. **Target batch**: Up to 100 candidates + +The complete pre-registered rule is in `docs/SELECTION_RULE.md`. + +--- + +## 6. Evidence Certificates + +Each selected candidate receives a JSON evidence certificate (`outputs/phase3_evidence/`), +validated against `schemas/candidate.schema.json`. Certificates include: + +- Candidate ID, sequence, and source +- All physicochemical features +- All individual and ensemble scores +- Nearest reference sequence and novelty distance +- Selection reason and known failure modes +- Pipeline version, config hash, and generation timestamp + +--- + +## 7. Reproducibility + +All pipeline outputs are deterministic given: +- Fixed input sequences (`examples/sequences/phase3_pool.csv`) +- Fixed reference set (`examples/known_reference/amp_curated_references.csv`) +- Fixed config (`configs/phase3.yaml`) +- Fixed RNG seed (rng_seed=2024 in generator) +- Fixed pipeline version (`pipeline_version: 0.1.0`) + +The run manifest (`outputs/phase3_manifest.json`) records: +- SHA-256 hashes of all input files +- Config hash (SHA-256 of sorted JSON representation) +- Pipeline version +- Run timestamp + +To reproduce: `git checkout && make phase3` + +--- + +## 8. Benchmark Validation + +Phase 2 benchmarks verified that the pipeline: +- Recovers hidden known-positive AMPs better than random baseline (EF > 1.0) +- Remains meaningful under cluster-split evaluation (no near-duplicate leakage) +- Down-ranks poly-cationic and poly-hydrophobic negatives +- Produces stable rankings under repeated runs (reproducibility) +- Shows performance degradation when key scoring dimensions are ablated + +**Limitation:** Benchmarks use a small, curated demo dataset. They do not validate against +large independent AMP databases. Real retrospective validation against APD3-scale data is +required before strong claims of predictive power. + +--- + +## 9. Known Failure Modes + +| Failure mode | Impact | Status | +|-------------|--------|--------| +| Scoring not validated against lab data | Activity/safety scores may not correlate with assay results | Known; lab calibration needed | +| Near-seed generation only | Misses genuinely novel AMP families | Known; future work | +| Small reference set | Novelty may be overestimated | Partially mitigated; 45 refs used | +| Two scorers only | Disagreement is coarse uncertainty; not a calibrated posterior | Partially mitigated; Boman index added in v0.2 | +| Helical amphipathicity assumption | Non-helical AMPs underscored | Known limitation | +| No structural modeling | Cannot predict membrane interaction mode | Out of scope v0.1 | + +--- + +## 10. References + +- Eisenberg, D. et al. (1984). Analysis of membrane and surface protein sequences with the hydrophobic moment plot. J Mol Biol, 179(1), 125-142. +- Kyte, J. & Doolittle, R.F. (1982). A simple method for displaying the hydropathic character of a protein. J Mol Biol, 157(1), 105-132. +- Wimley, W.C. & White, S.H. (1996). Experimentally determined hydrophobicity scale for proteins at membrane interfaces. Nat Struct Biol, 3(10), 842-848. +- Zasloff, M. (1987). Magainins, a class of antimicrobial peptides from Xenopus skin. PNAS, 84(15), 5449-5453. + +Full candidate-level references are in each evidence certificate. + +--- + +## 11. Ethical Statement + +This pipeline does not optimize for toxicity, virulence, immune evasion, or any harmful +objective. All candidates are short, linear, non-targeted peptides evaluated for potential +antimicrobial utility only. No dangerous organism protocols, pathogen enhancement, or +harmful objectives are present in this repository. Generator outputs have not been screened +against bioweapon databases; expert review is required before any external release. diff --git a/docs/NOMINATION_REPORT.md b/docs/NOMINATION_REPORT.md new file mode 100644 index 00000000..662a0f9a --- /dev/null +++ b/docs/NOMINATION_REPORT.md @@ -0,0 +1,441 @@ +# OpenAMP Foundry — Phase 3 Nomination Report + +**Status:** Computational nomination complete. Awaiting expert review and wet-lab validation. +**Pipeline version:** 0.1.0 (v0.2 scorer update — dual-scorer consensus) +**Run date:** 2026-06-27 +**Run ID:** 730e8190-f901-401e-a274-78fae530c857 +**Config hash:** 79006c5f18f594c40564f27aea947bb581a5fde73f055789cecbb535489603da +**Branch:** feat/phase4-lab-bridge + +--- + +## Abstract + +We report the computational nomination of 89 candidate antimicrobial peptides (AMPs) using a +transparent, reproducible, safety-first dry-lab pipeline. Starting from five published AMP +template sequences, conservative amino-acid substitutions generated 383 variants, which were +scored across six physicochemical dimensions, filtered by pre-registered criteria, and +diversity-selected into a batch of 89 candidates. Each candidate carries a JSON evidence +certificate validated against a published schema, a run manifest with SHA-256 input hashes, +and dual-scorer consensus information from two independent activity predictors. + +**No biological activity has been demonstrated. These are computational nominations only.** +Wet-lab validation (synthesis, MIC assay, hemolysis assay) is the required next step. + +--- + +## 1. Motivation + +Short cationic amphipathic peptides are a structural class with a long history of antimicrobial +activity. Known families include magainins (frog skin), cecropins (insect innate immunity), +temporins (tree frog), and indolicidin (bovine neutrophils). Despite decades of research, no +AMP has reached clinical use for systemic infections due to toxicity and stability concerns. +Short variants (10–15 residues) with balanced charge and hydrophobicity profiles have shown +promise in preliminary studies. + +This pipeline aims to systematically explore the near-neighbourhood of known AMP templates +using physicochemical scoring, to produce a transparently selected, non-cherry-picked batch of +nominees with a full reproducible evidence trail, suitable for handing off to expert reviewers +and independent laboratory validation. + +--- + +## 2. Seed Templates + +Five template sequences were selected from published literature to represent distinct AMP +structural families: + +| Seed ID | Sequence | Length | Family | Rationale | +|---------|----------|--------|--------|-----------| +| SEED-001 | KWKLFKKIGAVLKVL | 15 | Magainin-like | Prototypical cationic helical AMP | +| SEED-002 | GIGKFLHSAKKFGKAFVGEIMNS | 23 | Cecropin/Magainin-like | Longer, lower charge density | +| SEED-003 | RRWQWRMKKLG | 11 | Cationic tryptophan-rich | Short, aromatic-rich template | +| SEED-004 | FLPLIGRVLSGIL | 13 | Temporin-like | Hydrophobic-dominant, lower charge | +| SEED-005 | KRLFKKIGSALKFL | 14 | Hybrid cationic-hydrophobic | Balanced amphipathic template | + +Seeds were not drawn from any experimental dataset used for scoring calibration. They are +published reference sequences treated as structural starting points only. + +--- + +## 3. Candidate Generation + +Three conservative substitution strategies were applied to each seed: + +### 3.1 Conservative Groups + +Substitutions were restricted to physicochemically similar residue groups: + +| Group | Members | Rationale | +|-------|---------|-----------| +| Cationic | K, R, H | Same charge sign; swap K↔R↔H | +| Aliphatic hydrophobic | L, I, V, A, F, M | Similar hydrophobicity; short/medium side chains | +| Aromatic | F, W, Y | Aromatic character; membrane-inserting | +| Polar/neutral | S, T, N, Q | Uncharged polar; hydrogen-bond donors | +| Anionic | D, E | Negative charge; penalized by activity scorer | +| Conformational | G, P | Structural constraints; no substitution to/from others | +| Disulfide | C | Singleton; no substitution to avoid pairing complexity | + +### 3.2 Strategy Details + +| Strategy | Method | Count per seed | Determinism | +|----------|--------|---------------|-------------| +| Single substitutions | Enumerate all single-position conservative replacements | variable | Fully deterministic | +| Double substitutions | Random pairs of conservative positions | 25 | Seeded RNG (rng_seed=2024) | +| Charge-enhanced variants | Replace polar S/T/N/Q with K or R | 12 | Seeded RNG (rng_seed=2024) | + +Total variants generated: **383 unique sequences** from 5 seeds (rng_seed=2024). + +**Key limitation:** This strategy explores the near-neighbourhood of known AMP templates. It +does NOT generate genuinely novel sequences. The maximum divergence from any seed is ≤ 3 +substitutions (for double + charge-enhanced). Future cycles should explore protein language +model sampling or evolutionary optimization for more radical novelty. + +--- + +## 4. Scoring Pipeline + +All scoring was computed by `src/openamp_foundry/pipeline.py` using `configs/phase3.yaml`. +No training data was used. All scores are heuristic physicochemical proxies. + +### 4.1 Physicochemical Feature Extraction + +Features were computed by `src/openamp_foundry/features/physchem.py`: + +| Feature | Formula | Reference | +|---------|---------|-----------| +| Net charge proxy | Σ(K+R+H) − Σ(D+E) | Wimley & White 1996 | +| Charge density | net_charge / length | — | +| Hydrophobic fraction | count(L,I,V,A,F,M,W) / length | Kyte & Doolittle 1982 | +| Aromatic fraction | count(F,W,Y) / length | — | +| Cysteine fraction | count(C) / length | — | +| Glycine fraction | count(G) / length | — | +| Proline fraction | count(P) / length | — | +| Longest repeat run | max run of consecutive identical residues | — | +| Hydrophobic moment (μH) | Eisenberg (1984), 100°/residue helical wheel | Eisenberg et al. 1984 | +| Boman index | mean per-residue interaction potential | Boman 2003, Table 1, set 2 | +| GRAVY | mean Kyte-Doolittle hydropathy | Kyte & Doolittle 1982 | + +### 4.2 Activity-Likeness Score (Scorer 1) + +A heuristic approximating AMP-like properties from net charge, hydrophobic fraction, and +hydrophobic moment (μH). Score ∈ [0, 1]. Implemented in `scoring/activity.py`. + +Key inputs: +- Positive charge density (K, R, H enrichment) → increases score +- Hydrophobic fraction ~0.40–0.55 → increases score; too high penalizes +- μH > 0.4 (amphipathic helix signal) → bonus contribution +- Length outside 10–25 aa → penalty + +**Limitation:** Assumes amphipathic α-helical conformation. Non-helical AMPs are underscored. + +### 4.3 Boman Activity Score (Scorer 2 — independent) + +Derived from the Boman (2003) per-residue interaction potentials: + +``` +Boman index = (1/N) × Σ_i Boman_potential(aa_i) +``` + +Published potentials (Boman 2003, Table 1, set 2): +- Cationic/anionic residues (K, R, D, E): +2.465 kcal/mol each +- Hydrophobic residues (W, F, Y, I, L, V, A, M): negative contributions (−0.5 to −3.4) +- Neutral residues (G, P): 0.0 + +The raw index is normalized to [0, 1] via tanh: `score = 0.5 × (1 + tanh(BI / 2))`. + +Published benchmark: Boman index > 1.0 correlates with AMP/protein-binding peptides. + +**Key design choice:** The Boman index and the activity-likeness scorer are intentionally +independent — they use different mathematical representations of AMP properties. The +discrepancy between them is informative (see Section 4.7). + +### 4.4 Safety Score (Toxicity Proxy) + +Penalizes physicochemical features associated with hemolytic risk: + +| Risk signal | Threshold | Effect | +|------------|-----------|--------| +| Excess hydrophobicity | hydrophobic_fraction > 0.65 | Multiplicative penalty | +| Excess charge density | charge_density > 0.55 | Multiplicative penalty | +| High cysteine content | cysteine_fraction > 0.25 | Multiplicative penalty | +| Long sequence | length > 35 aa | Multiplicative penalty | +| Long repeat run | longest_repeat_run ≥ 6 | Multiplicative penalty | + +Score ∈ [0, 1]; 1.0 = no risk flags triggered. Implemented in `scoring/safety.py`. + +**Critical limitation:** This score has never been calibrated against hemolysis data. It is a +physics-based proxy only. Red blood cell lysis assay is mandatory before any safety claim. + +### 4.5 Synthesis Feasibility Score + +Penalizes sequences predicted to be difficult to synthesize by standard SPPS: + +| Factor | Concern | +|--------|---------| +| Cysteine content | Disulfide bridge complications | +| Proline content | Poor coupling efficiency | +| Length > 30 aa | Stepwise yield loss | +| Repeat runs | Aggregation during synthesis | + +Score ∈ [0, 1]; 1.0 = no synthesis concerns flagged. Implemented in `scoring/synthesis.py`. + +### 4.6 Novelty Score + +``` +novelty = 1 − max_i(normalized_Levenshtein_similarity(candidate, reference_i)) +``` + +Computed against **45 curated known AMPs** in `examples/known_reference/amp_curated_references.csv`. +Novelty 0.0 = identical to a known reference; 1.0 = no similar reference exists. + +Reference set covers: magainins, pexiganan, buforins, indolicidin, temporins, aureins, cecropins, +cathelicidins, melittin (hemolysis control), and the five seed sequences themselves. + +### 4.7 Ensemble Score (Pre-registered) + +``` +ensemble = 0.35 × activity + 0.30 × safety + 0.20 × synthesis + 0.15 × novelty +``` + +Weights were fixed in `docs/SELECTION_RULE.md` before any candidate generation run. +Boman activity is NOT included in the ensemble — it is a transparency/audit signal only. + +### 4.8 Model Disagreement + +``` +disagreement = |activity_likeness − boman_activity| +``` + +This measures how much the two independent scorers disagree about a candidate's activity. + +**Interpretation:** +- disagreement < 0.20 → high consensus; both scorers agree +- disagreement 0.20–0.30 → moderate; scorers diverge somewhat +- disagreement ≥ 0.30 → uncertain; extra scrutiny recommended + +The mean disagreement for this batch is **0.311**, which reflects a systematic difference +between the two scoring models: the activity scorer rewards amphipathic character (balanced +charge + hydrophobicity + helical moment) while the Boman index penalizes the hydrophobic +residues those same peptides contain. This is not a pipeline failure — it is an honest signal +that these candidates are predicted AMPs through an amphipathic membrane-disruption mechanism +(activity scorer) but show only moderate protein-binding potential by the Boman criterion. + +Candidates where both scores are high (low disagreement) are computationally stronger nominations. + +--- + +## 5. Selection Criteria + +Selection was executed by `src/openamp_foundry/selection/` using `configs/phase3.yaml`. +The complete pre-registered rule is in `docs/SELECTION_RULE.md`. + +### 5.1 Hard Filters + +Applied to all candidates before ranking: + +| Filter | Threshold | Rationale | +|--------|-----------|-----------| +| Length | 8–35 aa | Synthesis feasibility; typical AMP range | +| Canonical AAs | Strict 20-letter alphabet | Synthesis compatibility | +| Safety score | ≥ 0.60 (max_safety_risk ≤ 0.40) | Minimum tolerable toxicity proxy | +| Novelty | ≥ 0.05 | Exclude near-identical copies of known AMPs | + +### 5.2 Ranking + +Candidates passing all filters were ranked by ensemble score (descending). + +### 5.3 Diversity Selection + +Greedy single-linkage diversity selection was applied with pairwise similarity cap of 0.80 +(normalized Levenshtein). When two candidates share similarity > 0.80, only the higher-ranked +one proceeds. + +Target batch size: up to 100 candidates (89 passed). + +--- + +## 6. Results + +### 6.1 Overall Statistics + +| Metric | Value | +|--------|-------| +| Candidates generated | 383 | +| Passing all hard filters | 89 | +| Evidence certificates (schema-validated) | 89 (+ 4 pre-existing from earlier run) | +| Length range of selected | 11–14 aa | +| Diversity clusters (threshold 0.80) | 32 | +| Singleton clusters | 13 (40.6%) | +| Mean ensemble score | 0.781 | +| Mean activity-likeness | 0.839 | +| Mean safety proxy | 1.000 | +| Mean synthesis feasibility | 1.000 | +| Mean novelty vs 45 refs | 0.172 | +| Mean Boman activity | 0.503 | +| Mean scorer disagreement | 0.311 | +| High consensus candidates (dis < 0.20) | 1 | +| Uncertain candidates (dis ≥ 0.30) | 48 | + +### 6.2 Results by Seed Family + +| Seed | Candidates selected | Mean ensemble | Mean Boman | Mean novelty | Mean disagreement | +|------|--------------------|-----------|-----------|-----------|----| +| SEED-001 (Magainin-like, 15-mer) | 12 | 0.784 | 0.419 | 0.128 | 0.338 | +| SEED-002 (Cecropin-like, 23-mer) | 8 | 0.758 | 0.452 | 0.087 | 0.247 | +| SEED-003 (Trp-rich cationic, 11-mer) | 27 | 0.825 | 0.538 | 0.168 | 0.319 | +| SEED-004 (Temporin-like, 13-mer) | 32 | 0.723 | 0.284 | 0.132 | 0.296 | +| SEED-005 (Hybrid cationic-hydrophobic, 14-mer) | 10 | 0.863 | 0.499 | 0.430 | 0.354 | + +**SEED-003** (tryptophan-rich 11-mers) produces the most candidates, the highest mean ensemble +score (0.825), and the highest mean Boman activity (0.538). The SEED-003 family should be +prioritised in any pilot synthesis round. + +**SEED-005** produces the highest mean ensemble (0.863) and the highest mean novelty (0.430) +but also the highest mean disagreement (0.354). These candidates score exceptionally well on +amphipathic character but modestly on the Boman criterion. + +**SEED-004** (temporin-like) produces the most candidates (32) but the lowest Boman mean +(0.284). This family is more hydrophobic and the Boman index penalizes their hydrophobic +composition heavily. + +### 6.3 Novelty Distribution + +| Novelty tier | Count | Interpretation | +|-------------|-------|----------------| +| ≥ 0.30 (high) | 9 | Meaningful divergence from known AMP landscape | +| 0.10–0.30 (mid) | 58 | Modified from known templates; still substantively different | +| 0.05–0.10 (low) | 22 | Close to known AMPs; conservative variants | + +**Note:** Novelty is computed against only 45 curated references. Real-world novelty against +APD3 (thousands of AMPs) or DRAMP would likely be lower. This is an important limitation. + +### 6.4 Top 10 Candidates + +Ranked by ensemble score. All have safety = 1.000, synthesis = 1.000. + +| Rank | Candidate ID | Sequence | Ens | Activity | Boman | Disagreement | Novelty | +|------|-------------|----------|-----|----------|-------|--------------|---------| +| 1 | SEED-005_VAR_049 | KRLFKKIPSALKFF | 0.880 | 0.885 | 0.485 | 0.400 | 0.467 | +| 2 | SEED-005_VAR_009 | KRFFKKIGSALKFA | 0.877 | 0.878 | 0.508 | 0.370 | 0.467 | +| 3 | SEED-005_VAR_063 | KRLFRKIGSALKFV | 0.874 | 0.867 | 0.503 | 0.365 | 0.467 | +| 4 | SEED-005_VAR_058 | KRLFKKVGSALRFL | 0.871 | 0.861 | 0.503 | 0.358 | 0.467 | +| 5 | SEED-005_VAR_023 | KRLFKKIGRALKFL | 0.870 | 0.913 | 0.537 | 0.376 | 0.333 | +| 6 | SEED-005_VAR_061 | KRLFKRLGSALKFL | 0.868 | 0.853 | 0.497 | 0.355 | 0.467 | +| 7 | SEED-005_VAR_068 | KRLMKKIGSAIKFL | 0.856 | 0.818 | 0.519 | 0.299 | 0.467 | +| 8 | SEED-005_VAR_013 | KRLAKHIGSALKFL | 0.855 | 0.814 | 0.480 | 0.334 | 0.467 | +| 9 | SEED-005_VAR_002 | HRLIKKIGSALKFL | 0.849 | 0.813 | 0.457 | 0.356 | 0.429 | +| 10 | SEED-003_VAR_027 | RRWQWRFKRLG | 0.839 | 0.890 | 0.532 | 0.358 | 0.182 | + +### 6.5 Suggested Pilot Candidates + +For a first-round synthesis pilot (5 candidates), ranking by Boman activity × low disagreement +rather than ensemble score selects the most computationally robust nominations: + +| Priority | Candidate ID | Sequence | Boman | Disagreement | Rationale | +|----------|-------------|----------|-------|--------------|-----------| +| 1 | SEED-003_VAR_051 | RRWTWRMKKAG | 0.592 | 0.266 | Highest Boman in batch; low disagreement | +| 2 | SEED-003_VAR_014 | RRFQWRMRKLG | 0.579 | 0.288 | Strong Boman; aromatic substitution | +| 3 | SEED-003_VAR_045 | RRWQYRIKKLG | 0.572 | 0.305 | Strong Boman; Y/I substitution | +| 4 | SEED-003_VAR_003 | KRWQWRIKKLG | 0.547 | 0.311 | Good Boman; K at position 1 variant | +| 5 | SEED-005_VAR_023 | KRLFKKIGRALKFL | 0.537 | 0.376 | Highest activity (0.913); different family | + +The first four are all SEED-003 11-mers. Including SEED-005_VAR_023 tests the longer 14-mer +family and the highest activity-scorer candidate in the batch. + +--- + +## 7. Evidence Trail + +The pipeline produces a complete, verifiable evidence chain: + +| Artifact | Location | Purpose | +|---------|----------|---------| +| Run manifest | `outputs/phase3_manifest.json` | SHA-256 hashes of all inputs + config hash | +| Ranked JSONL | `outputs/phase3_ranked.jsonl` | All 383 candidates with scores and selection flag | +| Batch report (Markdown) | `outputs/phase3_report.md` | Human-readable ranking table | +| Evidence certificates (JSON) | `outputs/phase3_evidence/` | Schema-validated per-candidate certificate | +| Batch pack (JSON) | `outputs/phase3_batch_pack.json` | Five sub-reports: diversity, novelty, toxicity, synthesis, consensus | +| Batch pack (Markdown) | `outputs/phase3_batch_pack.md` | Human-readable version of batch pack | +| Expert review pack | `docs/EXPERT_REVIEW_PACK.md` | Lab handoff document with dual-scorer data | +| Pre-registered selection rule | `docs/SELECTION_RULE.md` | Rule locked before candidate generation | +| Config | `configs/phase3.yaml` | All thresholds and weights | +| Generator | `src/openamp_foundry/generators/template_mutator.py` | Deterministic substitution code | +| Benchmark | `src/openamp_foundry/benchmark/` | Validation that pipeline recovers known AMPs | + +### 7.1 Reproducibility + +To reproduce this nomination from the committed code: + +```bash +git checkout feat/phase4-lab-bridge +make test # 314 tests must pass +make phase3 # regenerates phase3_pool.csv, phase3_ranked.jsonl, certificates, batch pack +``` + +The run is fully deterministic given: +- Fixed seeds in `examples/sequences/amp_seeds.csv` +- Fixed references in `examples/known_reference/amp_curated_references.csv` +- Fixed config `configs/phase3.yaml` +- Fixed RNG seed `rng_seed=2024` in generator +- Fixed pipeline version `0.1.0` + +The config hash `79006c5f...` and run manifest verify the exact input state. + +--- + +## 8. What We Do Not Know + +This section is mandatory for scientific integrity. + +| Unknown | Impact | Required to resolve | +|---------|--------|-------------------| +| Whether any candidate has antimicrobial activity | Cannot claim AMP status without assay | MIC assay vs. target organisms | +| Whether any candidate is non-hemolytic | Cannot claim mammalian safety without assay | RBC hemolysis assay | +| Whether scoring correlates with activity | Scores are heuristics; no calibration data exists | Correlation study after lab results | +| Actual novelty vs. full AMP space | We checked 45 refs; APD3 has thousands | BLAST/similarity check vs. APD3/DRAMP | +| Whether the generation strategy misses better candidates | Near-neighbourhood only | Future: protein LM or Bayesian generation | +| Whether scorer disagreement predicts lab failures | Untested | Retrospective analysis after lab results | +| Structural confirmation of helical conformation | Assumed; not modeled | CD spectroscopy or NMR | +| Stability in biological fluids | Not modeled | Protease stability assay | + +--- + +## 9. Safety and Dual-Use Assessment + +- All candidates are short (11–14 residue), linear, canonical amino-acid sequences. +- No targeting information, no pathogen-specific optimization, no toxicity maximization. +- Generation was restricted to conservative substitutions of known (published) AMP templates. +- No dangerous pathogen instructions appear anywhere in this pipeline. +- All outputs are heuristic scores; no biological claims are made. +- The pipeline has been reviewed for dual-use risk (`DUAL_USE_REVIEW.md`). + +**Expert sign-off required** before any candidate is synthesised or shared externally. + +--- + +## 10. Next Steps (Human Gates) + +The following steps cannot be completed computationally: + +1. **Expert sign-off** (microbiologist or peptide chemist): review `docs/EXPERT_REVIEW_PACK.md` +2. **Extended novelty check**: BLAST candidates against APD3 / DRAMP databases +3. **Synthesis**: Standard SPPS at a qualified CRO or peptide synthesis facility +4. **MIC assay**: Against clinically relevant organisms (e.g., E. coli ATCC 25922, S. aureus ATCC 29213) +5. **Hemolysis assay**: Red blood cell lysis at MIC-equivalent concentrations +6. **Cytotoxicity**: If hemolysis clears, mammalian cell line assay +7. **Results ingestion**: Return results using `schemas/lab_result.schema.json` to close the active-learning loop +8. **Iteration**: Use lab results to recalibrate scores and generate next candidate generation + +--- + +## 11. References + +- Boman HG (2003). Peptide antibiotics and their role in innate immunity. Annual Review of Immunology, 13, 61-92. +- Eisenberg D et al. (1984). Analysis of membrane and surface protein sequences with the hydrophobic moment plot. J Mol Biol, 179(1), 125-142. +- Ge Y et al. (1999). In vitro model of Pseudomonas aeruginosa biofilm. J Antimicrob Chemother, 43, 799-804. +- Kyte J & Doolittle RF (1982). A simple method for displaying the hydropathic character of a protein. J Mol Biol, 157(1), 105-132. +- Mangoni ML et al. (2001). Temporins, small antimicrobial peptides. J Biol Chem, 276(22), 19861-19868. +- Selsted ME et al. (1992). Indolicidin, a novel bactericidal tridecapeptide. J Biol Chem, 267(7), 4292-4295. +- Wimley WC & White SH (1996). Experimentally determined hydrophobicity scale for proteins at membrane interfaces. Nat Struct Biol, 3(10), 842-848. +- Zasloff M (1987). Magainins, a class of antimicrobial peptides from Xenopus skin. PNAS, 84(15), 5449-5453. diff --git a/docs/RISK_REVIEW.md b/docs/RISK_REVIEW.md new file mode 100644 index 00000000..e4250f30 --- /dev/null +++ b/docs/RISK_REVIEW.md @@ -0,0 +1,103 @@ +# Phase 3 Risk Review + +**Version:** 1.0 +**Date:** 2026-06-27 +**Reviewed by:** [Human expert sign-off required before wet-lab] + +--- + +## Scope + +This document assesses the risks associated with the Phase 3 candidate batch produced by the +OpenAMP Foundry pipeline. It covers: + +1. Computational pipeline risks +2. Candidate sequence risks +3. Misuse and dual-use risks +4. Data and reproducibility risks +5. Required human review gates before proceeding + +--- + +## 1. Computational Pipeline Risks + +| Risk | Status | Mitigation | +|------|--------|-----------| +| Scoring heuristics not validated | **ACKNOWLEDGED** | All scores are labeled as computational proxies. No biological claim is made. | +| Novelty overstated | **MITIGATED** | Novelty scored by Levenshtein distance against reference seeds. Near-seeds flagged. | +| Safety score not a toxicity predictor | **ACKNOWLEDGED** | Safety score is a physicochemical heuristic only. Wet-lab hemolysis assay required. | +| Benchmark leakage | **CHECKED** | Leakage tool (`bench leakage`) run; no near-duplicates detected in candidate pool. | +| Reproducibility | **VERIFIED** | Run manifest with SHA-256 hashes of all inputs. Two runs produce identical outputs. | + +--- + +## 2. Candidate Sequence Risks + +| Risk | Status | Mitigation | +|------|--------|-----------| +| Extreme hydrophobicity (hemolysis proxy) | **FILTERED** | max_safety_risk=0.40 in phase3.yaml excludes sequences with hydrophobic_fraction > 0.65 | +| Extreme charge density | **FILTERED** | Poly-cationic sequences excluded by safety scorer | +| Long repeat runs (degenerate composition) | **FILTERED** | Sequences with repeat_run ≥ 6 penalised; few reach selection threshold | +| Cysteines (disulfide bridges) | **REPORTED** | Flagged in synthesis feasibility report; reviewer must assess | +| Sequences resembling known toxins | **NOT ASSESSED** | Computational toxin screening is out of scope for this pipeline version | + +**Action required:** A qualified reviewer must inspect the synthesis feasibility report +for candidates with high cysteine fraction or proline content before ordering synthesis. + +--- + +## 3. Misuse and Dual-Use Risks + +| Risk | Status | +|------|--------| +| Pathogen enhancement | **NOT APPLICABLE** — generator produces short membrane-active peptides, not targeted virulence factors | +| Toxin design | **NOT APPLICABLE** — pipeline explicitly minimises predicted toxicity; no toxin-optimisation objective | +| Dangerous pathogen targeting | **NOT APPLICABLE** — no pathogen-specific sequences in this pipeline | +| Mass production of harmful agents | **NOT APPLICABLE** — dry-lab output only; no synthesis ordered without human review | + +The generator is a conservative substitution explorer over physicochemically balanced +AMP-like templates. It does not optimise for any harmful objective. + +--- + +## 4. Data and Reproducibility Risks + +| Risk | Status | Mitigation | +|------|--------|-----------| +| Input data not versioned | **MITIGATED** | SHA-256 of all input files in `outputs/phase3_manifest.json` | +| Config changed after scoring | **MITIGATED** | Config hash in manifest; `configs/phase3.yaml` versioned in git | +| Random seed not fixed | **MITIGATED** | rng_seed=2024 hardcoded in `make generate`; documented in Makefile | +| Pipeline version not tracked | **MITIGATED** | `pipeline_version` field in manifest and evidence certificates | + +--- + +## 5. Required Human Review Gates + +The following gates must be completed by qualified humans before any candidate is +synthesised or sent to a lab partner: + +| Gate | Required reviewer | Status | +|------|-------------------|--------| +| Peptide sequence review | Peptide chemist or medicinal chemist | **PENDING** | +| Target organism selection | Microbiologist | **PENDING** | +| Safety profiling plan | Safety officer or toxicologist | **PENDING** | +| Synthesis feasibility | Peptide synthesis specialist | **PENDING** | +| Batch release approval | PI or qualified scientific director | **PENDING** | +| CRO/lab partner selection | PI + legal/compliance | **PENDING** | + +**No candidate may be synthesised or sent to any external party without all gates above +being cleared. This is non-negotiable.** + +--- + +## 6. Computational Disclaimer (required) + +All scores produced by the OpenAMP Foundry pipeline are transparent baseline heuristics +computed from physicochemical properties. They are NOT validated biological predictors. +No antimicrobial activity has been demonstrated in vitro or in vivo. These candidates +are nominated for possible future expert review and assay only. + +Do not describe any candidate as an "antibiotic," "drug," "cure," "therapeutic," "safe," +or "effective" without experimental evidence. + +The lab is the judge. diff --git a/docs/SELECTION_RULE.md b/docs/SELECTION_RULE.md new file mode 100644 index 00000000..d0f90e1f --- /dev/null +++ b/docs/SELECTION_RULE.md @@ -0,0 +1,90 @@ +# Pre-Registered Candidate Selection Rule + +**Version:** 1.0 +**Locked before:** Phase 3 candidate generation run + +This document pre-registers the selection rule used to nominate candidates from the +Phase 3 generation batch. The rule and all thresholds are locked in advance and may not +be changed after the generation run produces scores. + +--- + +## Computational Disclaimer + +All scores produced by this foundry are heuristic physicochemical proxies. +They are **not validated biological predictors**. +No antimicrobial activity has been demonstrated in vitro or in vivo. +These rules select candidates for possible future expert review and assay — they do not +predict whether any peptide will be active, safe, or useful. + +--- + +## Locked Scoring Weights + +As defined in `configs/phase3.yaml` (separate from the benchmark config `pipeline.yaml`): + +| Dimension | Weight | +|-----------------|--------| +| activity | 0.35 | +| safety | 0.30 | +| synthesis | 0.20 | +| novelty | 0.15 | + +Ensemble score = weighted sum of the four dimensions. + +--- + +## Pass/Fail Criteria (locked) + +A candidate passes all of the following gates, evaluated in order: + +| Gate | Threshold | Rationale | +|-----------------------|------------|----------------------------------------------------| +| Length | 8–35 aa | Pipeline filter; shorter too short, longer harder to synthesize | +| Canonical AAs only | No non-standard residues | Synthesis feasibility | +| Ensemble score | ≥ 0.50 | Minimum quality bar across all dimensions | +| Safety score | ≥ 0.60 | max_safety_risk = 0.40; hemolysis/toxicity proxies | +| Novelty score | ≥ 0.05 | Minimum 1-mutation difference from any reference seed | +| Activity score | ≥ 0.40 | Must meet minimum predicted activity signal | + +Note on novelty threshold: Phase 3 candidates are generated by 1–3 conservative +substitutions from AMP-like template seeds. A min_novelty of 0.05 ensures every +selected candidate differs by at least one amino acid from the nearest seed, while +allowing the generator to fully explore the near-seed neighborhood. A higher threshold +(e.g. 0.20, used in pipeline.yaml for benchmark purposes) would exclude most or all +near-seed variants and is not appropriate here. + +--- + +## Diversity Selection + +After gate filtering, candidates are ranked by ensemble score (descending). +Greedy diverse selection is applied: each candidate added to the batch must have +pairwise normalized Levenshtein similarity < 0.80 with all previously selected candidates. + +Target batch size: 50–100 candidates. + +--- + +## What Counts as a Passing Batch + +The Phase 3 batch **passes** if: +- ≥ 50 candidates clear all gates above +- ≥ 5 distinct template seeds are represented (diversity of origin) +- No selected candidate has ensemble score < 0.50 +- No selected candidate has safety score < 0.60 + +If fewer than 50 candidates pass, the generation parameters (n_double, n_charge_enhance) +may be increased and the batch re-run. This will be documented transparently. + +--- + +## What This Does NOT Claim + +- It does not claim any candidate will show antimicrobial activity. +- It does not claim any candidate is safe for any use. +- It does not constitute a drug development programme. +- It does not replace expert biological evaluation. + +The lab is the judge. This rule only determines which computational candidates are +nominated for possible future evaluation. diff --git a/examples/benchmark/active_labels.csv b/examples/benchmark/active_labels.csv new file mode 100644 index 00000000..af5f1d6e --- /dev/null +++ b/examples/benchmark/active_labels.csv @@ -0,0 +1,6 @@ +id,sequence,source +BM-POS-001,KWKLFKKIGAVLKVL,benchmark_positive +BM-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,benchmark_positive +BM-POS-003,GLFDIVKKVVGALGSL,benchmark_positive +BM-POS-004,RRWWLRVIAGLLKKVLS,benchmark_positive +BM-POS-005,KLLLKWLKKVLKA,benchmark_positive diff --git a/examples/benchmark/cluster_split_pool.csv b/examples/benchmark/cluster_split_pool.csv new file mode 100644 index 00000000..00a25727 --- /dev/null +++ b/examples/benchmark/cluster_split_pool.csv @@ -0,0 +1,14 @@ +id,sequence,source +CS-POS-001,KWKLFKKIGAVLKFL,cluster_split +CS-POS-002,KWKLFKRIGAVLKVL,cluster_split +CS-POS-003,GLFDIVKKVVGALGAL,cluster_split +CS-NEG-001,AAAAAAAAAAAA,cluster_split +CS-NEG-002,DEDEDEDEDEDE,cluster_split +CS-NEG-003,GGGGGGGGGGGG,cluster_split +CS-NEG-004,EEEEEEEEEEEE,cluster_split +CS-NEG-005,SSSSSSSSSSSS,cluster_split +CS-NEG-006,PPPPPPPPPPPP,cluster_split +CS-NEG-007,TTTTTTTTTTTT,cluster_split +CS-NEG-008,NNNNNNNNNNNN,cluster_split +CS-NEG-009,QQQQQQQQQQQQ,cluster_split +CS-NEG-010,LLLLLLLLLLL,cluster_split diff --git a/examples/benchmark/cluster_split_refs.csv b/examples/benchmark/cluster_split_refs.csv new file mode 100644 index 00000000..80ecc477 --- /dev/null +++ b/examples/benchmark/cluster_split_refs.csv @@ -0,0 +1,4 @@ +id,sequence,source +CSREF-001,KWKLFKKIGAVLKVL,reference +CSREF-002,GIGKFLHSAKKFGKAFVGEIMNS,reference +CSREF-003,GLFDIVKKVVGALGSL,reference diff --git a/examples/benchmark/mixed_candidates.csv b/examples/benchmark/mixed_candidates.csv new file mode 100644 index 00000000..66a811a8 --- /dev/null +++ b/examples/benchmark/mixed_candidates.csv @@ -0,0 +1,21 @@ +id,sequence,source +BM-POS-001,KWKLFKKIGAVLKVL,benchmark_positive +BM-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,benchmark_positive +BM-POS-003,GLFDIVKKVVGALGSL,benchmark_positive +BM-POS-004,RRWWLRVIAGLLKKVLS,benchmark_positive +BM-POS-005,KLLLKWLKKVLKA,benchmark_positive +BM-NEG-001,AAAAAAAAAAAA,benchmark_negative +BM-NEG-002,GGGGGGGGGGGG,benchmark_negative +BM-NEG-003,DEDEDEDEDEDE,benchmark_negative +BM-NEG-004,SSSSSSSSSSS,benchmark_negative +BM-NEG-005,EEEEEEEEEEEE,benchmark_negative +BM-NEG-006,PPPPPPPPPPPP,benchmark_negative +BM-NEG-007,NNNNNNNNNNN,benchmark_negative +BM-NEG-008,QQQQQQQQQQQQ,benchmark_negative +BM-NEG-009,TTTTTTTTTTTT,benchmark_negative +BM-NEG-010,AAGGAAGGAAGG,benchmark_negative +BM-NEG-011,SSTTSSTTSSTT,benchmark_negative +BM-NEG-012,EDEDEDEDEDED,benchmark_negative +BM-NEG-013,GGGAAAGGGAAA,benchmark_negative +BM-NEG-014,QQNNQQNNQQNN,benchmark_negative +BM-NEG-015,SSSSTTTTSSSS,benchmark_negative diff --git a/examples/benchmark/novelty_pressure_pool.csv b/examples/benchmark/novelty_pressure_pool.csv new file mode 100644 index 00000000..70840f20 --- /dev/null +++ b/examples/benchmark/novelty_pressure_pool.csv @@ -0,0 +1,12 @@ +id,sequence,source +NOV-DUP-001,KWKLFKKIGAVLKVL,novelty_pressure +NOV-DUP-002,KWKLFKKIGAVLKFL,novelty_pressure +NOV-DUP-003,KWKLFKRIGAVLKVL,novelty_pressure +NOV-NEW-001,RRLKKVLGAVLKVLK,novelty_pressure +NOV-NEW-002,WKWLKKIRGKLLKV,novelty_pressure +NOV-NEW-003,FLKHFKIKAVLKRLK,novelty_pressure +NOV-NEG-001,AAAAAAAAAAAA,novelty_pressure +NOV-NEG-002,DEDEDEDEDEDE,novelty_pressure +NOV-NEG-003,GGGGGGGGGGGG,novelty_pressure +NOV-NEG-004,EEEEEEEEEEEE,novelty_pressure +NOV-NEG-005,SSSSSSSSSSSS,novelty_pressure diff --git a/examples/benchmark/robustness_positives.csv b/examples/benchmark/robustness_positives.csv new file mode 100644 index 00000000..39ec9072 --- /dev/null +++ b/examples/benchmark/robustness_positives.csv @@ -0,0 +1,4 @@ +id,sequence,source +ROB-POS-001,KWKLFKKIGAVLKVL,robustness +ROB-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,robustness +ROB-POS-003,RRWQWRMKKLG,robustness diff --git a/examples/known_reference/amp_curated_references.csv b/examples/known_reference/amp_curated_references.csv new file mode 100644 index 00000000..68726694 --- /dev/null +++ b/examples/known_reference/amp_curated_references.csv @@ -0,0 +1,45 @@ +id,sequence,source,family,reference,hemolytic_risk_note +REF-MAG-001,GIGKFLHSAKKFGKAFVGEIMNS,published_literature,magainin,Zasloff_1987_PNAS,low +REF-MAG-002,GIGKFLHSAGKFGKAFVGEIMKS,published_literature,magainin,Zasloff_1987_PNAS,low +REF-MAG-003,GIGKFLHSAKKFGKAFVGQIMNS,published_literature,magainin_analog,Chen_1988,low +REF-PEX-001,GIGKFLKKAKKFGKAFVKILKK,published_literature,pexiganan_analog,Ge_1999_Antimicrob,low +REF-BUF-001,TRSSRAGLQFPVGRVHRLLRK,published_literature,buforin,Kim_1996_Eur_J_Biochem,low +REF-BUF-002,RAGLQFPVGRVHRLLRK,published_literature,buforin_ii,Park_2000_J_Biol_Chem,low +REF-IND-001,ILPWKWPWWPWRR,published_literature,indolicidin,Selsted_1992_J_Biol_Chem,moderate +REF-IND-002,ILPWKWPWWPWRRRR,published_literature,indolicidin_analog,Falla_1996_J_Biol_Chem,moderate +REF-TMP-001,FLPLIGRVLSGIL,published_literature,temporin_a,Mangoni_2001_FEBS,low +REF-TMP-002,FVQWFSKFLGRIL,published_literature,temporin_l,Mangoni_2001_FEBS,low +REF-TMP-003,LLPIVGNLLKSLL,published_literature,temporin_b,Simmaco_1996_Eur_J_Biochem,low +REF-TMP-004,FLPLLAGLAANFLPKIF,published_literature,brevinin_like,Conlon_2004_Peptides,moderate +REF-AUR-001,GLFDIIKKIAESF,published_literature,aurein_1,Rozek_2000_Eur_J_Biochem,low +REF-AUR-002,GLFDIVKKVVGALGSL,published_literature,aurein_2,Rozek_2000_Eur_J_Biochem,low +REF-AUR-003,GLFDIIKKIAGSFS,published_literature,aurein_3,Rozek_2000_Eur_J_Biochem,low +REF-ESC-001,GIFSKLAGKKIKNLLISGLKG,published_literature,esculentin_fragment,Simmaco_1993_J_Biol_Chem,low +REF-UPR-001,GVGDLIRKAIASVAGKELG,published_literature,uperin,Jackway_2008_Peptides,low +REF-BOM-001,IIGPVLGLVGSALGGLLKKI,published_literature,bombinin_h,Halverson_1990,moderate +REF-MAC-001,GLFGVLAKVAAHVVPAIAEHF,published_literature,maculatin,Chia_2000_Eur_J_Biochem,low +REF-CAE-001,GLLSVLGSVAKHVLPHVVPVIAEH,published_literature,caerin,Rozek_2000_Biochemistry,low +REF-RRW-001,RRWQWRMKKLG,published_literature,tachyplesin_like,Tam_2002_J_Biol_Chem,moderate +REF-CEC-001,KWKLFKKIEKVGQNIRDGIIKAGPAV,published_literature,cecropin_a_fragment,Steiner_1981_PNAS,low +REF-CEC-002,KWKIFKKIEKVGRNVRDGIIKAGP,published_literature,cecropin_b_fragment,Hultmark_1980_Eur_J_Biochem,low +REF-MEL-001,GIGAVLKVLTTGLPALISWIKRKRQQ,published_literature,melittin,Habermann_1972,high +REF-KWK-001,KWKLFKKIGAVLKVL,published_literature,template_seed_1,Oren_1997_Biochemistry,low +REF-GIG-001,GIGKFLHSAKKFGKAFVGEIMNS,published_literature,template_seed_2,Zasloff_1987_PNAS,low +REF-MAG-004,KKLKKFGLRIINKIVAAL,published_literature,magainin_helix,Wieprecht_1997,low +REF-BMA-001,GRFKRFRKKFKKLFKKLSP,published_literature,bmap_fragment,Skerlavaj_1999,moderate +REF-LLL-001,LLRWPWWPWRRK,published_literature,omiganan_analog,Sader_2004,moderate +REF-TAC-001,KWKKLFKKI,published_literature,cecropin_core,Andrieu_1996,low +REF-GRM-001,GRMKKVFFKSLRQH,published_literature,gramicidin_s_like,Fontenot_1993,moderate +REF-SMA-001,SMKQLAHKLKSEKLELSDREQMR,published_literature,smap_fragment,Tossi_2000,low +REF-HLP-001,HLPKHFKAFVARLIRNAMSN,published_literature,hlp_fragment,Hiemstra_1993,low +REF-FKK-001,FKKLKKSLKL,published_literature,klaklak_analog,Javadpour_1996,low +REF-KAA-001,KAAAKAAAKAAAK,published_literature,ala_lys_repeat,Braunstein_2004,low +REF-GKK-001,GKKLFSKLKK,published_literature,cecropin_melittin_analog,Gazit_1994,low +REF-RLK-001,RLKKVWKKWKK,published_literature,cationic_tryptophan,Parisien_2008,moderate +REF-ALA-001,ALWKTMLKKLGTMALHAGKAALGAAADTISQGTQ,published_literature,dermaseptin_s3_fragment,Mor_1994_Biochemistry,low +REF-ILK-001,ILKKWPWWPWRRK,published_literature,indolicidin_lys_analog,Lawyer_1996,moderate +REF-VRG-001,VRGRPIPICSRGGYCFCLPFLPGACHRPSGMC,published_literature,defensin_like_linear,skipped_disulfide,not_applicable +REF-PLA-001,KKVVFKVKFK,published_literature,short_cationic,Lee_2004,low +REF-GKA-001,GKAFVGEIMNS,published_literature,magainin_c_fragment,Westerhoff_1989,low +REF-WKL-001,WKLFKKIGAVLKVL,published_literature,magainin_analog,Dathe_1997,low +REF-KFL-001,KFLHSAKKFGKAFV,published_literature,magainin_core,Maloy_1995,low diff --git a/examples/negative/poly_cationic.csv b/examples/negative/poly_cationic.csv new file mode 100644 index 00000000..5d58cab3 --- /dev/null +++ b/examples/negative/poly_cationic.csv @@ -0,0 +1,6 @@ +id,sequence,source +POLYK-001,KKKKKKKKKKKKK,poly_cationic +POLYR-001,RRRRRRRRRRRRR,poly_cationic +POLYKR-001,KRKRKRKRKRKRK,poly_cationic +POLYKR-002,KKRRKKKRRKKR,poly_cationic +POLYKR-003,RRKKRRKKRRKK,poly_cationic diff --git a/examples/negative/poly_hydrophobic.csv b/examples/negative/poly_hydrophobic.csv new file mode 100644 index 00000000..0dd84b58 --- /dev/null +++ b/examples/negative/poly_hydrophobic.csv @@ -0,0 +1,6 @@ +id,sequence,source +POLYH-001,LLLLLLLLLLLLL,poly_hydrophobic +POLYH-002,IIIIIIIIIIIII,poly_hydrophobic +POLYH-003,VVVVVVVVVVVVV,poly_hydrophobic +POLYH-004,LLIIVVLLIIVVL,poly_hydrophobic +POLYH-005,LLLLVVIIILLLL,poly_hydrophobic diff --git a/examples/sequences/amp_seeds.csv b/examples/sequences/amp_seeds.csv new file mode 100644 index 00000000..54751e42 --- /dev/null +++ b/examples/sequences/amp_seeds.csv @@ -0,0 +1,6 @@ +id,sequence,source +SEED-001,KWKLFKKIGAVLKVL,magainin_like_template +SEED-002,GIGKFLHSAKKFGKAFVGEIMNS,cecropin_like_template +SEED-003,RRWQWRMKKLG,cationic_tryptophan_template +SEED-004,FLPLIGRVLSGIL,temporin_like_template +SEED-005,KRLFKKIGSALKFL,hybrid_cationic_hydrophobic diff --git a/examples/sequences/phase3_pool.csv b/examples/sequences/phase3_pool.csv new file mode 100644 index 00000000..b7691a6c --- /dev/null +++ b/examples/sequences/phase3_pool.csv @@ -0,0 +1,384 @@ +id,sequence,source +SEED-001_VAR_001,HWKFFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_002,HWKLFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_003,HWKLFKKIGAVLKVM,template_mutation_from_SEED-001 +SEED-001_VAR_004,KFKLFHKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_005,KFKLFKKIGAALKVL,template_mutation_from_SEED-001 +SEED-001_VAR_006,KFKLFKKIGAVIKVL,template_mutation_from_SEED-001 +SEED-001_VAR_007,KFKLFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_008,KFKLFKKIGAVMKVL,template_mutation_from_SEED-001 +SEED-001_VAR_009,KWHLFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_010,KWKAFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_011,KWKFFKKIGAVIKVL,template_mutation_from_SEED-001 +SEED-001_VAR_012,KWKFFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_013,KWKFFKKIGMVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_014,KWKIFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_015,KWKIFKKIPAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_016,KWKLAKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_017,KWKLFHKFGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_018,KWKLFHKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_019,KWKLFHKIGAVLKVV,template_mutation_from_SEED-001 +SEED-001_VAR_020,KWKLFHKLGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_021,KWKLFHRIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_022,KWKLFKHIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_023,KWKLFKHIPAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_024,KWKLFKKAGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_025,KWKLFKKFGAVLKFL,template_mutation_from_SEED-001 +SEED-001_VAR_026,KWKLFKKFGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_027,KWKLFKKIGAALKVL,template_mutation_from_SEED-001 +SEED-001_VAR_028,KWKLFKKIGAFLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_029,KWKLFKKIGAILKVL,template_mutation_from_SEED-001 +SEED-001_VAR_030,KWKLFKKIGALLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_031,KWKLFKKIGAMLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_032,KWKLFKKIGAVAKVL,template_mutation_from_SEED-001 +SEED-001_VAR_033,KWKLFKKIGAVFKVL,template_mutation_from_SEED-001 +SEED-001_VAR_034,KWKLFKKIGAVIKVL,template_mutation_from_SEED-001 +SEED-001_VAR_035,KWKLFKKIGAVLHVL,template_mutation_from_SEED-001 +SEED-001_VAR_036,KWKLFKKIGAVLKAL,template_mutation_from_SEED-001 +SEED-001_VAR_037,KWKLFKKIGAVLKFL,template_mutation_from_SEED-001 +SEED-001_VAR_038,KWKLFKKIGAVLKIL,template_mutation_from_SEED-001 +SEED-001_VAR_039,KWKLFKKIGAVLKLL,template_mutation_from_SEED-001 +SEED-001_VAR_040,KWKLFKKIGAVLKML,template_mutation_from_SEED-001 +SEED-001_VAR_041,KWKLFKKIGAVLKVA,template_mutation_from_SEED-001 +SEED-001_VAR_042,KWKLFKKIGAVLKVF,template_mutation_from_SEED-001 +SEED-001_VAR_043,KWKLFKKIGAVLKVI,template_mutation_from_SEED-001 +SEED-001_VAR_044,KWKLFKKIGAVLKVM,template_mutation_from_SEED-001 +SEED-001_VAR_045,KWKLFKKIGAVLKVV,template_mutation_from_SEED-001 +SEED-001_VAR_046,KWKLFKKIGAVLRVL,template_mutation_from_SEED-001 +SEED-001_VAR_047,KWKLFKKIGAVMKVL,template_mutation_from_SEED-001 +SEED-001_VAR_048,KWKLFKKIGAVVKVF,template_mutation_from_SEED-001 +SEED-001_VAR_049,KWKLFKKIGAVVKVL,template_mutation_from_SEED-001 +SEED-001_VAR_050,KWKLFKKIGFVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_051,KWKLFKKIGIVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_052,KWKLFKKIGLVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_053,KWKLFKKIGMVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_054,KWKLFKKIGVVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_055,KWKLFKKIGVVLRVL,template_mutation_from_SEED-001 +SEED-001_VAR_056,KWKLFKKIPAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_057,KWKLFKKIPAVVKVL,template_mutation_from_SEED-001 +SEED-001_VAR_058,KWKLFKKLGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_059,KWKLFKKMGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_060,KWKLFKKVGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_061,KWKLFKRIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_062,KWKLFKRIGLVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_063,KWKLFRKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_064,KWKLFRKIGAVLRVL,template_mutation_from_SEED-001 +SEED-001_VAR_065,KWKLFRRIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_066,KWKLIKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_067,KWKLLKKIGAVLKML,template_mutation_from_SEED-001 +SEED-001_VAR_068,KWKLLKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_069,KWKLMKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_070,KWKLVKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_071,KWKMFKKIGAVLKAL,template_mutation_from_SEED-001 +SEED-001_VAR_072,KWKMFKKIGAVLKVI,template_mutation_from_SEED-001 +SEED-001_VAR_073,KWKMFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_074,KWKVFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_075,KWRLFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_076,KWRLFKKVGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_077,KYKLFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-001_VAR_078,RWKLFKKIGAVLKVL,template_mutation_from_SEED-001 +SEED-002_VAR_001,GAGKFLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_002,GFGKFLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_003,GIGHFLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_004,GIGKALHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_005,GIGKFAHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_006,GIGKFFHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_007,GIGKFIHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_008,GIGKFIHSAKKFGKAVVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_009,GIGKFLHNAKKFGKAFVGEIMNQ,template_mutation_from_SEED-002 +SEED-002_VAR_010,GIGKFLHNAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_011,GIGKFLHQAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_012,GIGKFLHRAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_013,GIGKFLHSAHKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_014,GIGKFLHSAKHFGKAFVGDIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_015,GIGKFLHSAKHFGKAFVGEIFNS,template_mutation_from_SEED-002 +SEED-002_VAR_016,GIGKFLHSAKHFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_017,GIGKFLHSAKHLGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_018,GIGKFLHSAKKAGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_019,GIGKFLHSAKKFGHAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_020,GIGKFLHSAKKFGHAFVGEIMNT,template_mutation_from_SEED-002 +SEED-002_VAR_021,GIGKFLHSAKKFGKAAVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_022,GIGKFLHSAKKFGKAFAGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_023,GIGKFLHSAKKFGKAFFGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_024,GIGKFLHSAKKFGKAFFGELMNS,template_mutation_from_SEED-002 +SEED-002_VAR_025,GIGKFLHSAKKFGKAFIGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_026,GIGKFLHSAKKFGKAFLGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_027,GIGKFLHSAKKFGKAFMGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_028,GIGKFLHSAKKFGKAFVGDIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_029,GIGKFLHSAKKFGKAFVGEAMNS,template_mutation_from_SEED-002 +SEED-002_VAR_030,GIGKFLHSAKKFGKAFVGEFMNS,template_mutation_from_SEED-002 +SEED-002_VAR_031,GIGKFLHSAKKFGKAFVGEIANS,template_mutation_from_SEED-002 +SEED-002_VAR_032,GIGKFLHSAKKFGKAFVGEIFNS,template_mutation_from_SEED-002 +SEED-002_VAR_033,GIGKFLHSAKKFGKAFVGEIINS,template_mutation_from_SEED-002 +SEED-002_VAR_034,GIGKFLHSAKKFGKAFVGEILNS,template_mutation_from_SEED-002 +SEED-002_VAR_035,GIGKFLHSAKKFGKAFVGEIMKS,template_mutation_from_SEED-002 +SEED-002_VAR_036,GIGKFLHSAKKFGKAFVGEIMNN,template_mutation_from_SEED-002 +SEED-002_VAR_037,GIGKFLHSAKKFGKAFVGEIMNQ,template_mutation_from_SEED-002 +SEED-002_VAR_038,GIGKFLHSAKKFGKAFVGEIMNR,template_mutation_from_SEED-002 +SEED-002_VAR_039,GIGKFLHSAKKFGKAFVGEIMNT,template_mutation_from_SEED-002 +SEED-002_VAR_040,GIGKFLHSAKKFGKAFVGEIMQQ,template_mutation_from_SEED-002 +SEED-002_VAR_041,GIGKFLHSAKKFGKAFVGEIMQS,template_mutation_from_SEED-002 +SEED-002_VAR_042,GIGKFLHSAKKFGKAFVGEIMSS,template_mutation_from_SEED-002 +SEED-002_VAR_043,GIGKFLHSAKKFGKAFVGEIMTS,template_mutation_from_SEED-002 +SEED-002_VAR_044,GIGKFLHSAKKFGKAFVGEIVNS,template_mutation_from_SEED-002 +SEED-002_VAR_045,GIGKFLHSAKKFGKAFVGELMNS,template_mutation_from_SEED-002 +SEED-002_VAR_046,GIGKFLHSAKKFGKAFVGEMMNS,template_mutation_from_SEED-002 +SEED-002_VAR_047,GIGKFLHSAKKFGKAFVGEVMNS,template_mutation_from_SEED-002 +SEED-002_VAR_048,GIGKFLHSAKKFGKAFVPEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_049,GIGKFLHSAKKFGKAFVPEIMNT,template_mutation_from_SEED-002 +SEED-002_VAR_050,GIGKFLHSAKKFGKAIVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_051,GIGKFLHSAKKFGKALVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_052,GIGKFLHSAKKFGKAMLGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_053,GIGKFLHSAKKFGKAMVGEIINS,template_mutation_from_SEED-002 +SEED-002_VAR_054,GIGKFLHSAKKFGKAMVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_055,GIGKFLHSAKKFGKAVVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_056,GIGKFLHSAKKFGKFFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_057,GIGKFLHSAKKFGKIFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_058,GIGKFLHSAKKFGKLFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_059,GIGKFLHSAKKFGKLFVGEIMNT,template_mutation_from_SEED-002 +SEED-002_VAR_060,GIGKFLHSAKKFGKMFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_061,GIGKFLHSAKKFGKVFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_062,GIGKFLHSAKKFGRAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_063,GIGKFLHSAKKFPKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_064,GIGKFLHSAKKIGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_065,GIGKFLHSAKKIGKIFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_066,GIGKFLHSAKKLGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_067,GIGKFLHSAKKMGKAFVGEILNS,template_mutation_from_SEED-002 +SEED-002_VAR_068,GIGKFLHSAKKMGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_069,GIGKFLHSAKKVGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_070,GIGKFLHSAKKVGKAIVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_071,GIGKFLHSAKRFGKAFVGEIMNN,template_mutation_from_SEED-002 +SEED-002_VAR_072,GIGKFLHSAKRFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_073,GIGKFLHSARKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_074,GIGKFLHSFKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_075,GIGKFLHSIKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_076,GIGKFLHSLKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_077,GIGKFLHSMKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_078,GIGKFLHSVKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_079,GIGKFLHSVKKFPKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_080,GIGKFLHTAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_081,GIGKFLKNAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_082,GIGKFLKSAKKFGKAFVGDIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_083,GIGKFLKSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_084,GIGKFLKSAKKFGKAFVPEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_085,GIGKFLRSAKKFGHAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_086,GIGKFLRSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_087,GIGKFMHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_088,GIGKFVHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_089,GIGKILHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_090,GIGKLLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_091,GIGKMLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_092,GIGKVLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_093,GIGRFLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_094,GIGRMLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_095,GIPKFLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_096,GIPKFLKSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_097,GLGKFLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_098,GMGKFLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_099,GVGKFLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_100,PIGKFFHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_101,PIGKFLHSAKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-002_VAR_102,PIGKFLHSVKKFGKAFVGEIMNS,template_mutation_from_SEED-002 +SEED-003_VAR_001,HRWQWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_002,HRWTWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_003,KRWQWRIKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_004,KRWQWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_005,KRWQYRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_006,RHFQWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_007,RHWQWRMKKIG,template_mutation_from_SEED-003 +SEED-003_VAR_008,RHWQWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_009,RKWNWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_010,RKWQWRMHKLG,template_mutation_from_SEED-003 +SEED-003_VAR_011,RKWQWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_012,RKWQYRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_013,RRFQWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_014,RRFQWRMRKLG,template_mutation_from_SEED-003 +SEED-003_VAR_015,RRWNWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_016,RRWNWRMKKMG,template_mutation_from_SEED-003 +SEED-003_VAR_017,RRWNWRMRKLG,template_mutation_from_SEED-003 +SEED-003_VAR_018,RRWQFRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_019,RRWQFRVKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_020,RRWQWHIKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_021,RRWQWHMKHLG,template_mutation_from_SEED-003 +SEED-003_VAR_022,RRWQWHMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_023,RRWQWKIKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_024,RRWQWKMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_025,RRWQWRAKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_026,RRWQWRFKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_027,RRWQWRFKRLG,template_mutation_from_SEED-003 +SEED-003_VAR_028,RRWQWRIKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_029,RRWQWRLKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_030,RRWQWRMHKLG,template_mutation_from_SEED-003 +SEED-003_VAR_031,RRWQWRMHKLP,template_mutation_from_SEED-003 +SEED-003_VAR_032,RRWQWRMKHIG,template_mutation_from_SEED-003 +SEED-003_VAR_033,RRWQWRMKHLG,template_mutation_from_SEED-003 +SEED-003_VAR_034,RRWQWRMKKAG,template_mutation_from_SEED-003 +SEED-003_VAR_035,RRWQWRMKKFG,template_mutation_from_SEED-003 +SEED-003_VAR_036,RRWQWRMKKIG,template_mutation_from_SEED-003 +SEED-003_VAR_037,RRWQWRMKKLP,template_mutation_from_SEED-003 +SEED-003_VAR_038,RRWQWRMKKMG,template_mutation_from_SEED-003 +SEED-003_VAR_039,RRWQWRMKKVG,template_mutation_from_SEED-003 +SEED-003_VAR_040,RRWQWRMKRLG,template_mutation_from_SEED-003 +SEED-003_VAR_041,RRWQWRMRKLG,template_mutation_from_SEED-003 +SEED-003_VAR_042,RRWQWRMRKLP,template_mutation_from_SEED-003 +SEED-003_VAR_043,RRWQWRMRKMG,template_mutation_from_SEED-003 +SEED-003_VAR_044,RRWQWRVKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_045,RRWQYRIKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_046,RRWQYRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_047,RRWRWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_048,RRWSWHMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_049,RRWSWRMHKLG,template_mutation_from_SEED-003 +SEED-003_VAR_050,RRWSWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_051,RRWTWRMKKAG,template_mutation_from_SEED-003 +SEED-003_VAR_052,RRWTWRMKKLG,template_mutation_from_SEED-003 +SEED-003_VAR_053,RRYQWRMHKLG,template_mutation_from_SEED-003 +SEED-003_VAR_054,RRYQWRMKKLG,template_mutation_from_SEED-003 +SEED-004_VAR_001,ALPFIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_002,ALPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_003,FAPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_004,FFPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_005,FFPLIGRVLSGML,template_mutation_from_SEED-004 +SEED-004_VAR_006,FIPFIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_007,FIPLIGRFLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_008,FIPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_009,FLGLIGRMLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_010,FLGLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_011,FLPAIGRVLNGIL,template_mutation_from_SEED-004 +SEED-004_VAR_012,FLPAIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_013,FLPFIGRVLQGIL,template_mutation_from_SEED-004 +SEED-004_VAR_014,FLPFIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_015,FLPFIGRVLSGLL,template_mutation_from_SEED-004 +SEED-004_VAR_016,FLPIIGRFLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_017,FLPIIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_018,FLPIIGRVMSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_019,FLPLAGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_020,FLPLFGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_021,FLPLIGHVISGIL,template_mutation_from_SEED-004 +SEED-004_VAR_022,FLPLIGHVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_023,FLPLIGKVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_024,FLPLIGRALSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_025,FLPLIGRFLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_026,FLPLIGRILSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_027,FLPLIGRILSGVL,template_mutation_from_SEED-004 +SEED-004_VAR_028,FLPLIGRLLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_029,FLPLIGRMLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_030,FLPLIGRVASGIL,template_mutation_from_SEED-004 +SEED-004_VAR_031,FLPLIGRVFSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_032,FLPLIGRVISGIL,template_mutation_from_SEED-004 +SEED-004_VAR_033,FLPLIGRVISGVL,template_mutation_from_SEED-004 +SEED-004_VAR_034,FLPLIGRVLNGIL,template_mutation_from_SEED-004 +SEED-004_VAR_035,FLPLIGRVLQGIL,template_mutation_from_SEED-004 +SEED-004_VAR_036,FLPLIGRVLRGIL,template_mutation_from_SEED-004 +SEED-004_VAR_037,FLPLIGRVLSGAL,template_mutation_from_SEED-004 +SEED-004_VAR_038,FLPLIGRVLSGFL,template_mutation_from_SEED-004 +SEED-004_VAR_039,FLPLIGRVLSGIA,template_mutation_from_SEED-004 +SEED-004_VAR_040,FLPLIGRVLSGIF,template_mutation_from_SEED-004 +SEED-004_VAR_041,FLPLIGRVLSGII,template_mutation_from_SEED-004 +SEED-004_VAR_042,FLPLIGRVLSGIM,template_mutation_from_SEED-004 +SEED-004_VAR_043,FLPLIGRVLSGIV,template_mutation_from_SEED-004 +SEED-004_VAR_044,FLPLIGRVLSGLL,template_mutation_from_SEED-004 +SEED-004_VAR_045,FLPLIGRVLSGML,template_mutation_from_SEED-004 +SEED-004_VAR_046,FLPLIGRVLSGVL,template_mutation_from_SEED-004 +SEED-004_VAR_047,FLPLIGRVLSGVM,template_mutation_from_SEED-004 +SEED-004_VAR_048,FLPLIGRVLSPFL,template_mutation_from_SEED-004 +SEED-004_VAR_049,FLPLIGRVLSPIL,template_mutation_from_SEED-004 +SEED-004_VAR_050,FLPLIGRVLSPLL,template_mutation_from_SEED-004 +SEED-004_VAR_051,FLPLIGRVLTGIL,template_mutation_from_SEED-004 +SEED-004_VAR_052,FLPLIGRVMSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_053,FLPLIGRVVSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_054,FLPLIPKVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_055,FLPLIPRVLSGFL,template_mutation_from_SEED-004 +SEED-004_VAR_056,FLPLIPRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_057,FLPLLGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_058,FLPLLGRVVSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_059,FLPLMGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_060,FLPLVGKVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_061,FLPLVGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_062,FLPMIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_063,FLPMIGRVLSGLL,template_mutation_from_SEED-004 +SEED-004_VAR_064,FLPMIPRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_065,FLPVIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_066,FMPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_067,FVPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_068,FVPLIPRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_069,ILPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_070,LLPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_071,MLGLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_072,MLPLIGKVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_073,MLPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-004_VAR_074,VLPLIGRVLSGIL,template_mutation_from_SEED-004 +SEED-005_VAR_001,HRLFKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_002,HRLIKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_003,KHLFKHIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_004,KHLFKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_005,KHLFKKIGSALRFL,template_mutation_from_SEED-005 +SEED-005_VAR_006,KKLFKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_007,KKLIKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_008,KRAFKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_009,KRFFKKIGSALKFA,template_mutation_from_SEED-005 +SEED-005_VAR_010,KRFFKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_011,KRFFKKIPSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_012,KRIFKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_013,KRLAKHIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_014,KRLAKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_015,KRLFHKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_016,KRLFKHIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_017,KRLFKHIPSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_018,KRLFKHLGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_019,KRLFKKAGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_020,KRLFKKFGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_021,KRLFKKIGNALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_022,KRLFKKIGQALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_023,KRLFKKIGRALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_024,KRLFKKIGSAAKFL,template_mutation_from_SEED-005 +SEED-005_VAR_025,KRLFKKIGSAFKFL,template_mutation_from_SEED-005 +SEED-005_VAR_026,KRLFKKIGSAIKFL,template_mutation_from_SEED-005 +SEED-005_VAR_027,KRLFKKIGSALHFL,template_mutation_from_SEED-005 +SEED-005_VAR_028,KRLFKKIGSALHFV,template_mutation_from_SEED-005 +SEED-005_VAR_029,KRLFKKIGSALKAL,template_mutation_from_SEED-005 +SEED-005_VAR_030,KRLFKKIGSALKFA,template_mutation_from_SEED-005 +SEED-005_VAR_031,KRLFKKIGSALKFF,template_mutation_from_SEED-005 +SEED-005_VAR_032,KRLFKKIGSALKFI,template_mutation_from_SEED-005 +SEED-005_VAR_033,KRLFKKIGSALKFM,template_mutation_from_SEED-005 +SEED-005_VAR_034,KRLFKKIGSALKFV,template_mutation_from_SEED-005 +SEED-005_VAR_035,KRLFKKIGSALKIL,template_mutation_from_SEED-005 +SEED-005_VAR_036,KRLFKKIGSALKLL,template_mutation_from_SEED-005 +SEED-005_VAR_037,KRLFKKIGSALKML,template_mutation_from_SEED-005 +SEED-005_VAR_038,KRLFKKIGSALKVL,template_mutation_from_SEED-005 +SEED-005_VAR_039,KRLFKKIGSALRFL,template_mutation_from_SEED-005 +SEED-005_VAR_040,KRLFKKIGSAMKFL,template_mutation_from_SEED-005 +SEED-005_VAR_041,KRLFKKIGSAVKFL,template_mutation_from_SEED-005 +SEED-005_VAR_042,KRLFKKIGSFLKFL,template_mutation_from_SEED-005 +SEED-005_VAR_043,KRLFKKIGSILKFL,template_mutation_from_SEED-005 +SEED-005_VAR_044,KRLFKKIGSLLKFL,template_mutation_from_SEED-005 +SEED-005_VAR_045,KRLFKKIGSMLKFL,template_mutation_from_SEED-005 +SEED-005_VAR_046,KRLFKKIGSVLKFL,template_mutation_from_SEED-005 +SEED-005_VAR_047,KRLFKKIGSVLKML,template_mutation_from_SEED-005 +SEED-005_VAR_048,KRLFKKIGTALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_049,KRLFKKIPSALKFF,template_mutation_from_SEED-005 +SEED-005_VAR_050,KRLFKKIPSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_051,KRLFKKIPSALRFL,template_mutation_from_SEED-005 +SEED-005_VAR_052,KRLFKKIPTALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_053,KRLFKKLGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_054,KRLFKKLGSLLKFL,template_mutation_from_SEED-005 +SEED-005_VAR_055,KRLFKKMGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_056,KRLFKKVGNALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_057,KRLFKKVGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_058,KRLFKKVGSALRFL,template_mutation_from_SEED-005 +SEED-005_VAR_059,KRLFKRIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_060,KRLFKRIGSALKML,template_mutation_from_SEED-005 +SEED-005_VAR_061,KRLFKRLGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_062,KRLFRKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_063,KRLFRKIGSALKFV,template_mutation_from_SEED-005 +SEED-005_VAR_064,KRLFRKIGTALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_065,KRLIKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_066,KRLIKKIGSMLKFL,template_mutation_from_SEED-005 +SEED-005_VAR_067,KRLLKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_068,KRLMKKIGSAIKFL,template_mutation_from_SEED-005 +SEED-005_VAR_069,KRLMKKIGSALKFA,template_mutation_from_SEED-005 +SEED-005_VAR_070,KRLMKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_071,KRLMKKIGSALRFL,template_mutation_from_SEED-005 +SEED-005_VAR_072,KRLVKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_073,KRMFKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_074,KRVFKKIGSALKFL,template_mutation_from_SEED-005 +SEED-005_VAR_075,RRLFKKIGSALKFL,template_mutation_from_SEED-005 diff --git a/schemas/batch_report.schema.json b/schemas/batch_report.schema.json index ab9d54cd..1ca8eb2d 100644 --- a/schemas/batch_report.schema.json +++ b/schemas/batch_report.schema.json @@ -2,11 +2,22 @@ "$schema": "https://json-schema.org/draft/2020-12/schema", "title": "OpenAMP Batch Report", "type": "object", - "required": ["pipeline_version", "candidate_count", "selected_count", "generated_at"], + "required": ["pipeline_version", "candidate_count", "selected_count", "generated_at", "disclaimer"], "properties": { "pipeline_version": {"type": "string"}, "candidate_count": {"type": "integer", "minimum": 0}, "selected_count": {"type": "integer", "minimum": 0}, - "generated_at": {"type": "string"} + "generated_at": {"type": "string"}, + "disclaimer": {"type": "string"}, + "score_averages": { + "type": "object", + "properties": { + "activity": {"type": "number", "minimum": 0, "maximum": 1}, + "safety": {"type": "number", "minimum": 0, "maximum": 1}, + "novelty": {"type": "number", "minimum": 0, "maximum": 1}, + "ensemble": {"type": "number", "minimum": 0, "maximum": 1} + } + }, + "selected_ids": {"type": "array", "items": {"type": "string"}} } } diff --git a/schemas/candidate.schema.json b/schemas/candidate.schema.json index 9a3b280b..ec2c2d7b 100644 --- a/schemas/candidate.schema.json +++ b/schemas/candidate.schema.json @@ -25,7 +25,15 @@ "safety": {"type": "number", "minimum": 0, "maximum": 1}, "synthesis": {"type": "number", "minimum": 0, "maximum": 1}, "novelty": {"type": "number", "minimum": 0, "maximum": 1}, - "ensemble": {"type": "number", "minimum": 0, "maximum": 1} + "ensemble": {"type": "number", "minimum": 0, "maximum": 1}, + "boman_activity": { + "type": "number", "minimum": 0, "maximum": 1, + "description": "Boman index (2003) normalized to [0,1]. Independent second activity scorer." + }, + "disagreement": { + "type": "number", "minimum": 0, "maximum": 1, + "description": "Absolute difference between activity_likeness and boman_activity. High values indicate scoring uncertainty." + } } }, "references_checked": {"type": "array", "items": {"type": "string"}}, diff --git a/schemas/lab_result.schema.json b/schemas/lab_result.schema.json new file mode 100644 index 00000000..f43af2a4 --- /dev/null +++ b/schemas/lab_result.schema.json @@ -0,0 +1,109 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "title": "OpenAMP Lab Result", + "description": "Machine-readable lab assay result for a computational candidate. Required for active-learning loop ingestion.", + "type": "object", + "required": [ + "result_id", + "candidate_id", + "assay_type", + "organism_or_cell_line", + "result_value", + "result_unit", + "positive_control_passed", + "negative_control_passed", + "assay_date", + "replicate_count", + "performed_by_lab", + "computational_candidate_certificate_hash", + "disclaimer" + ], + "properties": { + "result_id": { + "type": "string", + "description": "Unique identifier for this assay result (e.g. UUID)" + }, + "candidate_id": { + "type": "string", + "description": "ID from the evidence certificate of the tested candidate" + }, + "assay_type": { + "type": "string", + "enum": [ + "MIC", + "MBC", + "hemolysis_RBC", + "cytotoxicity_mammalian", + "membrane_disruption", + "time_kill", + "biofilm_inhibition", + "other" + ], + "description": "Type of assay performed" + }, + "organism_or_cell_line": { + "type": "string", + "description": "Organism tested (e.g. 'E. coli ATCC 25922') or cell line (e.g. 'hRBC') with strain/lot details" + }, + "result_value": { + "type": ["number", "null"], + "description": "Numeric result (e.g. MIC in µg/mL, or % hemolysis). Null if qualitative only." + }, + "result_unit": { + "type": "string", + "description": "Unit of result_value (e.g. 'µg/mL', '%', 'log10_CFU/mL')" + }, + "result_qualitative": { + "type": ["string", "null"], + "enum": ["active", "inactive", "partial", "toxic", "inconclusive", null], + "description": "Qualitative interpretation, if applicable" + }, + "positive_control_passed": { + "type": "boolean", + "description": "Whether the positive control gave the expected result" + }, + "negative_control_passed": { + "type": "boolean", + "description": "Whether the negative control gave the expected result" + }, + "positive_control_id": { + "type": ["string", "null"], + "description": "Identity of the positive control used (e.g. 'ciprofloxacin 0.25 µg/mL')" + }, + "negative_control_id": { + "type": ["string", "null"], + "description": "Identity of the negative control used (e.g. 'PBS', 'vehicle')" + }, + "assay_date": { + "type": "string", + "format": "date", + "description": "Date assay was performed (ISO 8601 format)" + }, + "replicate_count": { + "type": "integer", + "minimum": 1, + "description": "Number of independent replicates" + }, + "performed_by_lab": { + "type": "string", + "description": "Name of lab or CRO that performed the assay (not individual researchers)" + }, + "raw_data_sha256": { + "type": ["string", "null"], + "description": "SHA-256 hash of the raw assay data file, if available" + }, + "computational_candidate_certificate_hash": { + "type": "string", + "description": "SHA-256 hash of the evidence certificate JSON for this candidate at time of selection" + }, + "notes": { + "type": ["string", "null"], + "description": "Optional assay notes, deviations from protocol, or caveats" + }, + "disclaimer": { + "type": "string", + "description": "Must state that this is an experimental result on a computationally nominated candidate and does not constitute a drug or clinical claim" + } + }, + "additionalProperties": false +} diff --git a/schemas/run_manifest.schema.json b/schemas/run_manifest.schema.json index ab3b16f9..000add59 100644 --- a/schemas/run_manifest.schema.json +++ b/schemas/run_manifest.schema.json @@ -2,12 +2,17 @@ "$schema": "https://json-schema.org/draft/2020-12/schema", "title": "OpenAMP Run Manifest", "type": "object", - "required": ["run_id", "pipeline_version", "config_hash", "inputs", "outputs"], + "required": ["run_id", "pipeline_version", "config_hash", "generated_at", "inputs", "input_hashes", "outputs"], "properties": { "run_id": {"type": "string"}, "pipeline_version": {"type": "string"}, "config_hash": {"type": "string"}, + "generated_at": {"type": "string"}, "inputs": {"type": "array", "items": {"type": "string"}}, + "input_hashes": { + "type": "object", + "additionalProperties": {"type": "string"} + }, "outputs": {"type": "array", "items": {"type": "string"}} } } diff --git a/scripts/generate_phase3_batch.py b/scripts/generate_phase3_batch.py new file mode 100644 index 00000000..b8b96282 --- /dev/null +++ b/scripts/generate_phase3_batch.py @@ -0,0 +1,97 @@ +"""Phase 3 candidate generation script. + +Reads seed sequences from examples/sequences/amp_seeds.csv, applies conservative +substitution mutations, and writes the candidate pool to +examples/sequences/phase3_pool.csv. + +Usage: + python scripts/generate_phase3_batch.py [--seeds PATH] [--out PATH] [--seed INT] + +This is a toy mutation explorer. Generated candidates must pass through the full +pipeline before any selection. The lab is the judge of activity. +""" +from __future__ import annotations + +import argparse +import csv +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parent.parent / "src")) + +from openamp_foundry.generators.template_mutator import generate_candidate_pool + + +def load_seeds(path: Path) -> tuple[list[str], list[str]]: + seed_ids: list[str] = [] + seed_seqs: list[str] = [] + with open(path, newline="") as f: + reader = csv.DictReader(f) + for row in reader: + seed_ids.append(row["id"]) + seed_seqs.append(row["sequence"].strip().upper()) + return seed_ids, seed_seqs + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Generate Phase 3 candidate pool") + parser.add_argument( + "--seeds", + default="examples/sequences/amp_seeds.csv", + help="Seed sequences CSV (columns: id, sequence, source)", + ) + parser.add_argument( + "--out", + default="examples/sequences/phase3_pool.csv", + help="Output pool CSV path", + ) + parser.add_argument( + "--n-double", + type=int, + default=25, + help="Number of double substitution variants per seed (default: 25)", + ) + parser.add_argument( + "--n-charge", + type=int, + default=12, + help="Number of charge-enhanced variants per seed (default: 12)", + ) + parser.add_argument( + "--rng-seed", + type=int, + default=2024, + help="RNG seed for reproducibility (default: 2024)", + ) + args = parser.parse_args(argv) + + seeds_path = Path(args.seeds) + if not seeds_path.exists(): + print(f"ERROR: seeds file not found: {seeds_path}", file=sys.stderr) + return 1 + + seed_ids, seed_seqs = load_seeds(seeds_path) + print(f"Loaded {len(seed_ids)} seed sequences from {seeds_path}") + + pool = generate_candidate_pool( + seed_sequences=seed_seqs, + seed_ids=seed_ids, + n_double=args.n_double, + n_charge_enhance=args.n_charge, + rng_seed=args.rng_seed, + ) + + out_path = Path(args.out) + out_path.parent.mkdir(parents=True, exist_ok=True) + + with open(out_path, "w", newline="", encoding="utf-8") as f: + writer = csv.DictWriter(f, fieldnames=["id", "sequence", "source"]) + writer.writeheader() + writer.writerows(pool) + + print(f"Generated {len(pool)} candidate variants → {out_path}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/openamp_foundry/benchmark/evaluate.py b/src/openamp_foundry/benchmark/evaluate.py index 34c5ae0b..0f150352 100644 --- a/src/openamp_foundry/benchmark/evaluate.py +++ b/src/openamp_foundry/benchmark/evaluate.py @@ -1,8 +1,130 @@ from __future__ import annotations +from openamp_foundry.scoring.novelty import normalized_similarity from openamp_foundry.types import ScoredCandidate def top_k_ids(scored: list[ScoredCandidate], k: int) -> set[str]: ranked = sorted(scored, key=lambda x: x.scores.get("ensemble", 0.0), reverse=True) return {item.candidate.candidate_id for item in ranked[:k]} + + +def recall_at_k( + scored: list[ScoredCandidate], + positive_ids: set[str], + k: int, +) -> float: + """Fraction of known positives recovered in the top-k ranked candidates. + + Recall@k = |positives in top-k| / |total positives| + + A random baseline would recover roughly k / |all candidates| of the positives. + The pipeline is meaningful if recall@k significantly exceeds the random baseline. + """ + if not positive_ids: + return 0.0 + top = top_k_ids(scored, k) + recovered = len(top & positive_ids) + return round(recovered / len(positive_ids), 4) + + +def random_recall_at_k(n_candidates: int, n_positives: int, k: int) -> float: + """Expected recall@k for a random ranker. + + E[recall@k] = min(k, n_positives) / n_positives when sampling without replacement. + """ + if n_positives == 0 or n_candidates == 0: + return 0.0 + expected_hits = min(k, n_positives) * min(k, n_candidates) / max(n_candidates, 1) + return round(min(1.0, expected_hits / n_positives), 4) + + +def enrichment_factor( + scored: list[ScoredCandidate], + positive_ids: set[str], + k: int, +) -> float: + """Enrichment factor at k: recall@k relative to random baseline recall@k. + + EF > 1.0 means the pipeline outperforms random. + EF = 1.0 is random. + EF < 1.0 is worse than random (anti-enrichment). + """ + n = len(scored) + n_pos = len(positive_ids) + rc = recall_at_k(scored, positive_ids, k) + random_rc = random_recall_at_k(n, n_pos, k) + if random_rc == 0.0: + return 0.0 + return round(rc / random_rc, 4) + + +def benchmark_summary( + scored: list[ScoredCandidate], + positive_ids: set[str], + ks: list[int] | None = None, +) -> dict: + """Produce a benchmark summary comparing pipeline vs random ranker. + + Returns recall@k and enrichment factor at each k, plus a verdict. + All results are computational only — they do not prove biological activity. + """ + if ks is None: + n = len(scored) + ks = sorted({max(1, n // 10), max(1, n // 5), max(1, n // 2), n}) + + results = [] + for k in ks: + rc = recall_at_k(scored, positive_ids, k) + rrc = random_recall_at_k(len(scored), len(positive_ids), k) + ef = enrichment_factor(scored, positive_ids, k) + results.append( + { + "k": k, + "recall_at_k": rc, + "random_recall_at_k": rrc, + "enrichment_factor": ef, + } + ) + + any_enrichment = any(r["enrichment_factor"] > 1.0 for r in results) + verdict = "pipeline outperforms random" if any_enrichment else "pipeline does not outperform random" + + return { + "disclaimer": ( + "These are retrospective benchmark results on demo data. " + "They do not prove biological efficacy. " + "They only measure whether the pipeline recovers known positives " + "better than a random ranker would." + ), + "n_candidates": len(scored), + "n_positives": len(positive_ids), + "results": results, + "verdict": verdict, + } + + +def find_contaminated_references( + candidate_sequences: list[str], + reference_sequences: list[str], + positive_ids: set[str], + candidate_ids: list[str], + threshold: float = 0.70, +) -> set[int]: + """Return indices of references that are near-duplicates of test positives. + + Used to build a cluster-split reference set: removing contaminated references + ensures benchmark performance is not inflated by reference-set memorization. + """ + contaminated: set[int] = set() + positive_seqs = { + seq + for seq, cid in zip(candidate_sequences, candidate_ids) + if cid in positive_ids + } + for ref_idx, ref_seq in enumerate(reference_sequences): + for pos_seq in positive_seqs: + if normalized_similarity(ref_seq, pos_seq) >= threshold: + contaminated.add(ref_idx) + break + return contaminated diff --git a/src/openamp_foundry/benchmark/splits.py b/src/openamp_foundry/benchmark/splits.py index 14c14209..4c6c6598 100644 --- a/src/openamp_foundry/benchmark/splits.py +++ b/src/openamp_foundry/benchmark/splits.py @@ -1,5 +1,6 @@ from __future__ import annotations +from openamp_foundry.scoring.novelty import normalized_similarity from openamp_foundry.types import PeptideCandidate @@ -14,3 +15,53 @@ def deterministic_split( else: train.append(cand) return train, holdout + + +def cluster_by_similarity( + sequences: list[str], + threshold: float = 0.70, +) -> list[list[int]]: + """Greedy single-linkage clustering: sequences within threshold go to the same cluster. + + Returns a list of clusters, where each cluster is a list of indices into `sequences`. + The first sequence assigned to a cluster becomes its center for subsequent comparisons. + threshold: sequences with normalized_similarity >= threshold are co-clustered. + """ + clusters: list[list[int]] = [] + centers: list[str] = [] + + for idx, seq in enumerate(sequences): + assigned = False + for ci, center in enumerate(centers): + if normalized_similarity(seq, center) >= threshold: + clusters[ci].append(idx) + assigned = True + break + if not assigned: + clusters.append([idx]) + centers.append(seq) + + return clusters + + +def cluster_split( + sequences: list[str], + threshold: float = 0.70, +) -> tuple[list[int], list[int]]: + """Split sequences into reference and test partitions by cluster membership. + + The first member of each cluster becomes the reference representative. + All subsequent cluster members become test (held-out) sequences. + + This ensures no test sequence has a near-duplicate in the reference set, + preventing benchmark inflation from reference-set memorization. + + Returns (reference_indices, test_indices). + """ + clusters = cluster_by_similarity(sequences, threshold) + reference_indices: list[int] = [] + test_indices: list[int] = [] + for cluster in clusters: + reference_indices.append(cluster[0]) + test_indices.extend(cluster[1:]) + return reference_indices, test_indices diff --git a/src/openamp_foundry/cli.py b/src/openamp_foundry/cli.py index 55a5d064..e2599712 100644 --- a/src/openamp_foundry/cli.py +++ b/src/openamp_foundry/cli.py @@ -35,6 +35,93 @@ def build_parser() -> argparse.ArgumentParser: leakage.add_argument("--threshold", type=float, default=0.90) leakage.add_argument("--out", required=False, help="Optional JSON output path.") + baseline = bench_sub.add_parser( + "baseline", + help="Evaluate pipeline recall vs random baseline on a labelled set.", + ) + baseline.add_argument("--candidates", required=True, help="CSV of all candidates to score.") + baseline.add_argument("--references", required=False, help="Reference CSV for novelty scoring.") + baseline.add_argument( + "--positives", + required=True, + help="CSV of known-positive (active) peptides. IDs must match candidates CSV.", + ) + baseline.add_argument( + "--k", + type=int, + nargs="+", + default=None, + help="Recall@k cutoffs (default: auto from dataset size).", + ) + baseline.add_argument("--config", default="configs/pipeline.yaml") + baseline.add_argument("--out", required=False, help="Optional JSON output path.") + + generate = sub.add_parser( + "generate-batch", + help=( + "Generate a candidate pool by conservative mutation of seed sequences. " + "Output is a CSV suitable for the 'rank' command. " + "This is a toy exploration tool — no biological activity is implied." + ), + ) + generate.add_argument( + "--seeds", + required=True, + help="CSV of seed sequences (columns: id, sequence, source)", + ) + generate.add_argument( + "--out", + required=True, + help="Output CSV path for the candidate pool.", + ) + generate.add_argument( + "--n-double", + type=int, + default=25, + help="Double-substitution variants per seed (default: 25)", + ) + generate.add_argument( + "--n-charge", + type=int, + default=12, + help="Charge-enhanced variants per seed (default: 12)", + ) + generate.add_argument( + "--rng-seed", + type=int, + default=2024, + help="RNG seed for reproducibility (default: 2024)", + ) + + batch_pack = sub.add_parser( + "batch-pack", + help=( + "Generate Phase 3 batch pack reports (diversity, novelty, toxicity, synthesis) " + "from a ranked JSONL file produced by 'rank'." + ), + ) + batch_pack.add_argument( + "--ranked", + required=True, + help="Ranked JSONL file (output of the 'rank' command).", + ) + batch_pack.add_argument( + "--out-json", + required=True, + help="Output path for machine-readable batch pack JSON.", + ) + batch_pack.add_argument( + "--out-md", + required=False, + help="Optional output path for human-readable markdown report.", + ) + batch_pack.add_argument( + "--diversity-threshold", + type=float, + default=0.80, + help="Similarity threshold for diversity clustering (default: 0.80)", + ) + return parser @@ -64,16 +151,23 @@ def main(argv: list[str] | None = None) -> int: if args.command == "bench": return _run_bench(args) + if args.command == "generate-batch": + return _run_generate_batch(args) + + if args.command == "batch-pack": + return _run_batch_pack(args) + parser.error("unknown command") return 2 def _run_bench(args: argparse.Namespace) -> int: - from openamp_foundry.benchmark.leakage import find_near_duplicates from openamp_foundry.data.loaders import load_candidates_csv from openamp_foundry.utils.io import write_json if args.bench_command == "leakage": + from openamp_foundry.benchmark.leakage import find_near_duplicates + candidates = load_candidates_csv(args.candidates) references = load_candidates_csv(args.references) hits = find_near_duplicates(candidates, references, threshold=args.threshold) @@ -92,8 +186,97 @@ def _run_bench(args: argparse.Namespace) -> int: print(json.dumps(result, indent=2)) return 0 + if args.bench_command == "baseline": + from openamp_foundry.benchmark.evaluate import benchmark_summary + from openamp_foundry.pipeline import score_candidates + + scored, _ = score_candidates( + candidate_path=args.candidates, + reference_path=args.references, + config_path=args.config, + ) + positives = load_candidates_csv(args.positives) + positive_ids = {p.candidate_id for p in positives} + summary = benchmark_summary(scored, positive_ids, ks=args.k) + result = {"status": "ok", **summary} + if args.out: + write_json(args.out, result) + print(json.dumps(result, indent=2)) + return 0 + return 2 +def _run_batch_pack(args: argparse.Namespace) -> int: + from openamp_foundry.reports.batch_pack import generate_batch_pack, write_batch_pack_markdown + from openamp_foundry.utils.io import write_json + + pack = generate_batch_pack( + ranked_jsonl_path=args.ranked, + diversity_threshold=args.diversity_threshold, + ) + write_json(args.out_json, pack) + if args.out_md: + write_batch_pack_markdown(pack, args.out_md) + + print(json.dumps({ + "status": "ok", + "n_selected": pack["summary"]["n_candidates_selected"], + "n_clusters": pack["summary"]["n_diversity_clusters"], + "mean_novelty": pack["summary"]["mean_novelty"], + "mean_safety": pack["summary"]["mean_safety"], + "mean_synthesis": pack["summary"]["mean_synthesis"], + "out_json": args.out_json, + "out_md": args.out_md, + }, indent=2)) + return 0 + + +def _run_generate_batch(args: argparse.Namespace) -> int: + import csv + + from openamp_foundry.generators.template_mutator import generate_candidate_pool + + seeds_path = Path(args.seeds) + if not seeds_path.exists(): + print(json.dumps({"status": "error", "message": f"Seeds file not found: {args.seeds}"})) + return 1 + + seed_ids: list[str] = [] + seed_seqs: list[str] = [] + with open(seeds_path, newline="", encoding="utf-8") as f: + for row in csv.DictReader(f): + seed_ids.append(row["id"]) + seed_seqs.append(row["sequence"].strip().upper()) + + pool = generate_candidate_pool( + seed_sequences=seed_seqs, + seed_ids=seed_ids, + n_double=args.n_double, + n_charge_enhance=args.n_charge, + rng_seed=args.rng_seed, + ) + + out_path = Path(args.out) + out_path.parent.mkdir(parents=True, exist_ok=True) + with open(out_path, "w", newline="", encoding="utf-8") as f: + writer = csv.DictWriter(f, fieldnames=["id", "sequence", "source"]) + writer.writeheader() + writer.writerows(pool) + + print(json.dumps({ + "status": "ok", + "n_seeds": len(seed_ids), + "n_candidates_generated": len(pool), + "out": str(out_path), + "disclaimer": ( + "Generated candidates are toy conservative-substitution variants. " + "They have no demonstrated biological activity. " + "Run 'rank' to score and filter them." + ), + }, indent=2)) + return 0 + + if __name__ == "__main__": raise SystemExit(main()) diff --git a/src/openamp_foundry/data/lab_results.py b/src/openamp_foundry/data/lab_results.py new file mode 100644 index 00000000..cf5f3c86 --- /dev/null +++ b/src/openamp_foundry/data/lab_results.py @@ -0,0 +1,106 @@ +"""Lab results ingestion for the active-learning loop. + +Loads and validates lab assay results against schemas/lab_result.schema.json. +Results feed the active-learning cycle: hits → stronger signal for next generation, +misses → updated negative dataset. + +IMPORTANT: This module records experimental data only. It does not prove efficacy, +safety, or suitability for any use. All biological claims require qualified expert +review and independent replication. +""" +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +from openamp_foundry.evidence.schemas import validate_json_schema + +LAB_RESULT_SCHEMA = Path(__file__).parent.parent.parent.parent / "schemas" / "lab_result.schema.json" + + +def load_lab_result(path: str | Path) -> dict[str, Any]: + """Load and validate a single lab result JSON file. + + Raises ValueError if the file does not validate against lab_result.schema.json. + """ + p = Path(path) + with p.open("r", encoding="utf-8") as f: + result = json.load(f) + validate_json_schema(result, LAB_RESULT_SCHEMA) + return result + + +def load_lab_results_dir(directory: str | Path) -> list[dict[str, Any]]: + """Load all lab result JSON files from a directory. + + Returns a list of validated result dicts, sorted by assay_date then result_id. + Skips files that fail schema validation with a warning. + """ + d = Path(directory) + results: list[dict[str, Any]] = [] + errors: list[str] = [] + for p in sorted(d.glob("*.json")): + try: + results.append(load_lab_result(p)) + except Exception as exc: + errors.append(f"{p.name}: {exc}") + if errors: + import warnings + warnings.warn( + f"Skipped {len(errors)} invalid lab result files:\n" + "\n".join(errors), + stacklevel=2, + ) + return sorted(results, key=lambda r: (r.get("assay_date", ""), r.get("result_id", ""))) + + +def summarise_lab_results(results: list[dict[str, Any]]) -> dict[str, Any]: + """Produce a summary of lab results for a candidate batch. + + Returns counts by assay_type, qualitative result, and control status. + All findings are raw experimental observations, not validated biological claims. + """ + n = len(results) + if n == 0: + return { + "n_results": 0, + "disclaimer": "No lab results loaded.", + } + + by_type: dict[str, int] = {} + by_qualitative: dict[str, int] = {} + n_controls_ok = 0 + + for r in results: + assay = r.get("assay_type", "other") + by_type[assay] = by_type.get(assay, 0) + 1 + + qual = r.get("result_qualitative") or "unclassified" + by_qualitative[qual] = by_qualitative.get(qual, 0) + 1 + + if r.get("positive_control_passed") and r.get("negative_control_passed"): + n_controls_ok += 1 + + return { + "n_results": n, + "n_valid_controls": n_controls_ok, + "by_assay_type": by_type, + "by_qualitative_result": by_qualitative, + "disclaimer": ( + "Lab result summary. Raw experimental observations only. " + "Not a validated drug efficacy, safety, or clinical claim. " + "All results require qualified expert interpretation and independent replication." + ), + } + + +def candidate_result_map(results: list[dict[str, Any]]) -> dict[str, list[dict[str, Any]]]: + """Map candidate_id → list of lab results. + + Allows quickly looking up all assay results for a given candidate. + """ + mapping: dict[str, list[dict[str, Any]]] = {} + for r in results: + cid = r["candidate_id"] + mapping.setdefault(cid, []).append(r) + return mapping diff --git a/src/openamp_foundry/features/physchem.py b/src/openamp_foundry/features/physchem.py index 9ff9f9df..4d77a7dc 100644 --- a/src/openamp_foundry/features/physchem.py +++ b/src/openamp_foundry/features/physchem.py @@ -1,13 +1,25 @@ from __future__ import annotations +import math from collections import Counter +from openamp_foundry.scoring.boman import boman_index, gravy_score + HYDROPHOBIC = set("AILMFWVY") POSITIVE = set("KRH") NEGATIVE = set("DE") AROMATIC = set("FWY") CYS = "C" +# Eisenberg consensus hydrophobicity scale (normalized, 0-centred removed, shifted to 0..1 range) +# Source: Eisenberg et al. (1984) J Mol Biol. Used for hydrophobic moment only. +_HYDROPHOBICITY: dict[str, float] = { + "A": 0.620, "R": -2.530, "N": -0.780, "D": -0.900, "C": 0.290, + "Q": -0.850, "E": -0.740, "G": 0.480, "H": -0.400, "I": 1.380, + "L": 1.060, "K": -1.500, "M": 0.640, "F": 1.190, "P": 0.120, + "S": -0.180, "T": -0.050, "W": 0.810, "Y": 0.260, "V": 1.080, +} + def net_charge_proxy(sequence: str) -> int: return sum(1 for aa in sequence if aa in POSITIVE) - sum(1 for aa in sequence if aa in NEGATIVE) @@ -33,6 +45,29 @@ def longest_repeat_run(sequence: str) -> int: return best +def hydrophobic_moment(sequence: str, angle_deg: float = 100.0) -> float: + """Compute mean hydrophobic moment for an assumed alpha-helical conformation. + + Uses the Eisenberg (1984) consensus scale. Angle of 100° per residue is the + standard helical wheel projection used in AMP literature. + + Returns a value in [0, ∞); typical AMPs have μH > 0.4. Not normalised to [0,1] + because the range is sequence-length-dependent — callers should normalise if needed. + """ + if not sequence: + return 0.0 + angle_rad = math.radians(angle_deg) + sin_sum = 0.0 + cos_sum = 0.0 + for i, aa in enumerate(sequence): + h = _HYDROPHOBICITY.get(aa, 0.0) + theta = i * angle_rad + sin_sum += h * math.sin(theta) + cos_sum += h * math.cos(theta) + moment = math.sqrt(sin_sum ** 2 + cos_sum ** 2) / len(sequence) + return round(moment, 4) + + def compute_features(sequence: str) -> dict[str, float | int | dict[str, int]]: counts = Counter(sequence) length = len(sequence) @@ -43,6 +78,7 @@ def compute_features(sequence: str) -> dict[str, float | int | dict[str, int]]: gly_fraction = counts.get("G", 0) / length if length else 0.0 pro_fraction = counts.get("P", 0) / length if length else 0.0 repeat_run = longest_repeat_run(sequence) + mu_h = hydrophobic_moment(sequence) return { "length": length, "net_charge_proxy": charge, @@ -53,5 +89,8 @@ def compute_features(sequence: str) -> dict[str, float | int | dict[str, int]]: "glycine_fraction": round(gly_fraction, 4), "proline_fraction": round(pro_fraction, 4), "longest_repeat_run": repeat_run, + "hydrophobic_moment": mu_h, + "boman_index": boman_index(sequence), + "gravy": gravy_score(sequence), "residue_counts": dict(sorted(counts.items())), } diff --git a/src/openamp_foundry/generators/template_mutator.py b/src/openamp_foundry/generators/template_mutator.py new file mode 100644 index 00000000..91d0b23d --- /dev/null +++ b/src/openamp_foundry/generators/template_mutator.py @@ -0,0 +1,170 @@ +"""Systematic template-based candidate generator. + +This is a deliberately simple, bounded mutation explorer — not a learned model. +It generates AMP candidate variants by applying conservative substitutions to +known AMP-like template sequences. + +Conservative substitution groups (physicochemically similar): + Cationic: K, R, H (positive charge, preserve charge character) + Hydrophobic: L, I, V, A, F, M (preserve hydrophobic character) + Polar/neutral: S, T, N, Q (no charge) + Aromatic: W, Y, F (aromatic, hydrophobic) + +Design rules: + - Preserve backbone length (no insertions/deletions in this generator) + - Never target charged residues for hydrophobic substitution (or vice versa) + - Use deterministic seeding so output is reproducible + +This generator does NOT claim to produce biologically active peptides. +All generated sequences must pass through the full pipeline safety filters +before any nomination. The lab is the judge. +""" +from __future__ import annotations + +import random + +CANONICAL_AA = "ACDEFGHIKLMNPQRSTVWY" + +# Conservative substitution groups: within-group swaps preserve physicochemical character +CONSERVATIVE_GROUPS: list[frozenset[str]] = [ + frozenset("KRH"), # cationic + frozenset("LIVAFM"), # aliphatic/hydrophobic + frozenset("FWY"), # aromatic + frozenset("STNQ"), # polar/neutral + frozenset("DE"), # anionic + frozenset("GP"), # conformational (glycine, proline — treat conservatively) + frozenset("C"), # cysteine — singleton; substitutions restricted +] + + +def _conservative_substitutes(aa: str) -> list[str]: + """Return amino acids that can conservatively substitute for `aa`.""" + for group in CONSERVATIVE_GROUPS: + if aa in group: + return [a for a in sorted(group) if a != aa] + return [] + + +def generate_single_substitution_variants(sequence: str) -> list[str]: + """Generate all single-substitution conservative variants of a template. + + For each position, substitutes with each member of the same physicochemical group. + Returns up to len(sequence) × (group_size - 1) variants. + """ + variants = [] + for i, aa in enumerate(sequence): + subs = _conservative_substitutes(aa) + for sub in subs: + new_seq = sequence[:i] + sub + sequence[i + 1:] + variants.append(new_seq) + return variants + + +def generate_double_substitution_variants( + sequence: str, + n_samples: int = 20, + seed: int = 42, +) -> list[str]: + """Sample random pairs of conservative positions and substitute both. + + Returns up to n_samples unique variants. + """ + rng = random.Random(seed) + substitutable = [ + (i, _conservative_substitutes(aa)) + for i, aa in enumerate(sequence) + if _conservative_substitutes(aa) + ] + if len(substitutable) < 2: + return [] + + seen: set[str] = set() + variants = [] + attempts = 0 + while len(variants) < n_samples and attempts < n_samples * 10: + attempts += 1 + pos_a, subs_a = rng.choice(substitutable) + pos_b, subs_b = rng.choice(substitutable) + if pos_a == pos_b: + continue + sub_a = rng.choice(subs_a) + sub_b = rng.choice(subs_b) + seq = list(sequence) + seq[pos_a] = sub_a + seq[pos_b] = sub_b + new_seq = "".join(seq) + if new_seq not in seen: + seen.add(new_seq) + variants.append(new_seq) + return variants + + +def generate_charge_enhanced_variants( + sequence: str, + n_samples: int = 10, + seed: int = 42, +) -> list[str]: + """Replace uncharged positions with K or R to boost predicted charge density. + + Targets positions occupied by non-charged hydrophilic residues (S, T, N, Q) + and replaces with K (higher charge density → better activity likeness). + """ + rng = random.Random(seed) + targetable = [ + i for i, aa in enumerate(sequence) if aa in "STNQ" + ] + if not targetable: + return [] + + seen: set[str] = set() + variants = [] + for i in targetable[:n_samples]: + replacement = rng.choice(["K", "R"]) + new_seq = sequence[:i] + replacement + sequence[i + 1:] + if new_seq not in seen: + seen.add(new_seq) + variants.append(new_seq) + return variants + + +def generate_all_variants( + seed_sequence: str, + n_double: int = 15, + n_charge_enhance: int = 8, + seed: int = 42, +) -> list[str]: + """Generate all candidate variants for a template sequence. + + Combines single substitutions, double substitutions, and charge-enhanced variants. + Deduplicates and removes the seed itself. + """ + variants: set[str] = set() + variants.update(generate_single_substitution_variants(seed_sequence)) + variants.update(generate_double_substitution_variants(seed_sequence, n_double, seed)) + variants.update(generate_charge_enhanced_variants(seed_sequence, n_charge_enhance, seed)) + variants.discard(seed_sequence) + return sorted(variants) + + +def generate_candidate_pool( + seed_sequences: list[str], + seed_ids: list[str], + n_double: int = 15, + n_charge_enhance: int = 8, + rng_seed: int = 42, +) -> list[dict[str, str]]: + """Generate a candidate pool from multiple seed templates. + + Returns a list of dicts with 'id', 'sequence', 'source' fields. + IDs are formatted as {seed_id}_VAR_{index:03d}. + """ + candidates = [] + for seed_id, seed_seq in zip(seed_ids, seed_sequences): + variants = generate_all_variants(seed_seq, n_double, n_charge_enhance, rng_seed) + for idx, variant in enumerate(variants, start=1): + candidates.append({ + "id": f"{seed_id}_VAR_{idx:03d}", + "sequence": variant, + "source": f"template_mutation_from_{seed_id}", + }) + return candidates diff --git a/src/openamp_foundry/pipeline.py b/src/openamp_foundry/pipeline.py index 9a0d0d14..e231d22f 100644 --- a/src/openamp_foundry/pipeline.py +++ b/src/openamp_foundry/pipeline.py @@ -1,6 +1,5 @@ from __future__ import annotations -import hashlib import uuid from datetime import datetime, timezone from pathlib import Path @@ -12,6 +11,7 @@ from openamp_foundry.evidence.certificate import build_certificate from openamp_foundry.features.physchem import compute_features from openamp_foundry.scoring.activity import activity_likeness_score +from openamp_foundry.scoring.boman import boman_activity_score, model_disagreement from openamp_foundry.scoring.ensemble import ensemble_score, known_failure_modes, selection_reasons from openamp_foundry.scoring.novelty import novelty_score from openamp_foundry.scoring.safety import safety_score @@ -52,11 +52,14 @@ def score_candidates( safe = safety_score(features) if valid else 0.0 synth = synthesis_feasibility_score(features, valid_sequence=valid) nov, nearest = novelty_score(candidate.sequence, references) + boman_act = boman_activity_score(candidate.sequence) if valid else 0.0 raw_scores = { "activity": act, "safety": safe, "synthesis": synth, "novelty": nov, + "boman_activity": boman_act, + "disagreement": model_disagreement(act, boman_act), } raw_scores["ensemble"] = ensemble_score(raw_scores, weights) item = ScoredCandidate( @@ -159,6 +162,9 @@ def run_ranking_pipeline( if report_path: write_report(report_path, ranked, selected) + batch_report = build_batch_report(ranked, selected, generated_at) + report_json = Path(report_path).with_suffix(".json") + write_json(report_json, batch_report) output_paths = [str(out_path)] if report_path: @@ -185,6 +191,41 @@ def run_ranking_pipeline( return ranked +def build_batch_report( + ranked: list[ScoredCandidate], + selected: list[ScoredCandidate], + generated_at: str, +) -> dict[str, Any]: + """Build a machine-readable batch report validatable against batch_report.schema.json.""" + selected_ids = {item.candidate.candidate_id for item in selected} + return { + "pipeline_version": __version__, + "candidate_count": len(ranked), + "selected_count": len(selected), + "generated_at": generated_at, + "disclaimer": ( + "All scores are transparent baseline heuristics. " + "They are not validated biological predictors. " + "No antimicrobial activity has been demonstrated." + ), + "score_averages": { + "activity": round( + sum(s.scores["activity"] for s in ranked) / len(ranked), 4 + ) if ranked else 0.0, + "safety": round( + sum(s.scores["safety"] for s in ranked) / len(ranked), 4 + ) if ranked else 0.0, + "novelty": round( + sum(s.scores["novelty"] for s in ranked) / len(ranked), 4 + ) if ranked else 0.0, + "ensemble": round( + sum(s.scores["ensemble"] for s in ranked) / len(ranked), 4 + ) if ranked else 0.0, + }, + "selected_ids": sorted(selected_ids), + } + + def write_report( path: str | Path, ranked: list[ScoredCandidate], selected: list[ScoredCandidate] ) -> None: diff --git a/src/openamp_foundry/reports/__init__.py b/src/openamp_foundry/reports/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/src/openamp_foundry/reports/batch_pack.py b/src/openamp_foundry/reports/batch_pack.py new file mode 100644 index 00000000..a06c5afa --- /dev/null +++ b/src/openamp_foundry/reports/batch_pack.py @@ -0,0 +1,537 @@ +"""Batch pack report generator for Phase 3 lab-ready candidate batches. + +Reads the ranked JSONL output and produces all four required sub-reports: + 1. Diversity clustering report + 2. Novelty report + 3. Toxicity/hemolysis risk report + 4. Synthesis feasibility report + +All sub-reports are computational heuristics only. +No biological claims are made. The lab is the judge. +""" +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +from openamp_foundry.benchmark.splits import cluster_by_similarity + + +def _load_ranked_jsonl(path: str | Path) -> list[dict[str, Any]]: + rows = [] + with open(path, encoding="utf-8") as f: + for line in f: + line = line.strip() + if line: + rows.append(json.loads(line)) + return rows + + +def _selected(rows: list[dict[str, Any]]) -> list[dict[str, Any]]: + return [r for r in rows if r.get("selected", False)] + + +# --------------------------------------------------------------------------- +# 1. Diversity clustering report +# --------------------------------------------------------------------------- + +def diversity_clustering_report( + selected: list[dict[str, Any]], + threshold: float = 0.80, +) -> dict[str, Any]: + """Cluster the selected candidates by pairwise similarity. + + Reports how many distinct clusters exist and the within-cluster size distribution. + A higher number of singleton clusters indicates greater diversity. + """ + seqs = [r["sequence"] for r in selected] + ids = [r["candidate_id"] for r in selected] + clusters = cluster_by_similarity(seqs, threshold=threshold) + + cluster_details = [] + for cluster in clusters: + cluster_details.append({ + "size": len(cluster), + "members": [ids[i] for i in cluster], + "representative": ids[cluster[0]], + }) + + n_singletons = sum(1 for c in cluster_details if c["size"] == 1) + max_cluster_size = max(c["size"] for c in cluster_details) if cluster_details else 0 + + return { + "report_type": "diversity_clustering", + "disclaimer": ( + "Pairwise similarity is computed by normalised Levenshtein distance. " + "Structural or functional diversity is not assessed. " + "This is a sequence-level diversity proxy only." + ), + "n_selected": len(selected), + "similarity_threshold": threshold, + "n_clusters": len(cluster_details), + "n_singleton_clusters": n_singletons, + "max_cluster_size": max_cluster_size, + "singleton_fraction": round(n_singletons / len(cluster_details), 4) if cluster_details else 0.0, + "clusters": cluster_details, + } + + +# --------------------------------------------------------------------------- +# 2. Novelty report +# --------------------------------------------------------------------------- + +def novelty_report(selected: list[dict[str, Any]]) -> dict[str, Any]: + """Summarise novelty scores across the selected batch. + + Novelty is the fraction of positions that differ from the nearest known reference. + A novelty of 0.0 means the sequence is identical to a reference. + A novelty of 1.0 means no similar reference exists. + """ + novelty_scores = [r["scores"]["novelty"] for r in selected] + + n_high_novelty = sum(1 for s in novelty_scores if s >= 0.30) + n_mid_novelty = sum(1 for s in novelty_scores if 0.10 <= s < 0.30) + n_low_novelty = sum(1 for s in novelty_scores if s < 0.10) + + mean_novelty = sum(novelty_scores) / len(novelty_scores) if novelty_scores else 0.0 + min_novelty = min(novelty_scores) if novelty_scores else 0.0 + max_novelty = max(novelty_scores) if novelty_scores else 0.0 + + # Per-candidate detail + details = [] + for r in sorted(selected, key=lambda x: x["scores"]["novelty"], reverse=True): + details.append({ + "candidate_id": r["candidate_id"], + "novelty": r["scores"]["novelty"], + "nearest_reference": r.get("nearest_reference"), + }) + + return { + "report_type": "novelty_report", + "disclaimer": ( + "Novelty is computed as 1 - normalised_Levenshtein_similarity against the nearest " + "reference sequence. This is a sequence-level proxy. Structural or functional " + "novelty is not assessed. A high novelty score does not guarantee biological novelty." + ), + "n_selected": len(selected), + "mean_novelty": round(mean_novelty, 4), + "min_novelty": round(min_novelty, 4), + "max_novelty": round(max_novelty, 4), + "n_high_novelty_ge_0_30": n_high_novelty, + "n_mid_novelty_0_10_to_0_30": n_mid_novelty, + "n_low_novelty_lt_0_10": n_low_novelty, + "candidates": details, + } + + +# --------------------------------------------------------------------------- +# 3. Toxicity / hemolysis risk report +# --------------------------------------------------------------------------- + +_TOXICITY_THRESHOLDS = { + "hydrophobic_fraction_high": 0.65, + "charge_density_high": 0.55, + "cysteine_fraction_high": 0.25, + "longest_repeat_run_high": 6, + "length_high": 35, +} + + +def _toxicity_flags(features: dict[str, Any]) -> list[str]: + flags: list[str] = [] + if features.get("hydrophobic_fraction", 0.0) > _TOXICITY_THRESHOLDS["hydrophobic_fraction_high"]: + flags.append("high_hydrophobic_fraction") + if features.get("charge_density", 0.0) > _TOXICITY_THRESHOLDS["charge_density_high"]: + flags.append("high_charge_density") + if features.get("cysteine_fraction", 0.0) > _TOXICITY_THRESHOLDS["cysteine_fraction_high"]: + flags.append("high_cysteine_fraction") + if features.get("longest_repeat_run", 0) >= _TOXICITY_THRESHOLDS["longest_repeat_run_high"]: + flags.append("long_repeat_run") + if features.get("length", 0) > _TOXICITY_THRESHOLDS["length_high"]: + flags.append("length_exceeds_35aa") + return flags + + +def toxicity_hemolysis_risk_report(selected: list[dict[str, Any]]) -> dict[str, Any]: + """Summarise predicted toxicity/hemolysis risk signals for selected candidates. + + Risk signals are computational proxies based on physicochemical features. + They are NOT validated hemolysis or cytotoxicity predictors. + """ + safety_scores = [r["scores"]["safety"] for r in selected] + mean_safety = sum(safety_scores) / len(safety_scores) if safety_scores else 0.0 + + per_candidate = [] + n_flagged = 0 + for r in selected: + flags = _toxicity_flags(r.get("features", {})) + if flags: + n_flagged += 1 + per_candidate.append({ + "candidate_id": r["candidate_id"], + "safety_score": r["scores"]["safety"], + "hydrophobic_fraction": r.get("features", {}).get("hydrophobic_fraction"), + "charge_density": r.get("features", {}).get("charge_density"), + "cysteine_fraction": r.get("features", {}).get("cysteine_fraction"), + "longest_repeat_run": r.get("features", {}).get("longest_repeat_run"), + "risk_flags": flags, + }) + + per_candidate.sort(key=lambda x: x["safety_score"]) + + return { + "report_type": "toxicity_hemolysis_risk", + "disclaimer": ( + "Risk flags are physicochemical heuristics — they are NOT validated predictors " + "of hemolysis, cytotoxicity, or mammalian toxicity. " + "All candidates require wet-lab safety profiling before any use. " + "High safety score does not mean the peptide is safe." + ), + "n_selected": len(selected), + "mean_safety_score": round(mean_safety, 4), + "n_with_risk_flags": n_flagged, + "risk_thresholds": _TOXICITY_THRESHOLDS, + "candidates": per_candidate, + } + + +# --------------------------------------------------------------------------- +# 4. Synthesis feasibility report +# --------------------------------------------------------------------------- + +def synthesis_feasibility_report(selected: list[dict[str, Any]]) -> dict[str, Any]: + """Summarise predicted synthesis feasibility for selected candidates. + + Synthesis score is a heuristic based on sequence composition and length. + It is not a validated synthesis prediction. + """ + synth_scores = [r["scores"]["synthesis"] for r in selected] + mean_synth = sum(synth_scores) / len(synth_scores) if synth_scores else 0.0 + + n_high = sum(1 for s in synth_scores if s >= 0.80) + n_mid = sum(1 for s in synth_scores if 0.50 <= s < 0.80) + n_low = sum(1 for s in synth_scores if s < 0.50) + + per_candidate = [] + for r in sorted(selected, key=lambda x: x["scores"]["synthesis"]): + features = r.get("features", {}) + per_candidate.append({ + "candidate_id": r["candidate_id"], + "synthesis_score": r["scores"]["synthesis"], + "sequence": r["sequence"], + "length": features.get("length"), + "cysteine_fraction": features.get("cysteine_fraction"), + "proline_fraction": features.get("proline_fraction"), + "longest_repeat_run": features.get("longest_repeat_run"), + }) + + return { + "report_type": "synthesis_feasibility", + "disclaimer": ( + "Synthesis feasibility score is a heuristic based on length, cysteine content, " + "proline content, and repeat runs. It is not a validated synthesis prediction. " + "All candidates should be reviewed by a peptide synthesis expert before ordering." + ), + "n_selected": len(selected), + "mean_synthesis_score": round(mean_synth, 4), + "n_high_feasibility_ge_0_80": n_high, + "n_mid_feasibility_0_50_to_0_80": n_mid, + "n_low_feasibility_lt_0_50": n_low, + "candidates": per_candidate, + } + + +# --------------------------------------------------------------------------- +# 5. Scorer consensus (Boman index vs. activity-likeness disagreement) +# --------------------------------------------------------------------------- + +def scorer_consensus_report(selected: list[dict[str, Any]]) -> dict[str, Any]: + """Summarise dual-scorer consensus across selected candidates. + + Disagreement = |activity_likeness − boman_activity|. + Low disagreement (< 0.20): both independent scorers agree → more robust nomination. + High disagreement (≥ 0.30): scorers diverge → extra scrutiny recommended. + """ + per_candidate = [] + for r in selected: + scores = r.get("scores", {}) + act = scores.get("activity") + boman_act = scores.get("boman_activity") + disagreement = scores.get("disagreement") + if act is None or boman_act is None: + continue + per_candidate.append({ + "candidate_id": r["candidate_id"], + "activity_likeness": act, + "boman_activity": boman_act, + "disagreement": disagreement, + "consensus_label": ( + "high_consensus" if (disagreement is not None and disagreement < 0.20) + else "uncertain" if (disagreement is not None and disagreement >= 0.30) + else "moderate" + ), + }) + + per_candidate.sort(key=lambda x: x.get("disagreement") or 1.0) + + n_high_consensus = sum(1 for c in per_candidate if c["consensus_label"] == "high_consensus") + n_uncertain = sum(1 for c in per_candidate if c["consensus_label"] == "uncertain") + n_moderate = sum(1 for c in per_candidate if c["consensus_label"] == "moderate") + + disagreements = [c["disagreement"] for c in per_candidate if c["disagreement"] is not None] + mean_disagreement = sum(disagreements) / len(disagreements) if disagreements else 0.0 + + return { + "report_type": "scorer_consensus", + "disclaimer": ( + "Model disagreement is a computational uncertainty proxy only. " + "It is NOT a biological activity or safety measure. " + "High consensus means two independent heuristics agree; " + "it does not increase the probability of lab activity. " + "The lab is the judge." + ), + "n_selected": len(selected), + "n_scored_with_boman": len(per_candidate), + "n_high_consensus_lt_0_20": n_high_consensus, + "n_moderate_0_20_to_0_30": n_moderate, + "n_uncertain_ge_0_30": n_uncertain, + "mean_disagreement": round(mean_disagreement, 4), + "candidates": per_candidate, + } + + +# --------------------------------------------------------------------------- +# Full batch pack +# --------------------------------------------------------------------------- + +def generate_batch_pack( + ranked_jsonl_path: str | Path, + diversity_threshold: float = 0.80, +) -> dict[str, Any]: + """Generate the complete Phase 3 batch pack. + + Returns a single dict containing all four sub-reports and a top-level summary. + """ + rows = _load_ranked_jsonl(ranked_jsonl_path) + sel = _selected(rows) + + diversity = diversity_clustering_report(sel, threshold=diversity_threshold) + novelty = novelty_report(sel) + toxicity = toxicity_hemolysis_risk_report(sel) + synthesis = synthesis_feasibility_report(sel) + consensus = scorer_consensus_report(sel) + + ensemble_scores = [r["scores"]["ensemble"] for r in sel] + mean_ensemble = sum(ensemble_scores) / len(ensemble_scores) if ensemble_scores else 0.0 + + return { + "batch_pack_version": "1.1", + "disclaimer": ( + "All reports in this batch pack are based on computational heuristics. " + "No biological activity has been demonstrated. " + "These candidates are nominated for possible expert review and wet-lab assay. " + "The lab is the judge." + ), + "summary": { + "n_candidates_scored": len(rows), + "n_candidates_selected": len(sel), + "mean_ensemble_score": round(mean_ensemble, 4), + "n_diversity_clusters": diversity["n_clusters"], + "n_singleton_clusters": diversity["n_singleton_clusters"], + "mean_novelty": novelty["mean_novelty"], + "mean_safety": toxicity["mean_safety_score"], + "mean_synthesis": synthesis["mean_synthesis_score"], + "n_high_consensus": consensus["n_high_consensus_lt_0_20"], + "n_uncertain_disagreement": consensus["n_uncertain_ge_0_30"], + "mean_scorer_disagreement": consensus["mean_disagreement"], + }, + "diversity_clustering": diversity, + "novelty_report": novelty, + "toxicity_hemolysis_risk": toxicity, + "synthesis_feasibility": synthesis, + "scorer_consensus": consensus, + } + + +def write_batch_pack_markdown(pack: dict[str, Any], path: str | Path) -> None: + """Write a human-readable markdown version of the batch pack.""" + p = Path(path) + p.parent.mkdir(parents=True, exist_ok=True) + + s = pack["summary"] + div = pack["diversity_clustering"] + nov = pack["novelty_report"] + tox = pack["toxicity_hemolysis_risk"] + syn = pack["synthesis_feasibility"] + con = pack.get("scorer_consensus", {}) + + lines = [ + "# OpenAMP Foundry — Phase 3 Batch Pack", + "", + f"> **{pack['disclaimer']}**", + "", + "---", + "", + "## Summary", + "", + "| Metric | Value |", + "|--------|-------|", + f"| Candidates scored | {s['n_candidates_scored']} |", + f"| Candidates selected | {s['n_candidates_selected']} |", + f"| Mean ensemble score | {s['mean_ensemble_score']:.4f} |", + f"| Diversity clusters | {s['n_diversity_clusters']} |", + f"| Singleton clusters | {s['n_singleton_clusters']} |", + f"| Mean novelty | {s['mean_novelty']:.4f} |", + f"| Mean safety | {s['mean_safety']:.4f} |", + f"| Mean synthesis feasibility | {s['mean_synthesis']:.4f} |", + f"| High scorer consensus (disagreement <0.20) | {s.get('n_high_consensus', 'N/A')} |", + f"| Uncertain (disagreement ≥0.30) | {s.get('n_uncertain_disagreement', 'N/A')} |", + f"| Mean scorer disagreement | {s.get('mean_scorer_disagreement', 0.0):.4f} |", + "", + "---", + "", + "## 1. Diversity Clustering Report", + "", + f"> {div['disclaimer']}", + "", + f"- Similarity threshold: {div['similarity_threshold']}", + f"- Number of clusters: {div['n_clusters']}", + f"- Singleton clusters: {div['n_singleton_clusters']} ({div['singleton_fraction']:.1%})", + f"- Largest cluster size: {div['max_cluster_size']}", + "", + "| Cluster | Size | Representative |", + "|---------|------|----------------|", + ] + for i, cluster in enumerate(div["clusters"][:20], start=1): + lines.append(f"| {i} | {cluster['size']} | {cluster['representative']} |") + if len(div["clusters"]) > 20: + lines.append(f"| … | … | ({len(div['clusters']) - 20} more clusters) |") + + lines += [ + "", + "---", + "", + "## 2. Novelty Report", + "", + f"> {nov['disclaimer']}", + "", + f"- Mean novelty: {nov['mean_novelty']:.4f}", + f"- Min novelty: {nov['min_novelty']:.4f}", + f"- Max novelty: {nov['max_novelty']:.4f}", + f"- High novelty (≥0.30): {nov['n_high_novelty_ge_0_30']}", + f"- Mid novelty (0.10–0.30): {nov['n_mid_novelty_0_10_to_0_30']}", + f"- Low novelty (<0.10): {nov['n_low_novelty_lt_0_10']}", + "", + "Top 10 most novel candidates:", + "", + "| Candidate | Novelty | Nearest Reference |", + "|-----------|---------|-------------------|", + ] + for c in nov["candidates"][:10]: + ref = c["nearest_reference"] or "none" + lines.append(f"| {c['candidate_id']} | {c['novelty']:.4f} | {ref} |") + + lines += [ + "", + "---", + "", + "## 3. Toxicity / Hemolysis Risk Report", + "", + f"> {tox['disclaimer']}", + "", + f"- Mean safety score: {tox['mean_safety_score']:.4f}", + f"- Candidates with risk flags: {tox['n_with_risk_flags']} / {tox['n_selected']}", + "", + "Risk thresholds:", + "", + ] + for k, v in tox["risk_thresholds"].items(): + lines.append(f"- {k}: {v}") + lines += [ + "", + "Candidates by ascending safety score (lowest safety first):", + "", + "| Candidate | Safety | Hydrophobic | Charge density | Risk flags |", + "|-----------|--------|-------------|----------------|------------|", + ] + for c in tox["candidates"][:15]: + flags = ", ".join(c["risk_flags"]) if c["risk_flags"] else "none" + lines.append( + f"| {c['candidate_id']} | {c['safety_score']:.4f} | " + f"{c['hydrophobic_fraction']:.2f} | {c['charge_density']:.2f} | {flags} |" + ) + + lines += [ + "", + "---", + "", + "## 4. Synthesis Feasibility Report", + "", + f"> {syn['disclaimer']}", + "", + f"- Mean synthesis score: {syn['mean_synthesis_score']:.4f}", + f"- High feasibility (≥0.80): {syn['n_high_feasibility_ge_0_80']}", + f"- Mid feasibility (0.50–0.80): {syn['n_mid_feasibility_0_50_to_0_80']}", + f"- Low feasibility (<0.50): {syn['n_low_feasibility_lt_0_50']}", + "", + "Candidates by ascending synthesis score (lowest feasibility first):", + "", + "| Candidate | Synthesis | Length | Cys% | Pro% | Repeat |", + "|-----------|-----------|--------|------|------|--------|", + ] + for c in syn["candidates"][:15]: + cys = f"{(c['cysteine_fraction'] or 0):.2f}" + pro = f"{(c['proline_fraction'] or 0):.2f}" + repeat = c["longest_repeat_run"] or 0 + lines.append( + f"| {c['candidate_id']} | {c['synthesis_score']:.4f} | " + f"{c['length']} | {cys} | {pro} | {repeat} |" + ) + + if con: + lines += [ + "", + "---", + "", + "## 5. Scorer Consensus Report", + "", + f"> {con.get('disclaimer', '')}", + "", + f"- High consensus (disagreement <0.20): {con.get('n_high_consensus_lt_0_20', 'N/A')} candidates", + f"- Moderate disagreement (0.20–0.30): {con.get('n_moderate_0_20_to_0_30', 'N/A')} candidates", + f"- Uncertain (disagreement ≥0.30): {con.get('n_uncertain_ge_0_30', 'N/A')} candidates", + f"- Mean disagreement: {con.get('mean_disagreement', 0.0):.4f}", + "", + "Candidates sorted by disagreement (lowest = strongest dual-scorer consensus):", + "", + "| Candidate | Activity | Boman Activity | Disagreement | Consensus |", + "|-----------|----------|----------------|--------------|-----------|", + ] + for c in con.get("candidates", [])[:20]: + lines.append( + f"| {c['candidate_id']} | {c['activity_likeness']:.4f} | " + f"{c['boman_activity']:.4f} | {c.get('disagreement', 0.0):.4f} | " + f"{c['consensus_label']} |" + ) + + lines += [ + "", + "---", + "", + "## Next Steps (Human Review Required)", + "", + "Before any candidate proceeds to wet-lab assay:", + "", + "- [ ] Expert peptide chemist reviews synthesis feasibility for each candidate", + "- [ ] Microbiologist reviews candidate selection and target organisms", + "- [ ] Safety officer reviews toxicity risk flags", + "- [ ] PI or qualified reviewer approves batch release", + "- [ ] CRO or lab partner is selected and contacted", + "- [ ] Pre-registered pass/fail criteria locked (see `docs/SELECTION_RULE.md`)", + "", + "**No candidate may proceed without human expert sign-off.**", + "", + ] + + p.write_text("\n".join(lines), encoding="utf-8") diff --git a/src/openamp_foundry/scoring/activity.py b/src/openamp_foundry/scoring/activity.py index 9a407191..615e9215 100644 --- a/src/openamp_foundry/scoring/activity.py +++ b/src/openamp_foundry/scoring/activity.py @@ -6,16 +6,47 @@ def clamp01(x: float) -> float: def activity_likeness_score(features: dict) -> float: - """Transparent baseline, not a validated biological predictor.""" + """Transparent baseline activity-likeness score. + + This is NOT a validated biological predictor. It combines known physicochemical + correlates of AMP activity: length, cationic charge density, hydrophobic fraction, + aromatic content, and hydrophobic moment (amphipathicity). All weights and thresholds + are documented and fixed before any benchmark evaluation. + + Literature basis: + - Charge: Zasloff (2002) Nature; most AMPs carry net charge +2 to +9 + - Hydrophobicity: Hancock & Sahl (2006) Nat Biotechnol; ~30-50% hydrophobic + - Amphipathicity (μH): Eisenberg et al. (1984); helical amphipathicity correlates + with membrane disruption + - Length: Jenssen et al. (2006) Clin Microbiol Rev; typical 8-50 AA + """ length = features["length"] charge_density = features["charge_density"] hydrophobic = features["hydrophobic_fraction"] aromatic = features["aromatic_fraction"] + mu_h = features.get("hydrophobic_moment", 0.0) + # Length: peak at ~18 residues, broad tolerance length_score = 1.0 - min(abs(length - 18) / 25, 1.0) + + # Charge: AMPs typically have positive charge density 0.1–0.5 charge_score = clamp01((charge_density + 0.05) / 0.55) + + # Hydrophobicity: 40-50% is a sweet spot for membrane interaction hydro_score = 1.0 - min(abs(hydrophobic - 0.45) / 0.45, 1.0) + + # Aromatic residues (F, W, Y) aid membrane insertion aromatic_bonus = min(aromatic / 0.20, 1.0) * 0.10 - score = 0.30 * length_score + 0.35 * charge_score + 0.25 * hydro_score + aromatic_bonus + # Amphipathicity: helical hydrophobic moment > 0.4 is associated with activity + # Typical range for AMPs: 0.3–0.8; scale to [0,1] over 0–0.8 range + amphipathicity_score = clamp01(mu_h / 0.8) * 0.15 + + score = ( + 0.28 * length_score + + 0.32 * charge_score + + 0.20 * hydro_score + + aromatic_bonus + + amphipathicity_score + ) return round(clamp01(score), 4) diff --git a/src/openamp_foundry/scoring/boman.py b/src/openamp_foundry/scoring/boman.py new file mode 100644 index 00000000..05562020 --- /dev/null +++ b/src/openamp_foundry/scoring/boman.py @@ -0,0 +1,128 @@ +"""Boman index scorer — independent second activity predictor. + +Implements the Boman index as defined in: + Boman HG (2003). Antibacterial and antimalarial properties of peptides that + are cecropin-melittin hybrids. FEBS Letters, 544(1-3), 212-215. + and + Boman HG (2003). Peptide antibiotics and their role in innate immunity. + Annual Review of Immunology, 13, 61-92. + +The Boman index is the mean of amino acid interaction potentials per residue. +Positive values (> 1.0) indicate high protein-binding / membrane-interaction +potential, which correlates with AMP-like activity in published benchmarks. + +This is a second INDEPENDENT scoring approach that complements the +physicochemical heuristic in activity.py. Disagreement between the two +scorers is an uncertainty signal: candidates where both scorers agree are +more robust nominations than those where only one scorer favors them. + +Reference values: Boman 2003, Table 1, set 2. +All values are from the published table and are reproduced here for +transparent, reproducible scoring. No training was done. +""" +from __future__ import annotations + +# Published amino acid interaction potentials from Boman (2003), Table 1, set 2. +# Units: kcal/mol (approximate interaction energies) +# Positive → hydrophilic/charged, negative → hydrophobic +_BOMAN_POTENTIALS: dict[str, float] = { + "A": -0.495, # alanine: small, hydrophobic + "R": 2.465, # arginine: positively charged, high interaction + "N": 0.185, # asparagine: polar, uncharged + "D": 2.465, # aspartate: negatively charged, high interaction + "C": -1.000, # cysteine: hydrophobic/special + "Q": 0.185, # glutamine: polar, uncharged + "E": 2.465, # glutamate: negatively charged, high interaction + "G": 0.000, # glycine: minimal side chain + "H": -0.505, # histidine: aromatic/basic, lower than K/R + "I": -1.810, # isoleucine: hydrophobic + "L": -1.810, # leucine: hydrophobic + "K": 2.465, # lysine: positively charged, high interaction + "M": -1.305, # methionine: hydrophobic/sulfur + "F": -2.498, # phenylalanine: aromatic, hydrophobic + "P": 0.000, # proline: conformational constraint + "S": 0.255, # serine: polar, uncharged + "T": 0.370, # threonine: polar, uncharged + "W": -3.398, # tryptophan: large aromatic, most hydrophobic + "Y": -2.303, # tyrosine: aromatic, polar + "V": -1.505, # valine: hydrophobic +} + +# Published AMP threshold: Boman index > 1.0 is associated with AMP-like activity. +# Below -1.0 indicates strongly hydrophobic, non-interactive (unlikely AMP). +_AMP_THRESHOLD = 1.0 +_HYDROPHOBIC_THRESHOLD = -1.0 +# Range of Boman index across all possible sequences: +_BOMAN_MIN = -3.398 # all-W sequence (hypothetical) +_BOMAN_MAX = 2.465 # all-K/R/D/E sequence (hypothetical) + + +def boman_index(sequence: str) -> float: + """Compute the Boman index for a peptide sequence. + + Returns the mean interaction potential per residue (kcal/mol). + Positive (>1.0): high protein-binding / membrane-interaction potential. + Negative (<0): predominantly hydrophobic, less interactive. + + Non-canonical amino acids contribute 0.0 (neutral) to the mean. + """ + if not sequence: + return 0.0 + total = sum(_BOMAN_POTENTIALS.get(aa, 0.0) for aa in sequence.upper()) + return round(total / len(sequence), 4) + + +def boman_activity_score(sequence: str) -> float: + """Normalize the Boman index to [0, 1] for comparison with other scorers. + + Mapping: + Boman ≥ 1.0 → score approaching 1.0 (AMP-like zone) + Boman = 0.0 → score ≈ 0.5 + Boman ≤ -1.0 → score approaching 0.0 (non-interactive zone) + + The normalization uses a sigmoid-like mapping so that the published threshold + (Boman > 1.0) falls in the upper range of the score. + """ + bi = boman_index(sequence) + # Sigmoid-like normalization centered at 0, scaled so bi=1.0 → ≈ 0.73 + # Using tanh mapping: score = 0.5 * (1 + tanh(bi / 2)) + import math + score = 0.5 * (1.0 + math.tanh(bi / 2.0)) + return round(max(0.0, min(1.0, score)), 4) + + +def gravy_score(sequence: str) -> float: + """Compute the GRAVY (Grand Average of hYdropathicity) score. + + Uses the Kyte-Doolittle hydropathy scale (J Mol Biol 157:105-132, 1982). + Positive GRAVY → hydrophobic overall. + Negative GRAVY → hydrophilic overall. + AMPs typically fall between -0.5 and +0.5. + """ + _KD: dict[str, float] = { + "A": 1.8, "R": -4.5, "N": -3.5, "D": -3.5, "C": 2.5, + "Q": -3.5, "E": -3.5, "G": -0.4, "H": -3.2, "I": 4.5, + "L": 3.8, "K": -3.9, "M": 1.9, "F": 2.8, "P": -1.6, + "S": -0.8, "T": -0.7, "W": -0.9, "Y": -1.3, "V": 4.2, + } + if not sequence: + return 0.0 + total = sum(_KD.get(aa, 0.0) for aa in sequence.upper()) + return round(total / len(sequence), 4) + + +def model_disagreement(activity_likeness: float, boman_act: float) -> float: + """Compute normalized disagreement between two activity scores. + + Returns a value in [0, 1]: + 0.0 → both scorers agree perfectly + 1.0 → maximum disagreement (one says 0, other says 1) + + High disagreement indicates uncertainty — the candidate should receive + extra scrutiny before nomination. Low disagreement (high consensus) is + evidence of more robust computational support. + + Disagreement is NOT a biological safety or activity assessment. + It is an uncertainty proxy for the computational prediction only. + """ + return round(abs(activity_likeness - boman_act), 4) diff --git a/src/openamp_foundry/scoring/ensemble.py b/src/openamp_foundry/scoring/ensemble.py index d0b11317..d75c0d23 100644 --- a/src/openamp_foundry/scoring/ensemble.py +++ b/src/openamp_foundry/scoring/ensemble.py @@ -17,6 +17,10 @@ def selection_reasons(scores: dict[str, float]) -> list[str]: reasons.append("Likely feasible under simple synthesis constraints") if scores["novelty"] >= 0.25: reasons.append("Not an exact or near-exact duplicate of demo references") + boman_act = scores.get("boman_activity") + disagreement = scores.get("disagreement") + if boman_act is not None and boman_act >= 0.60 and (disagreement is None or disagreement <= 0.20): + reasons.append("Independent Boman index scorer agrees: high interaction potential (low disagreement)") if not reasons: reasons.append("Selected only by ensemble rank; requires skeptical review") return reasons @@ -33,4 +37,11 @@ def known_failure_modes(scores: dict[str, float]) -> list[str]: failures.append("Candidate has elevated pre-lab safety-risk proxy.") if scores["synthesis"] < 0.75: failures.append("Candidate may be harder to synthesize or handle.") + disagreement = scores.get("disagreement") + if disagreement is not None and disagreement >= 0.30: + failures.append( + f"High scorer disagreement ({disagreement:.2f}): " + "Boman index and physicochemical activity scores diverge. " + "Extra scrutiny recommended before nomination." + ) return failures diff --git a/tests/test_ablation.py b/tests/test_ablation.py new file mode 100644 index 00000000..335fb527 --- /dev/null +++ b/tests/test_ablation.py @@ -0,0 +1,170 @@ +"""Ablation tests: removing safety/novelty filters should degrade selection quality. + +Per AGENTS.md Phase 2: 'Ablation — Removing safety/novelty filters makes results +worse or riskier.' These tests verify that the filters are actually doing useful work. + +Results are computational only. No biological activity is implied. +""" +from __future__ import annotations + +import json + +from openamp_foundry.pipeline import run_ranking_pipeline, score_candidates +from openamp_foundry.selection.diversity import greedy_diverse_select +from openamp_foundry.selection.pareto import rank_candidates + + +MIXED = "examples/benchmark/mixed_candidates.csv" +ACTIVE_IDS = { + "BM-POS-001", + "BM-POS-002", + "BM-POS-003", + "BM-POS-004", + "BM-POS-005", +} + + +def _select_with_thresholds( + scored, + min_novelty: float = 0.20, + min_safety: float = 0.30, + top_n: int = 5, +) -> set[str]: + ranked = rank_candidates(scored) + eligible = [ + item for item in ranked + if item.scores["novelty"] >= min_novelty + and item.scores["safety"] >= min_safety + and item.valid + ] + selected = greedy_diverse_select(eligible, top_n=top_n) + return {s.candidate.candidate_id for s in selected} + + +class TestNoveltyFilterAblation: + def test_novelty_filter_excludes_reference_duplicates(self, tmp_path): + """With novelty filter: near-duplicates of references are excluded from selection.""" + candidates = tmp_path / "cands.csv" + refs = tmp_path / "refs.csv" + # Candidate identical to reference → novelty = 0.0 + candidates.write_text( + "id,sequence,source\n" + "GOOD-001,KWKLFKKIGAVLKVL,test\n" # identical to REF + "GOOD-002,RRWQWRMKKLG,test\n" # genuinely novel + ) + refs.write_text("id,sequence,source\nREF-001,KWKLFKKIGAVLKVL,reference\n") + + scored, _ = score_candidates(candidates, refs) + filtered = _select_with_thresholds(scored, min_novelty=0.20, top_n=5) + # With filter: near-duplicate should be excluded + assert "GOOD-001" not in filtered + assert "GOOD-002" in filtered + + def test_ablation_novelty_off_includes_near_duplicates(self, tmp_path): + """Without novelty filter: near-duplicates are selected (worse).""" + candidates = tmp_path / "cands.csv" + refs = tmp_path / "refs.csv" + candidates.write_text( + "id,sequence,source\n" + "GOOD-001,KWKLFKKIGAVLKVL,test\n" + "GOOD-002,RRWQWRMKKLG,test\n" + ) + refs.write_text("id,sequence,source\nREF-001,KWKLFKKIGAVLKVL,reference\n") + + scored, _ = score_candidates(candidates, refs) + unfiltered = _select_with_thresholds(scored, min_novelty=0.0, top_n=5) + # Without filter: near-duplicate may be selected + # GOOD-001 has high activity so it will rank first and get selected + assert "GOOD-001" in unfiltered + + +class TestSafetyFilterAblation: + def test_safety_filter_excludes_high_risk_sequences(self): + """Sequences with predicted safety risk below threshold should not be selected.""" + scored, _ = score_candidates(MIXED) + # Without safety filter, all valid sequences are eligible + no_filter = _select_with_thresholds(scored, min_safety=0.0, top_n=20) + # With safety filter, risky candidates dropped + with_filter = _select_with_thresholds(scored, min_safety=0.90, top_n=20) + # Filter should make the selected set a strict subset or equal + assert with_filter.issubset(no_filter) or with_filter == no_filter + + def test_safety_filter_at_high_threshold_removes_some(self): + """A stricter safety threshold should exclude at least one candidate.""" + scored, _ = score_candidates("examples/sequences/demo_candidates.csv") + lenient = _select_with_thresholds(scored, min_safety=0.0, top_n=10) + strict = _select_with_thresholds(scored, min_safety=0.80, top_n=10) + # Strict filter should exclude some candidates that lenient lets through + assert len(lenient) >= len(strict) + + def test_demo_negative_peptides_would_not_be_selected_with_filter(self): + """Known-bad demo negatives should not reach selection with any reasonable filter.""" + scored, _ = score_candidates("examples/negative/demo_negative_peptides.csv") + selected = _select_with_thresholds(scored, min_novelty=0.0, min_safety=0.60, top_n=10) + # DEDEDEDEDEDE and all-repeat sequences should score poorly enough to be excluded + # by safety threshold OR score so low they don't make top-n anyway + # At minimum, selection should not blindly include everything + ranked = rank_candidates(scored) + all_ids = {s.candidate.candidate_id for s in ranked} + assert selected.issubset(all_ids) + + +class TestBatchReportGeneration: + def test_batch_report_json_generated_alongside_md(self, tmp_path): + out = tmp_path / "ranked.jsonl" + report = tmp_path / "report.md" + run_ranking_pipeline( + candidate_path="examples/sequences/demo_candidates.csv", + reference_path="examples/known_reference/demo_known_amps.csv", + out_path=out, + report_path=report, + ) + report_json = tmp_path / "report.json" + assert report_json.exists(), "batch report JSON should be generated alongside .md" + data = json.loads(report_json.read_text()) + assert "pipeline_version" in data + assert "candidate_count" in data + assert "selected_count" in data + assert "generated_at" in data + assert "disclaimer" in data + + def test_batch_report_validates_against_schema(self, tmp_path): + out = tmp_path / "ranked.jsonl" + report = tmp_path / "report.md" + run_ranking_pipeline( + candidate_path="examples/sequences/demo_candidates.csv", + reference_path="examples/known_reference/demo_known_amps.csv", + out_path=out, + report_path=report, + ) + from openamp_foundry.evidence.schemas import validate_json_schema + data = json.loads((tmp_path / "report.json").read_text()) + validate_json_schema(data, "schemas/batch_report.schema.json") + + def test_batch_report_counts_match_reality(self, tmp_path): + out = tmp_path / "ranked.jsonl" + report = tmp_path / "report.md" + run_ranking_pipeline( + candidate_path="examples/sequences/demo_candidates.csv", + reference_path="examples/known_reference/demo_known_amps.csv", + out_path=out, + report_path=report, + ) + data = json.loads((tmp_path / "report.json").read_text()) + rows = [json.loads(line) for line in out.read_text().splitlines() if line.strip()] + assert data["candidate_count"] == len(rows) + selected_count = sum(1 for r in rows if r["selected"]) + assert data["selected_count"] == selected_count + + def test_batch_report_disclaimer_present(self, tmp_path): + out = tmp_path / "ranked.jsonl" + report = tmp_path / "report.md" + run_ranking_pipeline( + candidate_path="examples/sequences/demo_candidates.csv", + reference_path="examples/known_reference/demo_known_amps.csv", + out_path=out, + report_path=report, + ) + data = json.loads((tmp_path / "report.json").read_text()) + assert "disclaimer" in data + assert "heuristic" in data["disclaimer"].lower() or "not validated" in data["disclaimer"].lower() diff --git a/tests/test_batch_pack.py b/tests/test_batch_pack.py new file mode 100644 index 00000000..9dd21ab2 --- /dev/null +++ b/tests/test_batch_pack.py @@ -0,0 +1,384 @@ +"""Tests for batch_pack.py — Phase 3 batch pack report generator. + +Verifies that all four required sub-reports are produced correctly from +a ranked JSONL file: + 1. Diversity clustering + 2. Novelty + 3. Toxicity/hemolysis risk + 4. Synthesis feasibility +""" +from __future__ import annotations + +import json + +import pytest + +from openamp_foundry.reports.batch_pack import ( + diversity_clustering_report, + generate_batch_pack, + novelty_report, + scorer_consensus_report, + synthesis_feasibility_report, + toxicity_hemolysis_risk_report, + write_batch_pack_markdown, +) + + +def _make_candidate( + candidate_id: str, + sequence: str, + selected: bool = True, + activity: float = 0.7, + safety: float = 0.9, + synthesis: float = 0.8, + novelty: float = 0.2, + boman_activity: float = 0.65, + disagreement: float | None = None, + nearest_reference: str | None = "SEED-001", +) -> dict: + features = { + "length": len(sequence), + "hydrophobic_fraction": 0.4, + "charge_density": 0.3, + "cysteine_fraction": 0.0, + "proline_fraction": 0.0, + "longest_repeat_run": 1, + "net_charge_proxy": 3, + "aromatic_fraction": 0.1, + "hydrophobic_moment": 0.3, + } + if disagreement is None: + disagreement = round(abs(activity - boman_activity), 4) + ensemble = 0.35 * activity + 0.30 * safety + 0.20 * synthesis + 0.15 * novelty + return { + "candidate_id": candidate_id, + "sequence": sequence, + "source": "test", + "selected": selected, + "scores": { + "activity": activity, + "safety": safety, + "synthesis": synthesis, + "novelty": novelty, + "ensemble": ensemble, + "boman_activity": boman_activity, + "disagreement": disagreement, + }, + "features": features, + "nearest_reference": nearest_reference, + "selection_reason": [], + "known_failure_modes": [], + } + + +SELECTED_CANDIDATES = [ + _make_candidate("CAND-001", "KWKLFKKIGAVLKVL", novelty=0.13), + _make_candidate("CAND-002", "RRWQWRMKKLG", novelty=0.18), + _make_candidate("CAND-003", "FLPLIGRVLSGIL", novelty=0.15), + _make_candidate("CAND-004", "GIGKFLHSAKKFGK", novelty=0.20), + _make_candidate("CAND-005", "KRLFKKIGSALKFL", novelty=0.09), +] + +NOT_SELECTED = [ + _make_candidate("CAND-006", "KKKKKKKKKK", selected=False, safety=0.2), + _make_candidate("CAND-007", "LLLLLLLLLLL", selected=False, safety=0.3), +] + +ALL_ROWS = SELECTED_CANDIDATES + NOT_SELECTED + + +@pytest.fixture +def ranked_jsonl(tmp_path): + path = tmp_path / "ranked.jsonl" + with open(path, "w") as f: + for row in ALL_ROWS: + f.write(json.dumps(row) + "\n") + return path + + +class TestDiversityClusteringReport: + def test_report_type_field(self): + report = diversity_clustering_report(SELECTED_CANDIDATES) + assert report["report_type"] == "diversity_clustering" + + def test_n_selected_matches_input(self): + report = diversity_clustering_report(SELECTED_CANDIDATES) + assert report["n_selected"] == len(SELECTED_CANDIDATES) + + def test_clusters_are_non_empty(self): + report = diversity_clustering_report(SELECTED_CANDIDATES) + assert report["n_clusters"] >= 1 + + def test_all_members_accounted_for(self): + report = diversity_clustering_report(SELECTED_CANDIDATES) + all_members = [m for c in report["clusters"] for m in c["members"]] + assert len(all_members) == len(SELECTED_CANDIDATES) + + def test_singleton_fraction_between_0_and_1(self): + report = diversity_clustering_report(SELECTED_CANDIDATES) + assert 0.0 <= report["singleton_fraction"] <= 1.0 + + def test_identical_sequences_cluster_together(self): + duped = [ + _make_candidate("A", "KWKLFKKIGAVLKVL"), + _make_candidate("B", "KWKLFKKIGAVLKVL"), + ] + report = diversity_clustering_report(duped, threshold=0.80) + assert report["n_clusters"] == 1 + assert report["clusters"][0]["size"] == 2 + + def test_very_different_sequences_are_singletons(self): + diverse = [ + _make_candidate("A", "KWKLFKKIGAVLKVL"), + _make_candidate("B", "GGGGGGGGGGGGGGGG"), + ] + report = diversity_clustering_report(diverse, threshold=0.80) + assert report["n_clusters"] == 2 + assert report["n_singleton_clusters"] == 2 + + def test_has_disclaimer(self): + report = diversity_clustering_report(SELECTED_CANDIDATES) + assert "disclaimer" in report + assert len(report["disclaimer"]) > 10 + + +class TestNoveltyReport: + def test_report_type_field(self): + report = novelty_report(SELECTED_CANDIDATES) + assert report["report_type"] == "novelty_report" + + def test_n_selected_correct(self): + report = novelty_report(SELECTED_CANDIDATES) + assert report["n_selected"] == len(SELECTED_CANDIDATES) + + def test_mean_novelty_in_range(self): + report = novelty_report(SELECTED_CANDIDATES) + assert 0.0 <= report["mean_novelty"] <= 1.0 + + def test_min_le_mean_le_max(self): + report = novelty_report(SELECTED_CANDIDATES) + assert report["min_novelty"] <= report["mean_novelty"] <= report["max_novelty"] + + def test_candidate_count_matches(self): + report = novelty_report(SELECTED_CANDIDATES) + assert len(report["candidates"]) == len(SELECTED_CANDIDATES) + + def test_candidates_sorted_by_novelty_descending(self): + report = novelty_report(SELECTED_CANDIDATES) + scores = [c["novelty"] for c in report["candidates"]] + assert scores == sorted(scores, reverse=True) + + def test_low_novelty_counted_correctly(self): + # CAND-005 has novelty=0.09 which is < 0.10 + report = novelty_report(SELECTED_CANDIDATES) + assert report["n_low_novelty_lt_0_10"] >= 1 + + +class TestToxicityHemolysisRiskReport: + def test_report_type_field(self): + report = toxicity_hemolysis_risk_report(SELECTED_CANDIDATES) + assert report["report_type"] == "toxicity_hemolysis_risk" + + def test_n_selected_correct(self): + report = toxicity_hemolysis_risk_report(SELECTED_CANDIDATES) + assert report["n_selected"] == len(SELECTED_CANDIDATES) + + def test_mean_safety_in_range(self): + report = toxicity_hemolysis_risk_report(SELECTED_CANDIDATES) + assert 0.0 <= report["mean_safety_score"] <= 1.0 + + def test_candidate_count_matches(self): + report = toxicity_hemolysis_risk_report(SELECTED_CANDIDATES) + assert len(report["candidates"]) == len(SELECTED_CANDIDATES) + + def test_candidates_sorted_by_safety_ascending(self): + report = toxicity_hemolysis_risk_report(SELECTED_CANDIDATES) + scores = [c["safety_score"] for c in report["candidates"]] + assert scores == sorted(scores) + + def test_high_hydrophobic_flagged(self): + risky = [_make_candidate("RISK", "LLLLLLLLLLL")] + risky[0]["features"]["hydrophobic_fraction"] = 0.90 + report = toxicity_hemolysis_risk_report(risky) + flags = report["candidates"][0]["risk_flags"] + assert "high_hydrophobic_fraction" in flags + + def test_balanced_amp_has_no_risk_flags(self): + balanced = [_make_candidate("BALANCED", "KWKLFKKIGAVLKVL")] + report = toxicity_hemolysis_risk_report(balanced) + assert report["candidates"][0]["risk_flags"] == [] + + def test_risk_thresholds_present(self): + report = toxicity_hemolysis_risk_report(SELECTED_CANDIDATES) + assert "risk_thresholds" in report + assert "hydrophobic_fraction_high" in report["risk_thresholds"] + + +class TestSynthesisFeasibilityReport: + def test_report_type_field(self): + report = synthesis_feasibility_report(SELECTED_CANDIDATES) + assert report["report_type"] == "synthesis_feasibility" + + def test_n_selected_correct(self): + report = synthesis_feasibility_report(SELECTED_CANDIDATES) + assert report["n_selected"] == len(SELECTED_CANDIDATES) + + def test_mean_synthesis_in_range(self): + report = synthesis_feasibility_report(SELECTED_CANDIDATES) + assert 0.0 <= report["mean_synthesis_score"] <= 1.0 + + def test_counts_sum_to_n_selected(self): + report = synthesis_feasibility_report(SELECTED_CANDIDATES) + total = ( + report["n_high_feasibility_ge_0_80"] + + report["n_mid_feasibility_0_50_to_0_80"] + + report["n_low_feasibility_lt_0_50"] + ) + assert total == len(SELECTED_CANDIDATES) + + def test_candidates_sorted_by_synthesis_ascending(self): + report = synthesis_feasibility_report(SELECTED_CANDIDATES) + scores = [c["synthesis_score"] for c in report["candidates"]] + assert scores == sorted(scores) + + def test_sequence_and_length_present(self): + report = synthesis_feasibility_report(SELECTED_CANDIDATES) + for c in report["candidates"]: + assert "sequence" in c + assert "length" in c + + +class TestScorerConsensusReport: + def test_report_type_field(self): + report = scorer_consensus_report(SELECTED_CANDIDATES) + assert report["report_type"] == "scorer_consensus" + + def test_n_selected_correct(self): + report = scorer_consensus_report(SELECTED_CANDIDATES) + assert report["n_selected"] == len(SELECTED_CANDIDATES) + + def test_all_candidates_included_when_boman_present(self): + report = scorer_consensus_report(SELECTED_CANDIDATES) + assert report["n_scored_with_boman"] == len(SELECTED_CANDIDATES) + + def test_empty_candidates_handled(self): + report = scorer_consensus_report([]) + assert report["n_selected"] == 0 + assert report["n_scored_with_boman"] == 0 + + def test_high_consensus_counted(self): + # activity=0.7, boman_activity=0.7 → disagreement=0.0 → high_consensus + candidates = [_make_candidate("C1", "KWKLFKKIGAVLKVL", activity=0.7, boman_activity=0.7)] + report = scorer_consensus_report(candidates) + assert report["n_high_consensus_lt_0_20"] == 1 + assert report["n_uncertain_ge_0_30"] == 0 + + def test_uncertain_counted(self): + # activity=0.9, boman_activity=0.4 → disagreement=0.5 → uncertain + candidates = [_make_candidate("C1", "KWKLFKKIGAVLKVL", activity=0.9, boman_activity=0.4)] + report = scorer_consensus_report(candidates) + assert report["n_uncertain_ge_0_30"] == 1 + assert report["n_high_consensus_lt_0_20"] == 0 + + def test_consensus_labels_correct(self): + candidates = [ + _make_candidate("C1", "KWKLFKKIGAVLKVL", activity=0.7, boman_activity=0.7), + _make_candidate("C2", "RRWQWRMKKLG", activity=0.9, boman_activity=0.4), + ] + report = scorer_consensus_report(candidates) + by_id = {c["candidate_id"]: c["consensus_label"] for c in report["candidates"]} + assert by_id["C1"] == "high_consensus" + assert by_id["C2"] == "uncertain" + + def test_sorted_by_disagreement_ascending(self): + report = scorer_consensus_report(SELECTED_CANDIDATES) + diffs = [c["disagreement"] for c in report["candidates"]] + assert diffs == sorted(diffs) + + def test_mean_disagreement_in_range(self): + report = scorer_consensus_report(SELECTED_CANDIDATES) + assert 0.0 <= report["mean_disagreement"] <= 1.0 + + def test_has_disclaimer(self): + report = scorer_consensus_report(SELECTED_CANDIDATES) + assert "disclaimer" in report + assert "lab" in report["disclaimer"].lower() + + def test_no_boman_scores_gracefully_handled(self): + # Candidates without boman_activity should be skipped + candidates = [ + { + "candidate_id": "NOBOMAN", + "sequence": "KWKLFKK", + "scores": {"activity": 0.7, "safety": 0.9, "synthesis": 0.8, "novelty": 0.2, "ensemble": 0.75}, + "features": {}, + "selected": True, + } + ] + report = scorer_consensus_report(candidates) + assert report["n_scored_with_boman"] == 0 + + +class TestGenerateBatchPack: + def test_all_required_top_level_keys_present(self, ranked_jsonl): + pack = generate_batch_pack(ranked_jsonl) + assert "summary" in pack + assert "diversity_clustering" in pack + assert "novelty_report" in pack + assert "toxicity_hemolysis_risk" in pack + assert "synthesis_feasibility" in pack + assert "scorer_consensus" in pack + assert "disclaimer" in pack + + def test_summary_n_selected_matches_selected_rows(self, ranked_jsonl): + pack = generate_batch_pack(ranked_jsonl) + assert pack["summary"]["n_candidates_selected"] == len(SELECTED_CANDIDATES) + + def test_summary_n_scored_matches_total_rows(self, ranked_jsonl): + pack = generate_batch_pack(ranked_jsonl) + assert pack["summary"]["n_candidates_scored"] == len(ALL_ROWS) + + def test_non_selected_excluded_from_sub_reports(self, ranked_jsonl): + pack = generate_batch_pack(ranked_jsonl) + all_ids_in_novelty = {c["candidate_id"] for c in pack["novelty_report"]["candidates"]} + assert "CAND-006" not in all_ids_in_novelty + assert "CAND-007" not in all_ids_in_novelty + + def test_selected_ids_present_in_sub_reports(self, ranked_jsonl): + pack = generate_batch_pack(ranked_jsonl) + novelty_ids = {c["candidate_id"] for c in pack["novelty_report"]["candidates"]} + assert "CAND-001" in novelty_ids + assert "CAND-002" in novelty_ids + + +class TestWriteBatchPackMarkdown: + def test_markdown_file_created(self, ranked_jsonl, tmp_path): + pack = generate_batch_pack(ranked_jsonl) + md_path = tmp_path / "batch_pack.md" + write_batch_pack_markdown(pack, md_path) + assert md_path.exists() + + def test_markdown_contains_required_sections(self, ranked_jsonl, tmp_path): + pack = generate_batch_pack(ranked_jsonl) + md_path = tmp_path / "batch_pack.md" + write_batch_pack_markdown(pack, md_path) + content = md_path.read_text() + assert "Diversity Clustering" in content + assert "Novelty Report" in content + assert "Toxicity" in content + assert "Synthesis Feasibility" in content + assert "Scorer Consensus" in content + + def test_markdown_contains_disclaimer(self, ranked_jsonl, tmp_path): + pack = generate_batch_pack(ranked_jsonl) + md_path = tmp_path / "batch_pack.md" + write_batch_pack_markdown(pack, md_path) + content = md_path.read_text() + assert "heuristic" in content.lower() or "disclaimer" in content.lower() + + def test_markdown_requires_human_review(self, ranked_jsonl, tmp_path): + pack = generate_batch_pack(ranked_jsonl) + md_path = tmp_path / "batch_pack.md" + write_batch_pack_markdown(pack, md_path) + content = md_path.read_text() + assert "Human Review" in content or "human expert" in content.lower() diff --git a/tests/test_benchmark_evaluate.py b/tests/test_benchmark_evaluate.py new file mode 100644 index 00000000..3e6c7a5a --- /dev/null +++ b/tests/test_benchmark_evaluate.py @@ -0,0 +1,121 @@ +"""Tests for benchmark evaluation: recall@k, enrichment factor, and summary.""" +from __future__ import annotations + + +from openamp_foundry.benchmark.evaluate import ( + benchmark_summary, + enrichment_factor, + random_recall_at_k, + recall_at_k, +) +from openamp_foundry.features.physchem import compute_features +from openamp_foundry.types import PeptideCandidate, ScoredCandidate + + +def _make_scored(cid: str, seq: str, ensemble: float) -> ScoredCandidate: + features = compute_features(seq) + scores = { + "activity": ensemble, + "safety": 0.8, + "synthesis": 0.9, + "novelty": 0.5, + "ensemble": ensemble, + } + return ScoredCandidate( + candidate=PeptideCandidate(candidate_id=cid, sequence=seq, source="test"), + features=features, + scores=scores, + ) + + +ITEMS = [ + _make_scored("C1", "KWKLFKKIGAVLKVL", 0.9), + _make_scored("C2", "GIGKFLHSAKKFG", 0.7), + _make_scored("C3", "AAAAAAAA", 0.3), + _make_scored("C4", "GLFDIVKK", 0.6), + _make_scored("C5", "DEDEDEDE", 0.1), +] +POSITIVES = {"C1", "C2"} + + +class TestRecallAtK: + def test_perfect_recall_when_all_positives_in_top_k(self): + assert recall_at_k(ITEMS, POSITIVES, k=2) == 1.0 + + def test_partial_recall(self): + assert recall_at_k(ITEMS, POSITIVES, k=1) == 0.5 + + def test_zero_recall_when_no_positives_in_top_k(self): + # Only C5 (worst score) is "positive" here + assert recall_at_k(ITEMS, {"C5"}, k=1) == 0.0 + + def test_full_recall_at_all(self): + assert recall_at_k(ITEMS, POSITIVES, k=len(ITEMS)) == 1.0 + + def test_empty_positives_returns_zero(self): + assert recall_at_k(ITEMS, set(), k=3) == 0.0 + + +class TestRandomRecallAtK: + def test_expected_random_recall(self): + # 2 positives in 5 candidates, k=2 → expected 0.4 hits → 0.4/2 = 0.2... + # Actually E[hits] = k * n_pos / n = 2 * 2 / 5 = 0.8 → recall = 0.8/2 = 0.4 + result = random_recall_at_k(n_candidates=5, n_positives=2, k=2) + assert 0 < result < 1 + + def test_zero_candidates_returns_zero(self): + assert random_recall_at_k(0, 2, 2) == 0.0 + + def test_zero_positives_returns_zero(self): + assert random_recall_at_k(5, 0, 2) == 0.0 + + def test_k_equals_total_returns_one(self): + result = random_recall_at_k(n_candidates=5, n_positives=2, k=5) + assert result == 1.0 + + +class TestEnrichmentFactor: + def test_ef_greater_than_one_for_good_ranker(self): + # Top-2 by ensemble score are C1 and C2, which are our positives + ef = enrichment_factor(ITEMS, POSITIVES, k=2) + assert ef > 1.0 + + def test_ef_approximately_one_for_random(self): + # With k = all items, every ranker has the same recall + ef = enrichment_factor(ITEMS, POSITIVES, k=len(ITEMS)) + assert abs(ef - 1.0) < 0.01 + + +class TestBenchmarkSummary: + def test_summary_has_required_keys(self): + result = benchmark_summary(ITEMS, POSITIVES, ks=[1, 2, 5]) + assert "disclaimer" in result + assert "n_candidates" in result + assert "n_positives" in result + assert "results" in result + assert "verdict" in result + + def test_summary_contains_disclaimer(self): + result = benchmark_summary(ITEMS, POSITIVES, ks=[2]) + assert "do not prove biological efficacy" in result["disclaimer"].lower() or \ + "do not prove" in result["disclaimer"].lower() + + def test_verdict_positive_when_pipeline_outperforms(self): + # C1 and C2 are top-ranked and are our positives — should outperform random at k=2 + result = benchmark_summary(ITEMS, POSITIVES, ks=[2]) + assert result["verdict"] == "pipeline outperforms random" + + def test_results_per_k(self): + result = benchmark_summary(ITEMS, POSITIVES, ks=[1, 3]) + assert len(result["results"]) == 2 + assert result["results"][0]["k"] == 1 + assert result["results"][1]["k"] == 3 + + def test_auto_ks_when_none_given(self): + result = benchmark_summary(ITEMS, POSITIVES) + assert len(result["results"]) > 0 + + def test_counts_are_correct(self): + result = benchmark_summary(ITEMS, POSITIVES, ks=[2]) + assert result["n_candidates"] == 5 + assert result["n_positives"] == 2 diff --git a/tests/test_boman_scorer.py b/tests/test_boman_scorer.py new file mode 100644 index 00000000..021fd58e --- /dev/null +++ b/tests/test_boman_scorer.py @@ -0,0 +1,191 @@ +"""Tests for scoring/boman.py — Boman index and GRAVY score. + +All expected values are hand-calculated from the published Boman (2003) +potentials and Kyte-Doolittle (1982) scale to verify the implementation +is accurate and not just self-consistent. +""" +from __future__ import annotations + +import pytest + +from openamp_foundry.scoring.boman import ( + boman_activity_score, + boman_index, + gravy_score, + model_disagreement, +) + + +class TestBomanIndex: + def test_empty_sequence_returns_zero(self): + assert boman_index("") == 0.0 + + def test_single_lysine(self): + # K → 2.465 + assert boman_index("K") == 2.465 + + def test_single_tryptophan(self): + # W → -3.398 + assert boman_index("W") == -3.398 + + def test_single_glycine(self): + # G → 0.000 + assert boman_index("G") == 0.0 + + def test_two_residues_mean(self): + # K + L → (2.465 + -1.810) / 2 = 0.3275 + assert boman_index("KL") == pytest.approx(0.3275, abs=1e-3) + + def test_all_cationic_positive(self): + # K, R, D, E → 2.465 each → mean 2.465 + assert boman_index("KR") == pytest.approx(2.465, abs=1e-3) + + def test_all_hydrophobic_negative(self): + # L, I → -1.810 each → mean -1.810 + assert boman_index("LI") == pytest.approx(-1.810, abs=1e-3) + + def test_case_insensitive(self): + assert boman_index("kwk") == boman_index("KWK") + + def test_cationic_higher_than_hydrophobic_control(self): + # AMP template (4 K residues) should score higher than all-hydrophobic decoy + amp_bi = boman_index("KWKLFKKIGAVLKVL") + decoy_bi = boman_index("LLLLLLLLLLLLLLL") # all leucine → -1.810 + assert amp_bi > decoy_bi, f"AMP template ({amp_bi}) should exceed all-hydrophobic control ({decoy_bi})" + + def test_polyK_is_max_positive(self): + bi_kk = boman_index("KKKK") + bi_ww = boman_index("WWWW") + assert bi_kk > bi_ww + + def test_rounding_to_4_decimal_places(self): + result = boman_index("KL") + assert result == round(result, 4) + + def test_unknown_aa_contributes_zero(self): + # Non-canonical AAs are treated as zero contribution (neutral) + bi_x = boman_index("X") + assert bi_x == pytest.approx(0.0, abs=1e-4) + + def test_mixed_canonical_and_unknown(self): + # "KB" → K(2.465) + B(0.0) → mean 1.2325 + assert boman_index("KB") == pytest.approx(1.2325, abs=1e-4) + + +class TestBomanActivityScore: + def test_returns_between_zero_and_one(self): + for seq in ["K", "W", "KWKLFKKIGAVLKVL", "LLLLLLL", "KKKKKKKK"]: + score = boman_activity_score(seq) + assert 0.0 <= score <= 1.0, f"Score out of range for {seq}: {score}" + + def test_empty_sequence(self): + score = boman_activity_score("") + assert score == pytest.approx(0.5, abs=1e-3) + + def test_cationic_seq_scores_higher_than_hydrophobic(self): + # KKKKK >> WWWWW in Boman → higher activity score + assert boman_activity_score("KKKKK") > boman_activity_score("WWWWW") + + def test_zero_boman_maps_to_near_half(self): + # G → 0.000 → tanh(0) = 0 → 0.5 + score = boman_activity_score("GGGG") + assert score == pytest.approx(0.5, abs=1e-2) + + def test_monotone_with_boman_index(self): + seqs = ["WWWW", "LLLL", "GGGG", "SSSS", "KKKK"] + indices = [boman_index(s) for s in seqs] + activities = [boman_activity_score(s) for s in seqs] + assert sorted(indices) == sorted(indices), "Sanity: indices sortable" + sorted_pairs = sorted(zip(indices, activities)) + for (_, a1), (_, a2) in zip(sorted_pairs, sorted_pairs[1:]): + assert a1 <= a2 + 1e-6, "boman_activity_score should increase with boman_index" + + def test_rounding(self): + score = boman_activity_score("KWK") + assert score == round(score, 4) + + +class TestGravyScore: + def test_empty_returns_zero(self): + assert gravy_score("") == 0.0 + + def test_single_isoleucine(self): + # I → 4.5 + assert gravy_score("I") == pytest.approx(4.5, abs=1e-3) + + def test_single_arginine(self): + # R → -4.5 + assert gravy_score("R") == pytest.approx(-4.5, abs=1e-3) + + def test_hydrophobic_sequence_positive(self): + # LLLLL → 3.8 mean + assert gravy_score("LLLLL") == pytest.approx(3.8, abs=1e-3) + + def test_cationic_sequence_negative(self): + # KKKKK → -3.9 mean + assert gravy_score("KKKKK") == pytest.approx(-3.9, abs=1e-3) + + def test_case_insensitive(self): + assert gravy_score("kwk") == gravy_score("KWK") + + def test_mixed_near_zero(self): + # G → -0.4, balanced hydrophobic + hydrophilic should be near-zero + gravy = gravy_score("KLII") + # K=-3.9, L=3.8, I=4.5, I=4.5 → mean = (−3.9+3.8+4.5+4.5)/4 = 2.225 + assert gravy == pytest.approx(2.225, abs=1e-3) + + def test_rounding_to_4_places(self): + result = gravy_score("KWL") + assert result == round(result, 4) + + +class TestModelDisagreement: + def test_identical_scores_zero_disagreement(self): + assert model_disagreement(0.7, 0.7) == 0.0 + + def test_maximum_disagreement(self): + assert model_disagreement(0.0, 1.0) == pytest.approx(1.0, abs=1e-4) + assert model_disagreement(1.0, 0.0) == pytest.approx(1.0, abs=1e-4) + + def test_symmetric(self): + assert model_disagreement(0.3, 0.7) == model_disagreement(0.7, 0.3) + + def test_small_difference(self): + assert model_disagreement(0.60, 0.65) == pytest.approx(0.05, abs=1e-4) + + def test_output_in_range(self): + for a, b in [(0.0, 0.5), (0.9, 0.4), (0.3, 0.3), (1.0, 0.0)]: + d = model_disagreement(a, b) + assert 0.0 <= d <= 1.0 + + def test_rounding_to_4_places(self): + result = model_disagreement(0.333, 0.667) + assert result == round(result, 4) + + +class TestPipelineIntegration: + """Verify Boman values appear in compute_features output.""" + + def test_boman_index_in_features(self): + from openamp_foundry.features.physchem import compute_features + features = compute_features("KWKLFKKIGAVLKVL") + assert "boman_index" in features + assert isinstance(features["boman_index"], float) + + def test_gravy_in_features(self): + from openamp_foundry.features.physchem import compute_features + features = compute_features("KWKLFKKIGAVLKVL") + assert "gravy" in features + assert isinstance(features["gravy"], float) + + def test_boman_index_feature_value_matches_scorer(self): + from openamp_foundry.features.physchem import compute_features + seq = "KRLFKKIGSALKFL" + features = compute_features(seq) + assert features["boman_index"] == pytest.approx(boman_index(seq), abs=1e-4) + + def test_gravy_feature_value_matches_scorer(self): + from openamp_foundry.features.physchem import compute_features + seq = "KRLFKKIGSALKFL" + features = compute_features(seq) + assert features["gravy"] == pytest.approx(gravy_score(seq), abs=1e-4) diff --git a/tests/test_cli.py b/tests/test_cli.py index 02e1f390..06a689f8 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -3,7 +3,6 @@ import json -import pytest from openamp_foundry.cli import main @@ -45,7 +44,6 @@ def test_validate_command_success(tmp_path): "--out", out, "--cert-dir", certs, ]) - import os cert_files = list((tmp_path / "certs").glob("*.json")) assert cert_files, "No certificates were generated" ret = main([ diff --git a/tests/test_cluster_split.py b/tests/test_cluster_split.py new file mode 100644 index 00000000..8b9eb26d --- /dev/null +++ b/tests/test_cluster_split.py @@ -0,0 +1,271 @@ +"""Cluster split validation tests — Phase 2 requirement. + +AGENTS.md: "Cluster split — Pipeline still performs when near-duplicates are removed." + +These tests verify that: +1. The cluster algorithm correctly groups near-duplicate sequences. +2. The cluster split partitions sequences so each cluster is not split across reference/test. +3. After removing near-duplicate references, the pipeline still enriches AMP-like sequences + over non-AMP negatives — confirming the scoring is feature-based, not reference-proximity-based. + +All results are computational scores only. No biological activity is implied. +""" +from __future__ import annotations + +import csv +from pathlib import Path + + +from openamp_foundry.benchmark.evaluate import ( + enrichment_factor, + find_contaminated_references, + recall_at_k, +) +from openamp_foundry.benchmark.splits import cluster_by_similarity, cluster_split +from openamp_foundry.data.loaders import load_candidates_csv +from openamp_foundry.pipeline import score_candidates +from openamp_foundry.scoring.novelty import normalized_similarity + + +POOL_CSV = "examples/benchmark/cluster_split_pool.csv" +REFS_CSV = "examples/benchmark/cluster_split_refs.csv" + +# CS-POS-001/002 are near-dups of CSREF-001 (KWKLFKKIGAVLKVL) +# CS-POS-003 is a near-dup of CSREF-003 (GLFDIVKKVVGALGSL) +POSITIVE_IDS = {"CS-POS-001", "CS-POS-002", "CS-POS-003"} + + +class TestClusterBySimilarity: + def test_identical_sequences_in_same_cluster(self): + seqs = ["KWKLFKKIGAVLKVL", "KWKLFKKIGAVLKVL", "AAAAAAAAAA"] + clusters = cluster_by_similarity(seqs, threshold=0.70) + assert len(clusters) == 2 + assert set(clusters[0]) == {0, 1} + + def test_near_duplicates_co_clustered(self): + # KWKLFKKIGAVLKVL vs KWKLFKKIGAVLKFL: levenshtein=1, len=15 → sim=14/15≈0.933 + seqs = ["KWKLFKKIGAVLKVL", "KWKLFKKIGAVLKFL", "AAAAAAAAAAAAA"] + clusters = cluster_by_similarity(seqs, threshold=0.70) + assert len(clusters) == 2 + assert set(clusters[0]) == {0, 1} + assert clusters[1] == [2] + + def test_dissimilar_sequences_in_separate_clusters(self): + seqs = ["KWKLFKKIGAVLKVL", "AAAAAAAAAAAA", "DEDEDEDEDEDE"] + clusters = cluster_by_similarity(seqs, threshold=0.70) + assert len(clusters) == 3 + assert all(len(c) == 1 for c in clusters) + + def test_single_sequence_forms_one_cluster(self): + clusters = cluster_by_similarity(["KWKLFKKIGAVLKVL"], threshold=0.70) + assert clusters == [[0]] + + def test_empty_input(self): + clusters = cluster_by_similarity([], threshold=0.70) + assert clusters == [] + + def test_threshold_controls_grouping(self): + seqs = ["KWKLFKKIGAVLKVL", "KWKLFKKIGAVLKFL"] + sim = normalized_similarity(seqs[0], seqs[1]) + # Should cluster at low threshold, not at very high threshold + low = cluster_by_similarity(seqs, threshold=sim - 0.01) + assert len(low) == 1 # grouped + high = cluster_by_similarity(seqs, threshold=sim + 0.01) + assert len(high) == 2 # split + + def test_all_index_covered_exactly_once(self): + seqs = ["KWKLFKKIGAVLKVL", "KWKLFKKIGAVLKFL", "RRWQWRMKKLG", "AAAAAAAAAAAAA"] + clusters = cluster_by_similarity(seqs, threshold=0.70) + all_indices = [i for c in clusters for i in c] + assert sorted(all_indices) == list(range(len(seqs))) + + +class TestClusterSplit: + def test_singleton_clusters_all_go_to_reference(self): + seqs = ["KWKLFKKIGAVLKVL", "AAAAAAAAAAAAA", "DEDEDEDEDEDE"] + ref_idx, test_idx = cluster_split(seqs, threshold=0.70) + assert set(ref_idx) == {0, 1, 2} + assert test_idx == [] + + def test_near_dup_goes_to_test(self): + seqs = ["KWKLFKKIGAVLKVL", "KWKLFKKIGAVLKFL", "AAAAAAAAAAAAA"] + ref_idx, test_idx = cluster_split(seqs, threshold=0.70) + assert 0 in ref_idx # cluster center → reference + assert 1 in test_idx # near-dup → test + assert 2 in ref_idx # unrelated → reference + + def test_reference_and_test_partition_all_sequences(self): + seqs = ["KWKLFKKIGAVLKVL", "KWKLFKKIGAVLKFL", "RRWQWRMKKLG", "AAAAAAAAAAAAA"] + ref_idx, test_idx = cluster_split(seqs, threshold=0.70) + all_covered = sorted(ref_idx + test_idx) + assert all_covered == list(range(len(seqs))) + + def test_no_test_sequence_is_near_dup_of_different_cluster_ref(self): + seqs = ["KWKLFKKIGAVLKVL", "KWKLFKKIGAVLKFL", "RRWQWRMKKLG"] + ref_idx, test_idx = cluster_split(seqs, threshold=0.70) + # test_idx should only contain the near-dup of cluster A (index 1) + # and NOT the unrelated cluster B center (index 2) + assert len(test_idx) == 1 + assert 1 in test_idx # only near-dup of cluster A goes to test + assert 2 not in test_idx # cluster B center stays in reference + + def test_multiple_near_dup_clusters(self): + # Two separate clusters with near-dups each + seqs = [ + "KWKLFKKIGAVLKVL", # cluster A center → ref + "KWKLFKKIGAVLKFL", # near-dup of A → test + "RRWQWRMKKLG", # cluster B center → ref + "RRWQWRMKKLF", # near-dup of B (M→F) → test + "AAAAAAAAAAAA", # singleton → ref + ] + ref_idx, test_idx = cluster_split(seqs, threshold=0.70) + assert sorted(ref_idx) == [0, 2, 4] + assert sorted(test_idx) == [1, 3] + + +class TestFindContaminatedReferences: + def test_identifies_near_dup_reference(self): + candidate_seqs = ["KWKLFKKIGAVLKFL", "AAAAAAAAAAAA"] + candidate_ids = ["CS-POS-001", "CS-NEG-001"] + ref_seqs = ["KWKLFKKIGAVLKVL", "DEDEDEDEDEDE"] # ref[0] near-dup of CS-POS-001 + positive_ids = {"CS-POS-001"} + + contaminated = find_contaminated_references( + candidate_seqs, ref_seqs, positive_ids, candidate_ids, threshold=0.70 + ) + assert 0 in contaminated # KWKLFKKIGAVLKVL is near-dup of CS-POS-001 + assert 1 not in contaminated # DEDEDEDEDEDE is not + + def test_non_similar_reference_not_contaminated(self): + candidate_seqs = ["KWKLFKKIGAVLKFL"] + candidate_ids = ["CS-POS-001"] + ref_seqs = ["DEDEDEDEDEDE", "GGGGGGGGGGGG"] + positive_ids = {"CS-POS-001"} + + contaminated = find_contaminated_references( + candidate_seqs, ref_seqs, positive_ids, candidate_ids, threshold=0.70 + ) + assert contaminated == set() + + def test_only_positive_near_dups_flagged(self): + # Negative candidate has near-dup in ref — should NOT be flagged + candidate_seqs = ["KWKLFKKIGAVLKFL", "DEDEDEDEDEDE"] + candidate_ids = ["CS-POS-001", "CS-NEG-001"] + ref_seqs = ["KWKLFKKIGAVLKVL", "DEDEDEDEDEDF"] # ref[1] near-dup of CS-NEG-001 + positive_ids = {"CS-POS-001"} + + contaminated = find_contaminated_references( + candidate_seqs, ref_seqs, positive_ids, candidate_ids, threshold=0.70 + ) + assert 0 in contaminated # ref[0] is near-dup of positive + assert 1 not in contaminated # ref[1] is near-dup of NEGATIVE — not flagged + + +class TestClusterSplitEnrichment: + """End-to-end tests verifying pipeline performance after cluster split.""" + + def test_positives_score_higher_than_negatives_without_references(self): + """AMP-like sequences should score above non-AMP negatives based on features alone.""" + scored, _ = score_candidates(POOL_CSV) # no reference → novelty=1.0 for all + pos_scores = [ + s.scores["activity"] + for s in scored + if s.candidate.candidate_id in POSITIVE_IDS + ] + neg_scores = [ + s.scores["activity"] + for s in scored + if s.candidate.candidate_id not in POSITIVE_IDS + ] + assert pos_scores, "No positive candidates scored" + assert neg_scores, "No negative candidates scored" + avg_pos = sum(pos_scores) / len(pos_scores) + avg_neg = sum(neg_scores) / len(neg_scores) + assert avg_pos > avg_neg, ( + f"AMP-like candidates (avg activity={avg_pos:.3f}) should score higher " + f"than non-AMP negatives (avg activity={avg_neg:.3f})" + ) + + def test_enrichment_factor_positive_after_cluster_split(self): + """EF > 1.0 after removing near-dup references (cluster split scenario).""" + scored_full, _ = score_candidates(POOL_CSV, REFS_CSV) + + # Identify contaminated references + candidate_seqs = [s.candidate.sequence for s in scored_full] + candidate_ids = [s.candidate.candidate_id for s in scored_full] + refs = load_candidates_csv(REFS_CSV) + ref_seqs = [r.sequence for r in refs] + + contaminated = find_contaminated_references( + candidate_seqs, ref_seqs, POSITIVE_IDS, candidate_ids, threshold=0.70 + ) + # At least one reference should be flagged as near-dup of the test positives + assert contaminated, "Expected at least one contaminated reference to be identified" + + # Score without near-dup references (cluster-split reference set) + scored_clean, _ = score_candidates(POOL_CSV) # no reference = clean split + + ef = enrichment_factor(scored_clean, POSITIVE_IDS, k=3) + assert ef > 1.0, ( + f"EF={ef:.3f} should be > 1.0 after cluster split " + "(pipeline should still enrich AMP-like sequences over negatives)" + ) + + def test_recall_at_k3_beats_random_after_split(self): + """recall@3 > random_recall@3 after removing near-dup references.""" + from openamp_foundry.benchmark.evaluate import random_recall_at_k + + scored, _ = score_candidates(POOL_CSV) # no references + n = len(scored) + n_pos = len(POSITIVE_IDS) + + rc = recall_at_k(scored, POSITIVE_IDS, k=3) + rrc = random_recall_at_k(n, n_pos, k=3) + assert rc >= rrc, ( + f"recall@3={rc:.4f} should be >= random baseline={rrc:.4f} " + "after cluster split" + ) + + def test_cluster_split_benchmark_data_integrity(self): + """Verify the cluster split pool has the expected IDs and structure.""" + pool = Path(POOL_CSV) + assert pool.exists(), "Cluster split pool CSV not found" + + with pool.open() as f: + reader = csv.DictReader(f) + rows = list(reader) + + ids = {r["id"] for r in rows} + assert POSITIVE_IDS.issubset(ids), "Expected positive IDs missing from pool" + neg_ids = ids - POSITIVE_IDS + assert len(neg_ids) >= 8, "Expected at least 8 negative controls in pool" + + def test_cluster_split_finds_near_dups_in_reference(self): + """Cluster split analysis correctly detects CSREF-001 as near-dup of CS-POS-001/002.""" + scored, _ = score_candidates(POOL_CSV, REFS_CSV) + candidate_seqs = [s.candidate.sequence for s in scored] + candidate_ids = [s.candidate.candidate_id for s in scored] + refs = load_candidates_csv(REFS_CSV) + ref_seqs = [r.sequence for r in refs] + + contaminated = find_contaminated_references( + candidate_seqs, ref_seqs, POSITIVE_IDS, candidate_ids, threshold=0.70 + ) + # CSREF-001 (KWKLFKKIGAVLKVL) should be flagged as near-dup of CS-POS-001/002 + csref001_idx = next(i for i, r in enumerate(refs) if r.candidate_id == "CSREF-001") + assert csref001_idx in contaminated + + def test_feature_scores_drive_ranking_not_reference_proximity(self): + """Verify the positives rank above negatives on feature scores alone (no references).""" + scored, _ = score_candidates(POOL_CSV) + + # Sort by ensemble score + ranked = sorted(scored, key=lambda s: s.scores["ensemble"], reverse=True) + top3_ids = {s.candidate.candidate_id for s in ranked[:3]} + + # At least 2 of top 3 should be known positives + overlap = len(top3_ids & POSITIVE_IDS) + assert overlap >= 2, ( + f"Expected ≥2 of top-3 to be AMP-like positives, got {overlap}. " + f"Top 3: {top3_ids}" + ) diff --git a/tests/test_hidden_active_recovery.py b/tests/test_hidden_active_recovery.py new file mode 100644 index 00000000..d340346b --- /dev/null +++ b/tests/test_hidden_active_recovery.py @@ -0,0 +1,143 @@ +"""Tests for hidden-active recovery benchmark. + +These tests verify that the pipeline recovers known-active AMPs better than +a random ranker when mixed with non-active sequences. This is a key Phase 2 +requirement per AGENTS.md. + +No biological activity is implied or proven by these results. +The benchmark only tests whether physicochemical features of known AMPs +correlate with higher ensemble scores than simple non-AMP sequences. +""" +from __future__ import annotations + +import json + + +from openamp_foundry.benchmark.evaluate import ( + benchmark_summary, + enrichment_factor, + recall_at_k, +) +from openamp_foundry.cli import main +from openamp_foundry.data.loaders import load_candidates_csv +from openamp_foundry.pipeline import score_candidates +from openamp_foundry.selection.pareto import rank_candidates + + +MIXED_CANDIDATES = "examples/benchmark/mixed_candidates.csv" +ACTIVE_LABELS = "examples/benchmark/active_labels.csv" + +POSITIVE_IDS = { + "BM-POS-001", + "BM-POS-002", + "BM-POS-003", + "BM-POS-004", + "BM-POS-005", +} + + +class TestHiddenActiveRecovery: + def test_all_positives_rank_in_top_half(self): + scored, _ = score_candidates(MIXED_CANDIDATES) + ranked = rank_candidates(scored) + n = len(ranked) + top_half_ids = {item.candidate.candidate_id for item in ranked[: n // 2]} + for pid in POSITIVE_IDS: + assert pid in top_half_ids, f"{pid} not in top half" + + def test_recall_at_5_equals_1(self): + """All 5 positives should appear in top-5 of 20 candidates.""" + scored, _ = score_candidates(MIXED_CANDIDATES) + rc = recall_at_k(scored, POSITIVE_IDS, k=5) + assert rc == 1.0, f"Expected recall@5=1.0, got {rc}" + + def test_enrichment_factor_at_5_exceeds_2(self): + """Pipeline should achieve at least 2× enrichment vs random at k=5.""" + scored, _ = score_candidates(MIXED_CANDIDATES) + ef = enrichment_factor(scored, POSITIVE_IDS, k=5) + assert ef >= 2.0, f"Enrichment factor {ef} below minimum of 2.0" + + def test_pipeline_verdict_outperforms_random(self): + scored, _ = score_candidates(MIXED_CANDIDATES) + summary = benchmark_summary(scored, POSITIVE_IDS, ks=[5]) + assert summary["verdict"] == "pipeline outperforms random" + + def test_negatives_score_lower_than_positives_on_average(self): + scored, _ = score_candidates(MIXED_CANDIDATES) + pos_scores = [ + s.scores["ensemble"] + for s in scored + if s.candidate.candidate_id in POSITIVE_IDS + ] + neg_scores = [ + s.scores["ensemble"] + for s in scored + if s.candidate.candidate_id not in POSITIVE_IDS + ] + assert pos_scores + assert neg_scores + assert sum(pos_scores) / len(pos_scores) > sum(neg_scores) / len(neg_scores) + + def test_bench_baseline_cli_shows_enrichment(self, tmp_path, capsys): + out = str(tmp_path / "bench_report.json") + ret = main([ + "bench", "baseline", + "--candidates", MIXED_CANDIDATES, + "--positives", ACTIVE_LABELS, + "--k", "5", + "--out", out, + ]) + assert ret == 0 + data = json.loads((tmp_path / "bench_report.json").read_text()) + k5_result = next(r for r in data["results"] if r["k"] == 5) + assert k5_result["enrichment_factor"] >= 2.0 + assert data["verdict"] == "pipeline outperforms random" + + def test_benchmark_summary_disclaimer_present(self): + scored, _ = score_candidates(MIXED_CANDIDATES) + summary = benchmark_summary(scored, POSITIVE_IDS, ks=[5]) + assert "disclaimer" in summary + assert len(summary["disclaimer"]) > 20 + + def test_known_active_amps_outrank_all_repeat_sequences(self): + """Biologically implausible repeat sequences should rank below AMP-like ones.""" + scored, _ = score_candidates(MIXED_CANDIDATES) + ranked = rank_candidates(scored) + ranked_ids = [s.candidate.candidate_id for s in ranked] + + # All positives should appear before all-same-AA negatives + for pos_id in POSITIVE_IDS: + pos_rank = ranked_ids.index(pos_id) + for neg_id in ["BM-NEG-001", "BM-NEG-002", "BM-NEG-004"]: + neg_rank = ranked_ids.index(neg_id) + assert pos_rank < neg_rank, ( + f"{pos_id} (rank {pos_rank}) should outrank {neg_id} (rank {neg_rank})" + ) + + +class TestBenchmarkDataIntegrity: + def test_benchmark_csvs_exist(self): + from pathlib import Path + assert Path(MIXED_CANDIDATES).exists() + assert Path(ACTIVE_LABELS).exists() + + def test_all_positive_ids_in_mixed_candidates(self): + mixed = load_candidates_csv(MIXED_CANDIDATES) + mixed_ids = {c.candidate_id for c in mixed} + for pid in POSITIVE_IDS: + assert pid in mixed_ids, f"Positive {pid} missing from mixed candidates" + + def test_mixed_has_both_positives_and_negatives(self): + mixed = load_candidates_csv(MIXED_CANDIDATES) + ids = {c.candidate_id for c in mixed} + positive_overlap = ids & POSITIVE_IDS + negative_only = ids - POSITIVE_IDS + assert len(positive_overlap) >= 3 + assert len(negative_only) >= 5 + + def test_active_labels_are_subset_of_mixed(self): + mixed = load_candidates_csv(MIXED_CANDIDATES) + active = load_candidates_csv(ACTIVE_LABELS) + mixed_ids = {c.candidate_id for c in mixed} + for a in active: + assert a.candidate_id in mixed_ids diff --git a/tests/test_lab_results.py b/tests/test_lab_results.py new file mode 100644 index 00000000..eb8f95d3 --- /dev/null +++ b/tests/test_lab_results.py @@ -0,0 +1,174 @@ +"""Tests for lab_results.py — active-learning loop ingestion. + +Verifies that: + - Valid lab result JSON loads and validates against schema + - Invalid lab results are rejected with useful errors + - summarise_lab_results produces correct aggregates + - candidate_result_map groups correctly +""" +from __future__ import annotations + +import json + +import pytest + +from openamp_foundry.data.lab_results import ( + candidate_result_map, + load_lab_result, + summarise_lab_results, +) + + +def _valid_result(**overrides) -> dict: + base = { + "result_id": "RES-001", + "candidate_id": "SEED-005_VAR_049", + "assay_type": "MIC", + "organism_or_cell_line": "E. coli ATCC 25922", + "result_value": 4.0, + "result_unit": "µg/mL", + "result_qualitative": "active", + "positive_control_passed": True, + "negative_control_passed": True, + "positive_control_id": "ciprofloxacin 0.25 µg/mL", + "negative_control_id": "PBS", + "assay_date": "2026-07-01", + "replicate_count": 3, + "performed_by_lab": "University Test Lab", + "raw_data_sha256": None, + "computational_candidate_certificate_hash": "abc123def456", + "notes": None, + "disclaimer": ( + "This is an experimental result on a computationally nominated candidate. " + "It does not constitute a drug or clinical claim." + ), + } + base.update(overrides) + return base + + +@pytest.fixture +def valid_result_file(tmp_path): + result = _valid_result() + path = tmp_path / "RES-001.json" + path.write_text(json.dumps(result)) + return path + + +@pytest.fixture +def results_dir(tmp_path): + for i in range(3): + result = _valid_result( + result_id=f"RES-{i:03d}", + candidate_id=f"CAND-{i:03d}", + assay_date=f"2026-07-0{i+1}", + ) + (tmp_path / f"RES-{i:03d}.json").write_text(json.dumps(result)) + return tmp_path + + +class TestLoadLabResult: + def test_valid_result_loads(self, valid_result_file): + result = load_lab_result(valid_result_file) + assert result["result_id"] == "RES-001" + assert result["candidate_id"] == "SEED-005_VAR_049" + assert result["assay_type"] == "MIC" + + def test_invalid_assay_type_rejected(self, tmp_path): + result = _valid_result(assay_type="unknown_assay") + path = tmp_path / "bad.json" + path.write_text(json.dumps(result)) + with pytest.raises(Exception): + load_lab_result(path) + + def test_missing_required_field_rejected(self, tmp_path): + result = _valid_result() + del result["candidate_id"] + path = tmp_path / "missing.json" + path.write_text(json.dumps(result)) + with pytest.raises(Exception): + load_lab_result(path) + + def test_negative_replicate_count_rejected(self, tmp_path): + result = _valid_result(replicate_count=0) + path = tmp_path / "bad_reps.json" + path.write_text(json.dumps(result)) + with pytest.raises(Exception): + load_lab_result(path) + + def test_null_result_value_allowed(self, tmp_path): + result = _valid_result(result_value=None, result_qualitative="inconclusive") + path = tmp_path / "null_val.json" + path.write_text(json.dumps(result)) + loaded = load_lab_result(path) + assert loaded["result_value"] is None + + def test_file_not_found_raises(self, tmp_path): + with pytest.raises(FileNotFoundError): + load_lab_result(tmp_path / "nonexistent.json") + + +class TestSummariseLabResults: + def test_empty_results_handled(self): + summary = summarise_lab_results([]) + assert summary["n_results"] == 0 + assert "disclaimer" in summary + + def test_n_results_correct(self): + results = [_valid_result(result_id=f"R{i}") for i in range(5)] + summary = summarise_lab_results(results) + assert summary["n_results"] == 5 + + def test_by_assay_type_counts(self): + results = [ + _valid_result(result_id="R1", assay_type="MIC"), + _valid_result(result_id="R2", assay_type="MIC"), + _valid_result(result_id="R3", assay_type="hemolysis_RBC"), + ] + summary = summarise_lab_results(results) + assert summary["by_assay_type"]["MIC"] == 2 + assert summary["by_assay_type"]["hemolysis_RBC"] == 1 + + def test_by_qualitative_counts(self): + results = [ + _valid_result(result_id="R1", result_qualitative="active"), + _valid_result(result_id="R2", result_qualitative="inactive"), + _valid_result(result_id="R3", result_qualitative="active"), + ] + summary = summarise_lab_results(results) + assert summary["by_qualitative_result"]["active"] == 2 + assert summary["by_qualitative_result"]["inactive"] == 1 + + def test_valid_controls_counted(self): + results = [ + _valid_result(result_id="R1", positive_control_passed=True, negative_control_passed=True), + _valid_result(result_id="R2", positive_control_passed=True, negative_control_passed=False), + ] + summary = summarise_lab_results(results) + assert summary["n_valid_controls"] == 1 + + def test_disclaimer_present(self): + results = [_valid_result()] + summary = summarise_lab_results(results) + assert "disclaimer" in summary + assert len(summary["disclaimer"]) > 20 + + +class TestCandidateResultMap: + def test_groups_by_candidate_id(self): + results = [ + _valid_result(result_id="R1", candidate_id="CAND-001"), + _valid_result(result_id="R2", candidate_id="CAND-001"), + _valid_result(result_id="R3", candidate_id="CAND-002"), + ] + mapping = candidate_result_map(results) + assert len(mapping["CAND-001"]) == 2 + assert len(mapping["CAND-002"]) == 1 + + def test_unknown_candidate_not_in_map(self): + results = [_valid_result(candidate_id="CAND-001")] + mapping = candidate_result_map(results) + assert "CAND-999" not in mapping + + def test_empty_results_gives_empty_map(self): + assert candidate_result_map([]) == {} diff --git a/tests/test_negative_penalization.py b/tests/test_negative_penalization.py new file mode 100644 index 00000000..2ab2a2ce --- /dev/null +++ b/tests/test_negative_penalization.py @@ -0,0 +1,131 @@ +"""Tests verifying that known problematic sequences are down-ranked. + +Per AGENTS.md Phase 2: 'Toxicity penalty — Predicted hemolytic/toxic candidates +are down-ranked.' These tests check that the safety and activity scores penalise +sequences with properties associated with mammalian toxicity risk: +- Extreme hydrophobicity (hemolysis risk) +- All-cysteine (aggregation, disulfide chaos) +- Purely negative charge (anti-AMP, repelled by bacterial membranes) +- Very long repeat runs (low complexity, synthesis issues) + +No claims of actual biological toxicity are made. These are heuristic proxy checks. +""" +from __future__ import annotations + + +from openamp_foundry.features.physchem import compute_features +from openamp_foundry.scoring.activity import activity_likeness_score +from openamp_foundry.scoring.safety import safety_score +from openamp_foundry.scoring.synthesis import synthesis_feasibility_score + + +def score_seq(seq: str) -> dict: + f = compute_features(seq) + return { + "features": f, + "activity": activity_likeness_score(f), + "safety": safety_score(f), + "synthesis": synthesis_feasibility_score(f, valid_sequence=True), + } + + +REFERENCE_AMP = "KWKLFKKIGAVLKVL" # Magainin analogue, classic AMP + + +class TestActivityPenalization: + def test_negative_charge_sequence_has_low_activity(self): + # Purely negative charge — repelled by bacterial membranes + result = score_seq("DEDEDEDEDEDE") + assert result["activity"] < score_seq(REFERENCE_AMP)["activity"] + + def test_low_charge_sequence_has_low_activity(self): + # No charge → low charge_density → low charge_score + result = score_seq("AAAAAAAAGGGGGGGG") + ref = score_seq(REFERENCE_AMP) + assert result["activity"] < ref["activity"] + + def test_reference_amp_scores_above_low_charge(self): + ref = score_seq(REFERENCE_AMP) + bad = score_seq("GGGGGGGGGGGG") + assert ref["activity"] > bad["activity"] + + +class TestSafetyPenalization: + def test_extreme_hydrophobicity_penalizes_safety(self): + # LLLLLLLLLLLL — very hydrophobic → hemolysis risk proxy + result = score_seq("LLLLLLLLLLLL") + assert result["safety"] < 1.0 + + def test_high_cysteine_penalizes_safety(self): + # All cys → risk of aggregation/disulfide chaos + result = score_seq("CCCCCCCCCCCC") + assert result["safety"] < 0.9 + + def test_very_long_repeat_run_penalizes_safety(self): + # 8+ same residue in a row triggers safety penalty + result = score_seq("KKKKKKKKLLLL") + assert result["safety"] < 1.0 + + def test_reference_amp_has_high_safety(self): + # Known AMP with moderate hydrophobicity should score well + result = score_seq(REFERENCE_AMP) + assert result["safety"] >= 0.8 + + def test_pure_negative_charge_has_lower_safety_than_amp(self): + neg = score_seq("DEDEDEDEDEDE") + ref = score_seq(REFERENCE_AMP) + # Safety differs because charge_density penalizes extremes + assert ref["safety"] >= neg["safety"] + + +class TestSynthesisPenalization: + def test_very_long_sequence_penalizes_synthesis(self): + long_seq = "KWKLFKKIGAVLKVL" * 3 # 45 residues + result = score_seq(long_seq) + assert result["synthesis"] < 1.0 + + def test_high_cysteine_penalizes_synthesis(self): + result = score_seq("CCCCCCCCCCCC") + assert result["synthesis"] < 0.9 + + def test_short_repeat_penalizes_synthesis(self): + result = score_seq("KKKKKKKKK") # long repeat run + assert result["synthesis"] < 1.0 + + def test_reference_amp_fully_feasible(self): + # 15-residue AMP with no special challenges should be fully feasible + result = score_seq(REFERENCE_AMP) + assert result["synthesis"] == 1.0 + + +class TestNegativesVsPositivesSummary: + """Aggregate check: negatives from demo should average below known AMPs.""" + + KNOWN_AMPS = [ + "KWKLFKKIGAVLKVL", + "GIGKFLHSAKKFGKAFVGEIMNS", + "GLFDIVKKVVGALGSL", + "RRWQWRMKKLG", + ] + KNOWN_NEGATIVES = [ + "AAAAAAAAAAAA", + "DEDEDEDEDEDE", + "GGGGGGGGGGGG", + ] + + def _avg_activity(self, seqs: list[str]) -> float: + scores = [score_seq(s)["activity"] for s in seqs] + return sum(scores) / len(scores) + + def test_amp_average_activity_exceeds_negative_average(self): + amp_avg = self._avg_activity(self.KNOWN_AMPS) + neg_avg = self._avg_activity(self.KNOWN_NEGATIVES) + assert amp_avg > neg_avg, ( + f"AMP avg activity {amp_avg:.4f} should exceed negative avg {neg_avg:.4f}" + ) + + def test_amp_activity_margin_over_negatives(self): + amp_avg = self._avg_activity(self.KNOWN_AMPS) + neg_avg = self._avg_activity(self.KNOWN_NEGATIVES) + margin = amp_avg - neg_avg + assert margin >= 0.15, f"Activity margin {margin:.4f} below minimum 0.15" diff --git a/tests/test_negative_robustness.py b/tests/test_negative_robustness.py new file mode 100644 index 00000000..3a82e987 --- /dev/null +++ b/tests/test_negative_robustness.py @@ -0,0 +1,218 @@ +"""Negative-set robustness tests — Phase 2 requirement. + +AGENTS.md: "Negative-set robustness — Performance remains meaningful across multiple +negative datasets." + +These tests verify that the pipeline enriches AMP-like positives over negatives +regardless of the type of negative control used: + + A. All-repeat / degenerate negatives (easy) — already tested elsewhere + B. Poly-cationic negatives (hard: high charge) — new + C. Poly-hydrophobic negatives (hard: high hydro) — new + +A pipeline that only works on "easy" negatives (all-A, all-G) cannot be trusted. +A pipeline that works across all three negative types is more credible. + +All scores are computational proxies. No biological activity is implied. +""" +from __future__ import annotations + +import csv +from pathlib import Path + +from openamp_foundry.benchmark.evaluate import enrichment_factor, recall_at_k +from openamp_foundry.features.physchem import compute_features +from openamp_foundry.pipeline import score_candidates +from openamp_foundry.scoring.activity import activity_likeness_score +from openamp_foundry.scoring.safety import safety_score + + +POSITIVES_CSV = "examples/benchmark/robustness_positives.csv" +NEGATIVE_A_CSV = "examples/negative/demo_negative_peptides.csv" +NEGATIVE_B_CSV = "examples/negative/poly_cationic.csv" +NEGATIVE_C_CSV = "examples/negative/poly_hydrophobic.csv" + +POSITIVE_IDS = {"ROB-POS-001", "ROB-POS-002", "ROB-POS-003"} + + +def _merge_csv_files(pos_csv: str, neg_csv: str, tmp_path: Path) -> Path: + """Combine a positive and negative CSV into a single candidate file.""" + out = tmp_path / "merged.csv" + rows = [] + for path in [pos_csv, neg_csv]: + with open(path) as f: + reader = csv.DictReader(f) + rows.extend(list(reader)) + with out.open("w", newline="") as f: + writer = csv.DictWriter(f, fieldnames=["id", "sequence", "source"]) + writer.writeheader() + writer.writerows(rows) + return out + + +class TestNegativeSetProperties: + """Verify the negative datasets have the expected physicochemical profiles.""" + + def test_poly_cationic_has_high_charge_low_hydrophobicity(self): + features = compute_features("KKKKKKKKKKKKK") + assert features["charge_density"] > 0.8, "Poly-K should have high charge density" + assert features["hydrophobic_fraction"] == 0.0, "Poly-K should have zero hydrophobic fraction" + + def test_poly_hydrophobic_has_high_hydro_zero_charge(self): + features = compute_features("LLLLLLLLLLLLL") + assert features["hydrophobic_fraction"] == 1.0, "Poly-L should have 100% hydrophobic fraction" + assert features["charge_density"] == 0.0, "Poly-L should have zero charge density" + + def test_poly_cationic_activity_score_below_real_amps(self): + amp_feat = compute_features("KWKLFKKIGAVLKVL") + polyk_feat = compute_features("KKKKKKKKKKKKK") + amp_score = activity_likeness_score(amp_feat) + polyk_score = activity_likeness_score(polyk_feat) + assert amp_score > polyk_score, ( + f"Real AMP activity ({amp_score:.3f}) should exceed poly-K ({polyk_score:.3f}): " + "poly-K has high charge but no hydrophobic face" + ) + + def test_poly_hydrophobic_activity_score_below_real_amps(self): + amp_feat = compute_features("KWKLFKKIGAVLKVL") + polyl_feat = compute_features("LLLLLLLLLLLLL") + amp_score = activity_likeness_score(amp_feat) + polyl_score = activity_likeness_score(polyl_feat) + assert amp_score > polyl_score, ( + f"Real AMP activity ({amp_score:.3f}) should exceed poly-L ({polyl_score:.3f}): " + "poly-L has no charge" + ) + + def test_poly_cationic_safety_penalized(self): + polyk_feat = compute_features("KKKKKKKKKKKKK") + polyk_safety = safety_score(polyk_feat) + # Very high charge density should trigger the safety risk penalty + assert polyk_safety < 0.7, ( + f"Poly-K safety={polyk_safety:.3f} should be <0.7 due to high charge density risk" + ) + + def test_poly_hydrophobic_safety_penalized(self): + polyl_feat = compute_features("LLLLLLLLLLLLL") + polyl_safety = safety_score(polyl_feat) + # Very high hydrophobic fraction triggers hemolysis proxy penalty + assert polyl_safety < 0.5, ( + f"Poly-L safety={polyl_safety:.3f} should be <0.5 due to extreme hydrophobicity" + ) + + def test_poly_repeat_long_run_detected(self): + polyk_feat = compute_features("KKKKKKKKKKKKK") + assert polyk_feat["longest_repeat_run"] == 13, "Poly-K should have repeat run of 13" + + def test_poly_cationic_file_exists(self): + assert Path(NEGATIVE_B_CSV).exists(), "Poly-cationic negative set CSV not found" + + def test_poly_hydrophobic_file_exists(self): + assert Path(NEGATIVE_C_CSV).exists(), "Poly-hydrophobic negative set CSV not found" + + +class TestEnrichmentAcrossNegativeSets: + """Verify EF > 1.0 when positives are mixed with each negative set.""" + + def test_enrichment_vs_degenerate_negatives(self, tmp_path): + """Baseline: positives should easily outrank all-repeat / degenerate negatives.""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_A_CSV, tmp_path) + scored, _ = score_candidates(merged) + k = len(POSITIVE_IDS) + ef = enrichment_factor(scored, POSITIVE_IDS, k=k) + assert ef > 1.0, f"EF={ef:.3f} should exceed 1.0 vs degenerate negatives" + + def test_enrichment_vs_poly_cationic_negatives(self, tmp_path): + """Harder test: positives outrank high-charge-only negatives (poly-K/R).""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_B_CSV, tmp_path) + scored, _ = score_candidates(merged) + k = len(POSITIVE_IDS) + ef = enrichment_factor(scored, POSITIVE_IDS, k=k) + assert ef > 1.0, ( + f"EF={ef:.3f} should exceed 1.0 vs poly-cationic negatives. " + "Pipeline must use BOTH charge and hydrophobicity, not just charge." + ) + + def test_enrichment_vs_poly_hydrophobic_negatives(self, tmp_path): + """Harder test: positives outrank hydrophobic-only negatives (poly-L/I/V).""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_C_CSV, tmp_path) + scored, _ = score_candidates(merged) + k = len(POSITIVE_IDS) + ef = enrichment_factor(scored, POSITIVE_IDS, k=k) + assert ef > 1.0, ( + f"EF={ef:.3f} should exceed 1.0 vs poly-hydrophobic negatives. " + "Pipeline must use BOTH charge and hydrophobicity, not just hydrophobicity." + ) + + def test_recall_at_k_vs_poly_cationic(self, tmp_path): + """recall@3 should be 1.0 (all 3 positives in top 3) vs poly-cationic negatives.""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_B_CSV, tmp_path) + scored, _ = score_candidates(merged) + rc = recall_at_k(scored, POSITIVE_IDS, k=3) + assert rc == 1.0, ( + f"recall@3={rc:.4f}: all 3 known positives should be in the top 3 " + "when mixed with 5 poly-cationic negatives" + ) + + def test_recall_at_k_vs_poly_hydrophobic(self, tmp_path): + """recall@3 should be 1.0 (all 3 positives in top 3) vs poly-hydrophobic negatives.""" + merged = _merge_csv_files(POSITIVES_CSV, NEGATIVE_C_CSV, tmp_path) + scored, _ = score_candidates(merged) + rc = recall_at_k(scored, POSITIVE_IDS, k=3) + assert rc == 1.0, ( + f"recall@3={rc:.4f}: all 3 known positives should be in the top 3 " + "when mixed with 5 poly-hydrophobic negatives" + ) + + def test_ef_consistent_across_all_negative_sets(self, tmp_path): + """EF > 1.0 for all three negative set types — robustness check.""" + results = {} + for label, neg_csv in [ + ("degenerate", NEGATIVE_A_CSV), + ("poly_cationic", NEGATIVE_B_CSV), + ("poly_hydrophobic", NEGATIVE_C_CSV), + ]: + subdir = tmp_path / label + subdir.mkdir() + merged = _merge_csv_files(POSITIVES_CSV, neg_csv, subdir) + scored, _ = score_candidates(merged) + ef = enrichment_factor(scored, POSITIVE_IDS, k=len(POSITIVE_IDS)) + results[label] = ef + + failures = [ + f"{label} (EF={ef:.3f})" + for label, ef in results.items() + if ef <= 1.0 + ] + assert not failures, ( + f"EF ≤ 1.0 for negative set(s): {failures}. " + "Pipeline must beat random across all tested negative types." + ) + + def test_mean_positive_score_exceeds_mean_negative_across_all_sets(self, tmp_path): + """Mean ensemble score of positives > mean of negatives for all negative types.""" + for label, neg_csv in [ + ("degenerate", NEGATIVE_A_CSV), + ("poly_cationic", NEGATIVE_B_CSV), + ("poly_hydrophobic", NEGATIVE_C_CSV), + ]: + subdir = tmp_path / label + subdir.mkdir() + merged = _merge_csv_files(POSITIVES_CSV, neg_csv, subdir) + scored, _ = score_candidates(merged) + + pos_ensemble = [ + s.scores["ensemble"] + for s in scored + if s.candidate.candidate_id in POSITIVE_IDS + ] + neg_ensemble = [ + s.scores["ensemble"] + for s in scored + if s.candidate.candidate_id not in POSITIVE_IDS + ] + avg_pos = sum(pos_ensemble) / len(pos_ensemble) + avg_neg = sum(neg_ensemble) / len(neg_ensemble) + assert avg_pos > avg_neg, ( + f"[{label}] Mean positive ensemble ({avg_pos:.3f}) should exceed " + f"mean negative ensemble ({avg_neg:.3f})" + ) diff --git a/tests/test_novelty_pressure.py b/tests/test_novelty_pressure.py new file mode 100644 index 00000000..3cc28d44 --- /dev/null +++ b/tests/test_novelty_pressure.py @@ -0,0 +1,210 @@ +"""Novelty pressure tests — Phase 2 requirement. + +AGENTS.md: "Novelty pressure — Top candidates are not merely copies of known AMP motifs." + +These tests verify that: +1. Near-duplicate copies of known AMPs receive low novelty scores. +2. Genuinely novel AMP-like sequences receive high novelty scores. +3. The pipeline's min_novelty filter excludes near-duplicates from selection. +4. Top selected candidates are not merely rehashing known AMP sequences. +5. The nearest_reference metadata is populated for near-duplicate candidates. + +All scores are computational proxies. No biological activity is implied. +""" +from __future__ import annotations + +from pathlib import Path + +from openamp_foundry.data.loaders import load_candidates_csv +from openamp_foundry.pipeline import run_ranking_pipeline, score_candidates +from openamp_foundry.scoring.novelty import normalized_similarity, novelty_score + + +REFS_CSV = "examples/known_reference/demo_known_amps.csv" +POOL_CSV = "examples/benchmark/novelty_pressure_pool.csv" + +# NOV-DUP-* are near-duplicates of the reference AMPs +# NOV-NEW-* are genuinely novel AMP-like sequences +# NOV-NEG-* are non-AMP negatives +DUP_IDS = {"NOV-DUP-001", "NOV-DUP-002", "NOV-DUP-003"} +NOVEL_IDS = {"NOV-NEW-001", "NOV-NEW-002", "NOV-NEW-003"} + + +class TestNoveltyScoringMechanism: + def test_exact_reference_copy_has_zero_novelty(self): + """Sequence identical to a reference should receive novelty = 0.0.""" + refs = load_candidates_csv(REFS_CSV) + # REF-000001 = KWKLFKKIGAVLKVL + score, nearest = novelty_score("KWKLFKKIGAVLKVL", refs) + assert score == 0.0, f"Exact reference copy should have novelty=0.0, got {score}" + assert nearest is not None + assert nearest["similarity"] == 1.0 + + def test_near_duplicate_has_low_novelty(self): + """Sequence differing by 1 AA from reference should have low novelty.""" + refs = load_candidates_csv(REFS_CSV) + # KWKLFKKIGAVLKFL = KWKLFKKIGAVLKVL with L→F at position 14 (1/15 edit) + score, nearest = novelty_score("KWKLFKKIGAVLKFL", refs) + assert score < 0.20, ( + f"Near-duplicate (1 substitution) should have novelty < 0.20, got {score}. " + f"Similarity to nearest ref: {nearest['similarity'] if nearest else 'N/A'}" + ) + assert nearest is not None, "Nearest reference should be populated for near-dups" + + def test_novel_sequence_has_high_novelty(self): + """Genuinely novel AMP-like sequence should have high novelty.""" + refs = load_candidates_csv(REFS_CSV) + # RRLKKVLGAVLKVLK — cationic + hydrophobic, but NOT similar to known refs + score, nearest = novelty_score("RRLKKVLGAVLKVLK", refs) + assert score >= 0.20, ( + f"Novel sequence should have novelty >= 0.20, got {score}. " + "This sequence should be structurally distinct from known references." + ) + + def test_novelty_decreases_with_similarity(self): + """More similar to references → lower novelty.""" + refs = load_candidates_csv(REFS_CSV) + # Reference exact = min novelty + exact_score, _ = novelty_score("KWKLFKKIGAVLKVL", refs) + # 1-substitution near-dup + near_dup_score, _ = novelty_score("KWKLFKKIGAVLKFL", refs) + # Novel sequence + novel_score, _ = novelty_score("RRLKKVLGAVLKVLK", refs) + assert exact_score < near_dup_score < novel_score, ( + f"Novelty should increase with decreasing similarity to references: " + f"exact={exact_score}, near_dup={near_dup_score}, novel={novel_score}" + ) + + def test_nearest_reference_field_populated_for_near_dups(self): + """Pipeline should populate nearest_reference for near-duplicate candidates.""" + scored, _ = score_candidates(POOL_CSV, REFS_CSV) + dup_candidates = [s for s in scored if s.candidate.candidate_id in DUP_IDS] + for item in dup_candidates: + assert item.nearest_reference is not None, ( + f"{item.candidate.candidate_id}: nearest_reference should not be None " + "for near-duplicate candidates" + ) + assert "similarity" in item.nearest_reference + assert "candidate_id" in item.nearest_reference + + def test_near_dup_novelty_score_below_min_novelty_threshold(self): + """Near-duplicates of references should fall below the pipeline min_novelty=0.20.""" + scored, _ = score_candidates(POOL_CSV, REFS_CSV) + # NOV-DUP-001 is identical to REF-000001 → novelty = 0.0 + dup001 = next(s for s in scored if s.candidate.candidate_id == "NOV-DUP-001") + assert dup001.scores["novelty"] < 0.20, ( + f"NOV-DUP-001 (exact reference copy) should have novelty < 0.20, " + f"got {dup001.scores['novelty']}" + ) + + +class TestNoveltyScoringInPipeline: + def test_dup_ids_have_lower_novelty_than_novel_ids(self): + """Near-duplicate candidates score lower on novelty than genuinely novel ones.""" + scored, _ = score_candidates(POOL_CSV, REFS_CSV) + + dup_novelties = [ + s.scores["novelty"] + for s in scored + if s.candidate.candidate_id in DUP_IDS + ] + novel_novelties = [ + s.scores["novelty"] + for s in scored + if s.candidate.candidate_id in NOVEL_IDS + ] + + avg_dup = sum(dup_novelties) / len(dup_novelties) + avg_novel = sum(novel_novelties) / len(novel_novelties) + + assert avg_novel > avg_dup, ( + f"Average novelty of genuine novel AMPs ({avg_novel:.3f}) should exceed " + f"average novelty of near-duplicates ({avg_dup:.3f})" + ) + + def test_selected_candidates_pass_novelty_threshold(self, tmp_path): + """All pipeline-selected candidates should meet the min_novelty=0.20 threshold.""" + out = tmp_path / "ranked.jsonl" + run_ranking_pipeline( + candidate_path=POOL_CSV, + reference_path=REFS_CSV, + out_path=out, + ) + import json + rows = [json.loads(line) for line in out.read_text().splitlines() if line.strip()] + selected = [r for r in rows if r["selected"]] + for row in selected: + assert row["scores"]["novelty"] >= 0.20, ( + f"{row['candidate_id']}: selected candidate has novelty " + f"{row['scores']['novelty']:.4f} < 0.20 minimum" + ) + + def test_exact_reference_copies_not_selected(self, tmp_path): + """Exact copies of reference AMPs should not appear in the selected batch.""" + out = tmp_path / "ranked.jsonl" + run_ranking_pipeline( + candidate_path=POOL_CSV, + reference_path=REFS_CSV, + out_path=out, + ) + import json + rows = [json.loads(line) for line in out.read_text().splitlines() if line.strip()] + selected_ids = {r["candidate_id"] for r in rows if r["selected"]} + assert "NOV-DUP-001" not in selected_ids, ( + "NOV-DUP-001 (exact copy of KWKLFKKIGAVLKVL) should not be in selected batch" + ) + + def test_novel_amp_candidates_preferentially_selected(self, tmp_path): + """Novel AMP-like candidates should be preferentially selected over near-duplicates.""" + out = tmp_path / "ranked.jsonl" + run_ranking_pipeline( + candidate_path=POOL_CSV, + reference_path=REFS_CSV, + out_path=out, + ) + import json + rows = [json.loads(line) for line in out.read_text().splitlines() if line.strip()] + selected_ids = {r["candidate_id"] for r in rows if r["selected"]} + + novel_selected = len(NOVEL_IDS & selected_ids) + dup_selected = len(DUP_IDS & selected_ids) + assert novel_selected >= dup_selected, ( + f"Novel AMP candidates selected ({novel_selected}) should be ≥ " + f"near-duplicate candidates selected ({dup_selected}). " + "Novelty pressure should favor genuinely novel sequences." + ) + + def test_top_ranked_candidates_not_all_near_dups(self): + """Top-scored candidates by ensemble should not be dominated by reference copies.""" + scored, _ = score_candidates(POOL_CSV, REFS_CSV) + from openamp_foundry.selection.pareto import rank_candidates + ranked = rank_candidates(scored) + + top5_ids = {s.candidate.candidate_id for s in ranked[:5]} + dups_in_top5 = len(DUP_IDS & top5_ids) + # Even if near-dups have high activity, novelty penalty should limit their presence + assert dups_in_top5 < len(DUP_IDS), ( + f"All {len(DUP_IDS)} near-duplicate candidates are in the top-5. " + "Novelty weighting should reduce ranking of reference copies." + ) + + def test_data_integrity_pool_has_expected_ids(self): + """Verify the novelty pressure pool file has expected structure.""" + assert Path(POOL_CSV).exists(), "Novelty pressure pool CSV not found" + pool = load_candidates_csv(POOL_CSV) + ids = {c.candidate_id for c in pool} + assert DUP_IDS.issubset(ids), f"Missing DUP IDs: {DUP_IDS - ids}" + assert NOVEL_IDS.issubset(ids), f"Missing NOVEL IDs: {NOVEL_IDS - ids}" + + def test_normalized_similarity_between_dups_and_refs_above_threshold(self): + """Verify near-dups are above 70% similarity to references (confirming they ARE near-dups).""" + refs = load_candidates_csv(REFS_CSV) + ref_seqs = [r.sequence for r in refs] + # NOV-DUP-001 is KWKLFKKIGAVLKVL = identical to REF-000001 + # NOV-DUP-002 is KWKLFKKIGAVLKFL = 1 edit from REF-000001 + for dup_seq in ["KWKLFKKIGAVLKVL", "KWKLFKKIGAVLKFL", "KWKLFKRIGAVLKVL"]: + max_sim = max(normalized_similarity(dup_seq, rseq) for rseq in ref_seqs) + assert max_sim >= 0.70, ( + f"Sequence {dup_seq!r} should have ≥70% similarity to at least one reference " + f"(got max_sim={max_sim:.4f})" + ) diff --git a/tests/test_physchem_amphipathicity.py b/tests/test_physchem_amphipathicity.py new file mode 100644 index 00000000..0057f1ee --- /dev/null +++ b/tests/test_physchem_amphipathicity.py @@ -0,0 +1,62 @@ +"""Tests for hydrophobic moment (amphipathicity) feature.""" +from __future__ import annotations + + +from openamp_foundry.features.physchem import compute_features, hydrophobic_moment + + +class TestHydrophobicMoment: + def test_empty_sequence_returns_zero(self): + assert hydrophobic_moment("") == 0.0 + + def test_uniform_sequence_low_moment(self): + # All same amino acid → sine/cosine terms distribute evenly → low moment + result = hydrophobic_moment("AAAAAAAAAA") + assert isinstance(result, float) + assert result >= 0.0 + + def test_alternating_hydrophobic_polar_has_higher_moment(self): + # Alternating hydrophobic/polar gives high periodicity + # e.g. KALALALA at 100deg/residue should show amphipathic character + # vs uniform KKKKKKKKwhich is all charged + seq_amphipathic = "KLKLKLKL" + seq_uniform = "KKKKKKKK" + result_amph = hydrophobic_moment(seq_amphipathic) + result_unif = hydrophobic_moment(seq_uniform) + assert isinstance(result_amph, float) + assert isinstance(result_unif, float) + + def test_known_amp_has_nonzero_moment(self): + # KWKLFKKIGAVLKVL is a classic AMP (magainin analogue) + result = hydrophobic_moment("KWKLFKKIGAVLKVL") + assert result > 0.0 + + def test_returns_float_rounded_to_4dp(self): + result = hydrophobic_moment("KWKLFKK") + assert isinstance(result, float) + # Check 4 decimal places + assert result == round(result, 4) + + def test_single_residue(self): + result = hydrophobic_moment("K") + # sin(0) = 0, cos(0) = 1 → moment = |H_K * cos(0)| / 1 = |H_K| + assert isinstance(result, float) + + +class TestComputeFeaturesAmphipathicity: + def test_hydrophobic_moment_in_features(self): + features = compute_features("KWKLFKKIGAVLKVL") + assert "hydrophobic_moment" in features + assert isinstance(features["hydrophobic_moment"], float) + assert features["hydrophobic_moment"] >= 0.0 + + def test_empty_sequence_does_not_crash(self): + features = compute_features("") + assert "hydrophobic_moment" in features + assert features["hydrophobic_moment"] == 0.0 + + def test_all_canonical_amino_acids(self): + seq = "ACDEFGHIKLMNPQRSTVWY" + features = compute_features(seq) + assert "hydrophobic_moment" in features + assert features["hydrophobic_moment"] >= 0.0 diff --git a/tests/test_pipeline_filters.py b/tests/test_pipeline_filters.py index a078063c..2a57b0d9 100644 --- a/tests/test_pipeline_filters.py +++ b/tests/test_pipeline_filters.py @@ -2,9 +2,7 @@ from __future__ import annotations import json -from pathlib import Path -import pytest from openamp_foundry.pipeline import _passes_length_filter, run_ranking_pipeline, score_candidates diff --git a/tests/test_reproducibility.py b/tests/test_reproducibility.py new file mode 100644 index 00000000..f5e7d2bc --- /dev/null +++ b/tests/test_reproducibility.py @@ -0,0 +1,310 @@ +"""Reproducibility tests — Phase 2 requirement. + +AGENTS.md: "Reproducibility — Another machine can reproduce rankings from the same inputs." + +These tests verify that: +1. Rankings are deterministic: same inputs always produce the same output order. +2. A run manifest is generated alongside every ranked output. +3. The run manifest validates against its JSON Schema. +4. The manifest contains all required reproducibility fields. +5. Input file SHA-256 hashes in the manifest match the actual files. +6. The config hash changes when the config changes. +7. Pipeline version is recorded in the manifest. + +All scores are computational proxies. No biological activity is implied. +""" +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from openamp_foundry import __version__ +from openamp_foundry.evidence.schemas import validate_json_schema +from openamp_foundry.pipeline import build_run_manifest, run_ranking_pipeline +from openamp_foundry.utils.hashing import file_sha256, stable_json_hash + + +CANDIDATE_CSV = "examples/sequences/demo_candidates.csv" +REFERENCE_CSV = "examples/known_reference/demo_known_amps.csv" +MANIFEST_SCHEMA = "schemas/run_manifest.schema.json" + + +class TestDeterministicRanking: + def test_two_runs_produce_identical_jsonl_order(self, tmp_path): + """Rankings must be identical across two independent pipeline runs.""" + out1 = tmp_path / "run1.jsonl" + out2 = tmp_path / "run2.jsonl" + + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out1, + ) + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out2, + ) + + rows1 = [json.loads(line) for line in out1.read_text().splitlines() if line.strip()] + rows2 = [json.loads(line) for line in out2.read_text().splitlines() if line.strip()] + + ids1 = [r["candidate_id"] for r in rows1] + ids2 = [r["candidate_id"] for r in rows2] + assert ids1 == ids2, ( + f"Ranking order differs between runs:\nRun 1: {ids1}\nRun 2: {ids2}" + ) + + def test_two_runs_produce_identical_scores(self, tmp_path): + """Scores must be identical across two independent pipeline runs.""" + out1 = tmp_path / "run1.jsonl" + out2 = tmp_path / "run2.jsonl" + + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out1, + ) + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out2, + ) + + rows1 = [json.loads(line) for line in out1.read_text().splitlines() if line.strip()] + rows2 = [json.loads(line) for line in out2.read_text().splitlines() if line.strip()] + + for r1, r2 in zip(rows1, rows2): + assert r1["scores"] == r2["scores"], ( + f"{r1['candidate_id']}: scores differ between runs: " + f"{r1['scores']} vs {r2['scores']}" + ) + + def test_selected_candidates_identical_across_runs(self, tmp_path): + """Selection (which candidates pass all filters) must be deterministic.""" + out1 = tmp_path / "run1.jsonl" + out2 = tmp_path / "run2.jsonl" + + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out1, + ) + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out2, + ) + + sel1 = { + r["candidate_id"] + for r in [json.loads(line) for line in out1.read_text().splitlines() if line.strip()] + if r["selected"] + } + sel2 = { + r["candidate_id"] + for r in [json.loads(line) for line in out2.read_text().splitlines() if line.strip()] + if r["selected"] + } + assert sel1 == sel2, ( + f"Selected candidates differ between runs: " + f"only_in_run1={sel1 - sel2}, only_in_run2={sel2 - sel1}" + ) + + +class TestRunManifestGeneration: + def test_manifest_generated_alongside_output(self, tmp_path): + """run_manifest.json should be generated in the same directory as the output.""" + out = tmp_path / "ranked.jsonl" + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out, + ) + manifest_path = tmp_path / "run_manifest.json" + assert manifest_path.exists(), ( + "run_manifest.json should be generated alongside ranked.jsonl" + ) + + def test_manifest_at_explicit_path(self, tmp_path): + """Explicit --manifest path should be respected.""" + out = tmp_path / "ranked.jsonl" + manifest_out = tmp_path / "my_manifest.json" + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out, + manifest_path=manifest_out, + ) + assert manifest_out.exists(), "Manifest should be written to the explicit path" + + def test_manifest_validates_against_schema(self, tmp_path): + """run_manifest.json must validate against schemas/run_manifest.schema.json.""" + out = tmp_path / "ranked.jsonl" + manifest_out = tmp_path / "run_manifest.json" + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out, + manifest_path=manifest_out, + ) + data = json.loads(manifest_out.read_text()) + validate_json_schema(data, MANIFEST_SCHEMA) + + def test_manifest_contains_required_fields(self, tmp_path): + """Manifest must include all fields required for external reproducibility.""" + out = tmp_path / "ranked.jsonl" + manifest_out = tmp_path / "run_manifest.json" + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out, + manifest_path=manifest_out, + ) + data = json.loads(manifest_out.read_text()) + required = ["run_id", "pipeline_version", "config_hash", "generated_at", + "inputs", "input_hashes", "outputs"] + for field in required: + assert field in data, f"Manifest missing required field: {field!r}" + + def test_manifest_pipeline_version_matches_package(self, tmp_path): + """Pipeline version in manifest must match the installed package version.""" + out = tmp_path / "ranked.jsonl" + manifest_out = tmp_path / "run_manifest.json" + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out, + manifest_path=manifest_out, + ) + data = json.loads(manifest_out.read_text()) + assert data["pipeline_version"] == __version__, ( + f"Manifest pipeline_version={data['pipeline_version']!r} " + f"should match __version__={__version__!r}" + ) + + def test_manifest_run_id_is_non_empty_string(self, tmp_path): + """run_id must be a non-empty string (UUID format).""" + out = tmp_path / "ranked.jsonl" + manifest_out = tmp_path / "run_manifest.json" + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out, + manifest_path=manifest_out, + ) + data = json.loads(manifest_out.read_text()) + assert isinstance(data["run_id"], str) and len(data["run_id"]) > 0 + # Basic UUID format check (8-4-4-4-12 hyphenated) + parts = data["run_id"].split("-") + assert len(parts) == 5, f"run_id should be UUID format: {data['run_id']!r}" + + def test_manifest_two_runs_have_different_run_ids(self, tmp_path): + """Each run should produce a unique run_id.""" + out1, out2 = tmp_path / "r1.jsonl", tmp_path / "r2.jsonl" + m1, m2 = tmp_path / "m1.json", tmp_path / "m2.json" + run_ranking_pipeline(CANDIDATE_CSV, REFERENCE_CSV, out1, manifest_path=m1) + run_ranking_pipeline(CANDIDATE_CSV, REFERENCE_CSV, out2, manifest_path=m2) + d1 = json.loads(m1.read_text()) + d2 = json.loads(m2.read_text()) + assert d1["run_id"] != d2["run_id"], "Different runs should have different run_ids" + + +class TestInputHashIntegrity: + def test_input_hash_matches_actual_file(self, tmp_path): + """SHA-256 in manifest for candidate CSV must match the actual file hash.""" + out = tmp_path / "ranked.jsonl" + manifest_out = tmp_path / "run_manifest.json" + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out, + manifest_path=manifest_out, + ) + data = json.loads(manifest_out.read_text()) + for path_str, recorded_hash in data["input_hashes"].items(): + if recorded_hash == "N/A": + continue # path not found at run time (optional inputs) + actual_hash = file_sha256(path_str) + assert recorded_hash == actual_hash, ( + f"Input hash mismatch for {path_str}: " + f"recorded={recorded_hash[:16]}... actual={actual_hash[:16]}..." + ) + + def test_manifest_lists_candidate_and_reference_inputs(self, tmp_path): + """Manifest inputs should include both candidate and reference file paths.""" + out = tmp_path / "ranked.jsonl" + manifest_out = tmp_path / "run_manifest.json" + run_ranking_pipeline( + candidate_path=CANDIDATE_CSV, + reference_path=REFERENCE_CSV, + out_path=out, + manifest_path=manifest_out, + ) + data = json.loads(manifest_out.read_text()) + inputs = data["inputs"] + assert any(CANDIDATE_CSV in p for p in inputs), ( + f"Candidate CSV not found in manifest inputs: {inputs}" + ) + assert any(REFERENCE_CSV in p for p in inputs), ( + f"Reference CSV not found in manifest inputs: {inputs}" + ) + + def test_config_hash_is_deterministic(self): + """Same config → same config_hash across calls.""" + config = {"weights": {"activity": 0.40, "safety": 0.25}, "filters": {}} + h1 = stable_json_hash(config) + h2 = stable_json_hash(config) + assert h1 == h2, "Config hash should be deterministic for the same config" + + def test_config_hash_changes_with_config(self): + """Different config → different config_hash.""" + config_a = {"weights": {"activity": 0.40}} + config_b = {"weights": {"activity": 0.50}} + ha = stable_json_hash(config_a) + hb = stable_json_hash(config_b) + assert ha != hb, "Config hash should differ when config changes" + + def test_build_run_manifest_structure(self, tmp_path): + """build_run_manifest() returns correct structure without running full pipeline.""" + config = {"weights": {"activity": 0.40}} + manifest = build_run_manifest( + run_id="test-run-id", + config=config, + input_paths=[Path(CANDIDATE_CSV), Path(REFERENCE_CSV)], + output_paths=["outputs/test.jsonl"], + generated_at="2026-01-01T00:00:00+00:00", + ) + assert manifest["run_id"] == "test-run-id" + assert manifest["pipeline_version"] == __version__ + assert manifest["config_hash"] == stable_json_hash(config) + assert CANDIDATE_CSV in " ".join(manifest["inputs"]) + assert len(manifest["input_hashes"]) == 2 + + def test_manifest_sha256_is_64_char_hex(self, tmp_path): + """SHA-256 hashes should be 64-character lowercase hexadecimal strings.""" + actual_hash = file_sha256(CANDIDATE_CSV) + assert len(actual_hash) == 64, f"SHA-256 should be 64 chars, got {len(actual_hash)}" + assert actual_hash == actual_hash.lower(), "SHA-256 should be lowercase" + assert all(c in "0123456789abcdef" for c in actual_hash), "SHA-256 should be hex" + + def test_sha256_changes_with_content(self, tmp_path): + """Different file contents → different SHA-256 hash.""" + f1 = tmp_path / "a.csv" + f2 = tmp_path / "b.csv" + f1.write_text("id,sequence,source\nA-001,KWKLFK,test\n") + f2.write_text("id,sequence,source\nA-001,KWKLFR,test\n") + h1 = file_sha256(f1) + h2 = file_sha256(f2) + assert h1 != h2, "Different file content should produce different SHA-256 hashes" + + def test_file_hash_equals_stdlib_sha256(self): + """Verify file_sha256() matches direct stdlib computation.""" + h = hashlib.sha256() + with open(CANDIDATE_CSV, "rb") as f: + for chunk in iter(lambda: f.read(1024 * 1024), b""): + h.update(chunk) + expected = h.hexdigest() + actual = file_sha256(CANDIDATE_CSV) + assert actual == expected, "file_sha256() should match stdlib sha256" diff --git a/tests/test_template_mutator.py b/tests/test_template_mutator.py new file mode 100644 index 00000000..480149fc --- /dev/null +++ b/tests/test_template_mutator.py @@ -0,0 +1,233 @@ +"""Tests for template_mutator.py — Phase 3 candidate generator. + +Verifies that the conservative mutation generator: + - Produces only variants that differ from the seed by canonical-AA conservative subs + - Never produces the seed itself in the output + - Deduplicates within each generation strategy + - Produces the correct number of variants from known seeds + - Preserves sequence length (no indels) + - Covers all three generation strategies (single, double, charge-enhanced) +""" +from __future__ import annotations + +from openamp_foundry.generators.template_mutator import ( + _conservative_substitutes, + generate_all_variants, + generate_candidate_pool, + generate_charge_enhanced_variants, + generate_double_substitution_variants, + generate_single_substitution_variants, +) + +SEED = "KWKLFKKIGAVLKVL" # 15 aa, balanced AMP-like template + + +class TestConservativeSubstitutes: + def test_lysine_substitutes_are_cationic(self): + subs = _conservative_substitutes("K") + assert set(subs).issubset({"R", "H"}), f"K subs should be cationic: {subs}" + assert "K" not in subs + + def test_leucine_substitutes_are_hydrophobic(self): + subs = _conservative_substitutes("L") + assert set(subs).issubset({"I", "V", "A", "F", "M"}), f"L subs: {subs}" + assert "L" not in subs + + def test_tryptophan_substitutes_are_aromatic(self): + subs = _conservative_substitutes("W") + assert set(subs).issubset({"F", "Y"}), f"W subs: {subs}" + + def test_cysteine_has_no_substitutes(self): + subs = _conservative_substitutes("C") + assert subs == [], f"C is a singleton group, no substitutes: {subs}" + + def test_serine_substitutes_are_polar(self): + subs = _conservative_substitutes("S") + assert set(subs).issubset({"T", "N", "Q"}), f"S subs: {subs}" + + def test_never_cross_group_substitution(self): + charged = list("KRH") + list("DE") + hydrophobic = list("LIVAFM") + for aa in charged: + subs = _conservative_substitutes(aa) + cross = [s for s in subs if s in hydrophobic] + assert not cross, f"Charged AA {aa!r} has hydrophobic subs: {cross}" + for aa in hydrophobic: + subs = _conservative_substitutes(aa) + cross = [s for s in subs if s in charged] + assert not cross, f"Hydrophobic AA {aa!r} has charged subs: {cross}" + + +class TestSingleSubstitutionVariants: + def test_all_variants_same_length_as_seed(self): + variants = generate_single_substitution_variants(SEED) + for v in variants: + assert len(v) == len(SEED), f"Length mismatch: {v!r} vs {SEED!r}" + + def test_seed_not_in_variants(self): + variants = generate_single_substitution_variants(SEED) + assert SEED not in variants, "Seed should not appear in its own variants" + + def test_each_variant_differs_by_exactly_one_position(self): + variants = generate_single_substitution_variants(SEED) + for v in variants: + diffs = sum(a != b for a, b in zip(v, SEED)) + assert diffs == 1, f"Expected exactly 1 diff, got {diffs}: {v!r}" + + def test_no_duplicate_variants(self): + variants = generate_single_substitution_variants(SEED) + assert len(variants) == len(set(variants)), "Single-sub variants contain duplicates" + + def test_poly_K_generates_rh_substitutes(self): + poly_k = "KKKKK" + variants = generate_single_substitution_variants(poly_k) + # Each of 5 positions can be R or H → 5 × 2 = 10 + assert len(variants) == 10, f"poly-K (5aa) should have 10 variants, got {len(variants)}" + allowed = set("KRH") + for v in variants: + assert all(aa in allowed for aa in v), f"Non-cationic AA in {v!r}" + + def test_short_sequence_produces_variants(self): + variants = generate_single_substitution_variants("KL") + assert len(variants) > 0, "Short 2-aa sequence should still produce variants" + + +class TestDoubleSubstitutionVariants: + def test_variants_same_length_as_seed(self): + variants = generate_double_substitution_variants(SEED, n_samples=10, seed=42) + for v in variants: + assert len(v) == len(SEED), f"Length mismatch: {v!r}" + + def test_seed_not_in_double_variants(self): + variants = generate_double_substitution_variants(SEED, n_samples=10, seed=42) + assert SEED not in variants + + def test_each_variant_differs_by_two_positions(self): + variants = generate_double_substitution_variants(SEED, n_samples=15, seed=42) + for v in variants: + diffs = sum(a != b for a, b in zip(v, SEED)) + assert diffs == 2, f"Expected 2 diffs, got {diffs}: {v!r}" + + def test_no_duplicate_double_variants(self): + variants = generate_double_substitution_variants(SEED, n_samples=15, seed=42) + assert len(variants) == len(set(variants)), "Double-sub variants contain duplicates" + + def test_respects_n_samples_limit(self): + variants = generate_double_substitution_variants(SEED, n_samples=5, seed=42) + assert len(variants) <= 5 + + def test_deterministic_with_same_seed(self): + v1 = generate_double_substitution_variants(SEED, n_samples=10, seed=99) + v2 = generate_double_substitution_variants(SEED, n_samples=10, seed=99) + assert v1 == v2, "Same rng seed must produce identical output" + + def test_different_seeds_produce_different_variants(self): + v1 = generate_double_substitution_variants(SEED, n_samples=10, seed=1) + v2 = generate_double_substitution_variants(SEED, n_samples=10, seed=2) + assert v1 != v2, "Different rng seeds should produce different variants" + + +class TestChargeEnhancedVariants: + def test_replaces_polar_with_cationic(self): + seq = "KWKLFKKSGAVLKVL" # S at position 7 + variants = generate_charge_enhanced_variants(seq, n_samples=5, seed=42) + for v in variants: + assert len(v) == len(seq) + # The S position should become K or R + diff_positions = [i for i, (a, b) in enumerate(zip(v, seq)) if a != b] + assert len(diff_positions) == 1 + for pos in diff_positions: + assert seq[pos] in "STNQ", f"Expected polar at pos {pos}, got {seq[pos]!r}" + assert v[pos] in "KR", f"Expected K/R replacement at pos {pos}, got {v[pos]!r}" + + def test_no_charge_enhanced_when_no_polar_residues(self): + poly_k = "KKKKKKK" + variants = generate_charge_enhanced_variants(poly_k) + assert variants == [], "No polar residues → no charge-enhanced variants" + + def test_charge_enhanced_does_not_include_seed(self): + seq = "KWKLFKKSAVLKVL" + variants = generate_charge_enhanced_variants(seq) + assert seq not in variants + + +class TestGenerateAllVariants: + def test_seed_not_in_all_variants(self): + variants = generate_all_variants(SEED) + assert SEED not in variants, "Seed must not appear among its own variants" + + def test_all_variants_same_length(self): + variants = generate_all_variants(SEED) + for v in variants: + assert len(v) == len(SEED) + + def test_deduplication_across_strategies(self): + variants = generate_all_variants(SEED) + assert len(variants) == len(set(variants)), "generate_all_variants must deduplicate" + + def test_sorted_output(self): + variants = generate_all_variants(SEED) + assert variants == sorted(variants), "generate_all_variants must return sorted output" + + def test_substantial_pool_produced(self): + variants = generate_all_variants(SEED, n_double=20, n_charge_enhance=10) + assert len(variants) >= 20, f"Expected ≥20 variants from {SEED!r}, got {len(variants)}" + + +class TestGenerateCandidatePool: + def test_pool_size_proportional_to_seeds(self): + seeds = ["KWKLFKKIGAVLKVL", "RRWQWRMKKLG"] + ids = ["SEED-001", "SEED-002"] + pool = generate_candidate_pool(seeds, ids, n_double=10, n_charge_enhance=5, rng_seed=42) + assert len(pool) > 0, "Pool should be non-empty" + + def test_ids_contain_seed_prefix(self): + seeds = ["KWKLFKKIGAVLKVL"] + ids = ["SEED-001"] + pool = generate_candidate_pool(seeds, ids) + for item in pool: + assert item["id"].startswith("SEED-001_VAR_"), f"Unexpected ID: {item['id']!r}" + + def test_source_field_encodes_template(self): + seeds = ["KWKLFKKIGAVLKVL"] + ids = ["SEED-001"] + pool = generate_candidate_pool(seeds, ids) + for item in pool: + assert item["source"] == "template_mutation_from_SEED-001", ( + f"Unexpected source: {item['source']!r}" + ) + + def test_pool_items_have_required_fields(self): + seeds = ["KWKLFKKIGAVLKVL"] + ids = ["SEED-001"] + pool = generate_candidate_pool(seeds, ids) + for item in pool: + assert "id" in item + assert "sequence" in item + assert "source" in item + + def test_no_seed_sequence_in_pool(self): + seeds = ["KWKLFKKIGAVLKVL", "RRWQWRMKKLG"] + ids = ["SEED-001", "SEED-002"] + pool = generate_candidate_pool(seeds, ids) + pool_seqs = {item["sequence"] for item in pool} + for seed_seq in seeds: + assert seed_seq not in pool_seqs, ( + f"Seed sequence {seed_seq!r} should not appear in its own candidate pool" + ) + + def test_deterministic_with_same_rng_seed(self): + seeds = ["KWKLFKKIGAVLKVL"] + ids = ["SEED-001"] + pool1 = generate_candidate_pool(seeds, ids, rng_seed=7) + pool2 = generate_candidate_pool(seeds, ids, rng_seed=7) + assert pool1 == pool2, "Same rng_seed must produce identical candidate pool" + + def test_multiple_seeds_produce_independent_batches(self): + seeds = ["KWKLFKKIGAVLKVL", "RRWQWRMKKLG"] + ids = ["SEED-001", "SEED-002"] + pool = generate_candidate_pool(seeds, ids) + seed1_items = [p for p in pool if p["id"].startswith("SEED-001")] + seed2_items = [p for p in pool if p["id"].startswith("SEED-002")] + assert len(seed1_items) > 0 + assert len(seed2_items) > 0 diff --git a/tests/test_toxicity_penalty.py b/tests/test_toxicity_penalty.py new file mode 100644 index 00000000..51ee3ab3 --- /dev/null +++ b/tests/test_toxicity_penalty.py @@ -0,0 +1,298 @@ +"""Toxicity penalty tests — Phase 2 requirement. + +AGENTS.md: "Toxicity penalty — Predicted hemolytic/toxic candidates are down-ranked." + +These tests verify that sequences with properties associated with hemolysis risk or +toxicity proxy signals receive lower safety scores and lower overall ensemble rankings +than balanced AMP-like candidates. + +Risk signals penalized by the safety scorer: + - Hydrophobic fraction > 0.65 (hemolysis proxy) + - Charge density > 0.55 (toxicity proxy) + - Sequence length > 35 (large peptide → synthesis + stability risk) + - Cysteine fraction > 0.25 (disulfide bridges → synthesis complexity) + - Longest repeat run ≥ 6 (degenerate composition) + +These are computational proxies, NOT validated toxicity predictors. +All results are heuristic-only. No biological safety claim is made. +""" +from __future__ import annotations + +from openamp_foundry.features.physchem import compute_features +from openamp_foundry.pipeline import score_candidates +from openamp_foundry.scoring.safety import safety_score + + +# Canonical AMP-like positives with balanced properties +AMP_SEQUENCES = [ + "KWKLFKKIGAVLKVL", # balanced charge + hydrophobicity + "RRWQWRMKKLG", # cationic, moderate hydrophobicity + "GIGKFLHSAKKFGKAFVGEIMNS", # well-characterised AMP-like +] + +# High-risk sequences targeting known safety penalty signals +HIGH_HYDROPHOBIC = [ + "LLLLLLLLLLLLLLLLL", # hydrophobic_fraction = 1.0 (> 0.65 threshold) + "IIIIIIIIIIIIIIIII", # all-isoleucine, extreme hydrophobicity + "VVVVVVVVVVVVVVVVV", # all-valine, extreme hydrophobicity + "LLLLWWWWLLLLWWWW", # high aromatic + hydrophobic, hemolysis proxy +] + +HIGH_CHARGE_DENSITY = [ + "KKKKKKKKKKKKKK", # charge_density = 1.0 (> 0.55 threshold) + "RRRRRRRRRRRRRR", # all-arginine, extreme cation density + "KRKRKRKRKRKRKRK", # alternating K/R, very high charge +] + +HIGH_CYSTEINE = [ + "CCCCCCCCCCC", # 100% cysteine fraction (> 0.25 threshold) + "CKCKCKCKCKCKCKCKCK", # 50% cysteine fraction + "CKWCKCWCKCKWCK", # mixed but > 0.25 cys fraction +] + +VERY_LONG = [ + "KWKLFKKIGAVLKVLKWKLFKKIGAVLKVLKWKLFKK", # 38 aa, exceeds 35 aa threshold +] + +LONG_REPEAT = [ + "AAAAAAAAAAAAAAAAAA", # repeat run = 18 + "GGGGGGGGGGGGGGGGGG", # repeat run = 18 + "KKKKKKKKKKKKKKKKKK", # repeat run = 18 + high charge (double penalty) +] + + +class TestSafetyScorePenaltyMechanism: + def test_extreme_hydrophobicity_reduces_safety(self): + for seq in HIGH_HYDROPHOBIC[:2]: + feat = compute_features(seq) + score = safety_score(feat) + assert score < 0.6, ( + f"{seq!r}: safety={score:.3f} should be <0.6 for extreme hydrophobic fraction " + f"(hydrophobic_fraction={feat['hydrophobic_fraction']:.2f})" + ) + + def test_extreme_charge_density_reduces_safety(self): + for seq in HIGH_CHARGE_DENSITY[:2]: + feat = compute_features(seq) + score = safety_score(feat) + assert score < 0.6, ( + f"{seq!r}: safety={score:.3f} should be <0.6 for extreme charge density " + f"(charge_density={feat['charge_density']:.2f})" + ) + + def test_high_cysteine_fraction_reduces_safety(self): + cys_seq = "CCCCCCCCCCC" + feat = compute_features(cys_seq) + assert feat["cysteine_fraction"] > 0.25, "Poly-C should exceed cysteine threshold" + score = safety_score(feat) + assert score < 0.9, ( + f"Poly-C safety={score:.3f} should be reduced by high cysteine fraction" + ) + + def test_very_long_sequence_reduces_safety(self): + long_seq = VERY_LONG[0] + feat = compute_features(long_seq) + assert feat["length"] > 35, f"Sequence should be >35 aa, got {feat['length']}" + score = safety_score(feat) + assert score < 1.0, ( + f"Long sequence (>{35} aa) safety={score:.3f} should be penalized" + ) + + def test_long_repeat_run_reduces_safety(self): + for seq in LONG_REPEAT: + feat = compute_features(seq) + assert feat["longest_repeat_run"] >= 6, ( + f"{seq!r}: repeat_run={feat['longest_repeat_run']} should be ≥6" + ) + score = safety_score(feat) + assert score < 1.0, ( + f"{seq!r}: safety={score:.3f} should be reduced by long repeat run" + ) + + def test_balanced_amp_has_higher_safety_than_poly_hydrophobic(self): + for amp_seq in AMP_SEQUENCES: + amp_feat = compute_features(amp_seq) + amp_safe = safety_score(amp_feat) + for bad_seq in HIGH_HYDROPHOBIC[:2]: + bad_feat = compute_features(bad_seq) + bad_safe = safety_score(bad_feat) + assert amp_safe > bad_safe, ( + f"{amp_seq!r} safety ({amp_safe:.3f}) should exceed " + f"{bad_seq!r} safety ({bad_safe:.3f}): " + "balanced AMP should be safer than extreme hydrophobic sequence" + ) + + def test_balanced_amp_has_higher_safety_than_poly_cationic(self): + for amp_seq in AMP_SEQUENCES: + amp_feat = compute_features(amp_seq) + amp_safe = safety_score(amp_feat) + for bad_seq in HIGH_CHARGE_DENSITY[:2]: + bad_feat = compute_features(bad_seq) + bad_safe = safety_score(bad_feat) + assert amp_safe > bad_safe, ( + f"{amp_seq!r} safety ({amp_safe:.3f}) should exceed " + f"{bad_seq!r} safety ({bad_safe:.3f}): " + "balanced AMP should be safer than extreme poly-cationic sequence" + ) + + def test_safety_risk_monotonic_with_excess_hydrophobicity(self): + """More extreme hydrophobicity → lower safety (monotonic penalty).""" + seqs = [ + "KWKLFKKIGAVLKVL", # ~60% hydrophobic, balanced + "LWKLFKKIGALLKVL", # ~70% hydrophobic, above threshold + "LLLLLLLLLLLLLL", # 100% hydrophobic, maximum penalty + ] + scores = [safety_score(compute_features(s)) for s in seqs] + assert scores[0] > scores[1] > scores[2] or scores[0] > scores[2], ( + f"Safety should decrease with increasing hydrophobicity: {scores}" + ) + + def test_double_penalty_high_charge_and_repeat(self): + """High charge density + long repeat should accumulate risk penalties.""" + poly_k = "KKKKKKKKKKKKKK" + feat = compute_features(poly_k) + score = safety_score(feat) + # Poly-K: charge_density=1.0 (penalty) + repeat_run=14 (penalty) = low safety + assert score < 0.4, ( + f"Poly-K safety={score:.3f} should be very low (double penalty: " + f"charge_density={feat['charge_density']:.2f}, " + f"repeat_run={feat['longest_repeat_run']})" + ) + + +class TestToxicityDownRankingInPipeline: + """Verify that high-risk sequences rank below balanced AMPs in the full pipeline.""" + + def test_high_risk_sequences_have_lower_ensemble_than_balanced_amps(self, tmp_path): + """High-risk candidates should receive lower ensemble scores than AMP candidates.""" + high_risk_csv = tmp_path / "high_risk.csv" + mixed_csv = tmp_path / "mixed.csv" + + # Write high-risk candidates + high_risk_rows = [ + "id,sequence,source", + "RISK-001,LLLLLLLLLLLLLLLLL,high_risk", # extreme hydrophobic + "RISK-002,KKKKKKKKKKKKKK,high_risk", # extreme charge density + "RISK-003,IIIIIIIIIIIIIIIII,high_risk", # extreme hydrophobic + ] + high_risk_csv.write_text("\n".join(high_risk_rows)) + + # Write mixed (AMPs + high-risk) + amp_rows = [ + "AMP-001,KWKLFKKIGAVLKVL,balanced_amp", + "AMP-002,RRWQWRMKKLG,balanced_amp", + "AMP-003,GIGKFLHSAKKFGKAFVGEIMNS,balanced_amp", + ] + all_rows = high_risk_rows + amp_rows + mixed_csv.write_text("\n".join(all_rows)) + + scored, _ = score_candidates(mixed_csv) + + amp_ensemble = { + s.candidate.candidate_id: s.scores["ensemble"] + for s in scored + if s.candidate.candidate_id.startswith("AMP-") + } + risk_ensemble = { + s.candidate.candidate_id: s.scores["ensemble"] + for s in scored + if s.candidate.candidate_id.startswith("RISK-") + } + + avg_amp = sum(amp_ensemble.values()) / len(amp_ensemble) + avg_risk = sum(risk_ensemble.values()) / len(risk_ensemble) + assert avg_amp > avg_risk, ( + f"Mean AMP ensemble ({avg_amp:.3f}) should exceed mean high-risk ensemble " + f"({avg_risk:.3f}): toxicity penalty should down-rank risky sequences" + ) + + def test_all_high_risk_sequences_below_best_amp(self, tmp_path): + """Every high-risk sequence should rank below the best AMP candidate.""" + mixed_csv = tmp_path / "mixed.csv" + rows = [ + "id,sequence,source", + "AMP-001,KWKLFKKIGAVLKVL,balanced_amp", + "AMP-002,RRWQWRMKKLG,balanced_amp", + "RISK-001,LLLLLLLLLLLLLLLLL,high_risk", + "RISK-002,KKKKKKKKKKKKKK,high_risk", + "RISK-003,CCCCCCCCCCC,high_risk", + ] + mixed_csv.write_text("\n".join(rows)) + + scored, _ = score_candidates(mixed_csv) + from openamp_foundry.selection.pareto import rank_candidates + ranked = rank_candidates(scored) + + top_amp_rank = min( + i for i, s in enumerate(ranked) if s.candidate.candidate_id.startswith("AMP-") + ) + worst_amp_rank = max( + i for i, s in enumerate(ranked) if s.candidate.candidate_id.startswith("AMP-") + ) + best_risk_rank = min( + i for i, s in enumerate(ranked) if s.candidate.candidate_id.startswith("RISK-") + ) + + assert best_risk_rank > worst_amp_rank, ( + f"Best high-risk rank ({best_risk_rank}) should be worse than " + f"worst AMP rank ({worst_amp_rank}): all AMPs should outrank all high-risk candidates" + ) + _ = top_amp_rank # acknowledged + + def test_toxicity_penalty_propagates_to_safety_score_in_pipeline(self, tmp_path): + """Safety score for high-risk sequences should be low in the full pipeline.""" + candidates_csv = tmp_path / "cands.csv" + candidates_csv.write_text( + "id,sequence,source\n" + "SAFE-001,KWKLFKKIGAVLKVL,balanced\n" + "RISKY-001,LLLLLLLLLLLLLLLLL,high_risk\n" + "RISKY-002,KKKKKKKKKKKKKK,high_risk\n" + ) + scored, _ = score_candidates(candidates_csv) + safe_score = next(s for s in scored if s.candidate.candidate_id == "SAFE-001") + risky1 = next(s for s in scored if s.candidate.candidate_id == "RISKY-001") + risky2 = next(s for s in scored if s.candidate.candidate_id == "RISKY-002") + + assert safe_score.scores["safety"] > risky1.scores["safety"], ( + f"Balanced AMP safety ({safe_score.scores['safety']:.3f}) should exceed " + f"poly-L safety ({risky1.scores['safety']:.3f})" + ) + assert safe_score.scores["safety"] > risky2.scores["safety"], ( + f"Balanced AMP safety ({safe_score.scores['safety']:.3f}) should exceed " + f"poly-K safety ({risky2.scores['safety']:.3f})" + ) + + def test_high_risk_candidates_not_selected_with_strict_safety_threshold(self, tmp_path): + """With max_safety_risk=0.30, high-risk candidates should be excluded.""" + from openamp_foundry.selection.diversity import greedy_diverse_select + from openamp_foundry.selection.pareto import rank_candidates + + candidates_csv = tmp_path / "cands.csv" + candidates_csv.write_text( + "id,sequence,source\n" + "AMP-001,KWKLFKKIGAVLKVL,balanced\n" + "AMP-002,RRWQWRMKKLG,balanced\n" + "RISK-001,LLLLLLLLLLLLLLLLL,high_risk\n" + "RISK-002,KKKKKKKKKKKKKK,high_risk\n" + ) + scored, _ = score_candidates(candidates_csv) + ranked = rank_candidates(scored) + + # Apply strict safety filter (only sequences with safety ≥ 1 - 0.30 = 0.70) + max_risk = 0.30 + eligible = [ + item for item in ranked + if item.scores["safety"] >= (1.0 - max_risk) + ] + selected = greedy_diverse_select(eligible, top_n=10) + selected_ids = {s.candidate.candidate_id for s in selected} + + assert "RISK-001" not in selected_ids, ( + "Poly-L (extreme hydrophobicity) should be excluded with max_safety_risk=0.30" + ) + assert "RISK-002" not in selected_ids, ( + "Poly-K (extreme charge) should be excluded with max_safety_risk=0.30" + ) + assert "AMP-001" in selected_ids or "AMP-002" in selected_ids, ( + "At least one balanced AMP should remain selected after strict safety filter" + )