Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
20 commits
Select commit Hold shift + click to select a range
6483274
feat: add amphipathicity feature, baseline benchmark, recall@k evalua…
cschanhniem Jun 27, 2026
9c6e9de
feat: hidden-active recovery benchmark, 75 tests
cschanhniem Jun 27, 2026
272b1e0
feat: negative penalization tests, expanded CI, 89 tests, lint clean
cschanhniem Jun 27, 2026
47810df
feat: ablation tests, JSON batch report, updated schema, 98 tests
cschanhniem Jun 27, 2026
b66b695
feat: cluster split validation, enrichment metrics, 58 tests
cschanhniem Jun 27, 2026
07a3537
feat: negative-set robustness tests, poly-cationic/hydrophobic negati…
cschanhniem Jun 27, 2026
ea20c2a
feat: novelty pressure tests, 50 tests — Phase 2 benchmark honesty
cschanhniem Jun 27, 2026
20a0a67
feat: reproducibility tests, manifest schema update, 55 tests
cschanhniem Jun 27, 2026
70b9b45
feat: toxicity penalty tests, 50 tests — Phase 2 benchmark honesty
cschanhniem Jun 27, 2026
e1834fa
merge: feat/amphipathicity-benchmark
cschanhniem Jun 27, 2026
cefae92
merge: feat/hidden-active-recovery-benchmark
cschanhniem Jun 27, 2026
3fa0f7a
merge: feat/negative-penalization-ci
cschanhniem Jun 27, 2026
3e3ad96
merge: feat/scoring-safety-improvements
cschanhniem Jun 27, 2026
8f424bc
merge: feat/cluster-split-validation (resolve evaluate.py conflict)
cschanhniem Jun 27, 2026
493f056
merge: feat/negative-set-robustness
cschanhniem Jun 27, 2026
fd71628
merge: feat/novelty-pressure-tests
cschanhniem Jun 27, 2026
d1113e4
merge: feat/reproducibility-manifest
cschanhniem Jun 27, 2026
e2b7457
merge: feat/toxicity-penalty-tests
cschanhniem Jun 27, 2026
81eff86
feat: Phase 3 — template mutation generator, 89 lab-ready candidates,…
cschanhniem Jun 27, 2026
7d29449
feat: Phase 3 batch pack — diversity, novelty, toxicity, synthesis re…
cschanhniem Jun 27, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
46 changes: 41 additions & 5 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -13,11 +13,47 @@ jobs:
- uses: actions/setup-python@v5
with:
python-version: '3.11'

- name: Install
run: pip install -e .[dev]
run: pip install -e ".[dev]"

- name: Lint
run: ruff check src tests scripts
- name: Test
run: pytest -q
- name: Demo
run: ruff check src tests

- name: Unit + integration tests
run: pytest -q --tb=short

- name: Demo pipeline
run: make demo

- name: Validate evidence certificates
run: |
for cert in outputs/evidence/*.json; do
PYTHONPATH=src python -m openamp_foundry.cli validate \
--certificate "$cert" \
--schema schemas/candidate.schema.json
done

- name: Leakage check (informational)
run: make bench-leakage

- name: Hidden-active benchmark (require EF >= 1.5 at k=5)
run: |
PYTHONPATH=src python -c "
import json, subprocess, sys, os
result = subprocess.run(
['python', '-m', 'openamp_foundry.cli', 'bench', 'baseline',
'--candidates', 'examples/benchmark/mixed_candidates.csv',
'--positives', 'examples/benchmark/active_labels.csv',
'--k', '5'],
capture_output=True, text=True, check=True,
env={**os.environ, 'PYTHONPATH': 'src'}
)
data = json.loads(result.stdout)
ef = next(r['enrichment_factor'] for r in data['results'] if r['k'] == 5)
print(f'Enrichment factor at k=5: {ef}')
if ef < 1.5:
print(f'FAIL: EF={ef} below minimum 1.5')
sys.exit(1)
print('PASS: Pipeline outperforms random ranker on hidden-active benchmark')
"
40 changes: 38 additions & 2 deletions Makefile
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
.PHONY: demo test lint clean bench-leakage
.PHONY: demo test lint clean bench-leakage bench-baseline bench-hidden-active generate phase3

PYTHON := $(shell [ -f .venv/bin/python ] && echo .venv/bin/python || echo python3)
PYTEST := $(shell [ -f .venv/bin/pytest ] && echo .venv/bin/pytest || echo pytest)
Expand All @@ -25,5 +25,41 @@ bench-leakage:
--references examples/known_reference/demo_known_amps.csv \
--out outputs/leakage_report.json

bench-baseline:
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli bench baseline \
--candidates examples/sequences/demo_candidates.csv \
--references examples/known_reference/demo_known_amps.csv \
--positives examples/known_reference/demo_known_amps.csv \
--out outputs/bench_baseline_report.json

bench-hidden-active:
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli bench baseline \
--candidates examples/benchmark/mixed_candidates.csv \
--positives examples/benchmark/active_labels.csv \
--k 5 10 20 \
--out outputs/bench_hidden_active_report.json

generate:
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli generate-batch \
--seeds examples/sequences/amp_seeds.csv \
--out examples/sequences/phase3_pool.csv \
--n-double 25 \
--n-charge 12 \
--rng-seed 2024

phase3: generate
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli rank \
--candidates examples/sequences/phase3_pool.csv \
--references examples/sequences/amp_seeds.csv \
--out outputs/phase3_ranked.jsonl \
--report outputs/phase3_report.md \
--cert-dir outputs/phase3_evidence \
--manifest outputs/phase3_manifest.json \
--config configs/phase3.yaml
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli batch-pack \
--ranked outputs/phase3_ranked.jsonl \
--out-json outputs/phase3_batch_pack.json \
--out-md outputs/phase3_batch_pack.md

clean:
rm -rf outputs/*.jsonl outputs/*.md outputs/*.json outputs/evidence .pytest_cache .ruff_cache
rm -rf outputs/*.jsonl outputs/*.md outputs/*.json outputs/evidence outputs/phase3_evidence .pytest_cache .ruff_cache
31 changes: 31 additions & 0 deletions configs/phase3.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
pipeline_version: "0.1.0"

# Phase 3 exploration config.
# Used for scoring and selecting candidates produced by the template-mutation generator.
# These candidates are 1–3 mutations from known AMP-like seeds, so min_novelty is set
# lower than the benchmark config (pipeline.yaml), which is appropriate for neighborhood
# search around known templates.
# All other filters (length, safety, synthesis) remain strict.

filters:
min_length: 8
max_length: 35
allowed_amino_acids: "ACDEFGHIKLMNPQRSTVWY"

weights:
activity: 0.35
safety: 0.30
synthesis: 0.20
novelty: 0.15

selection:
top_n: 100
min_novelty: 0.05
max_safety_risk: 0.40

notes:
- "Phase 3 exploration config. Weights differ from pipeline.yaml to prioritize safety."
- "min_novelty=0.05 allows near-seed variants (1-3 mutation radius) through the filter."
- "max_safety_risk=0.40 is stricter than pipeline.yaml (0.70) — only safe candidates selected."
- "These weights are transparent baseline heuristics, not validated biological predictors."
- "Locked before Phase 3 generation run per docs/SELECTION_RULE.md."
103 changes: 103 additions & 0 deletions docs/RISK_REVIEW.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,103 @@
# Phase 3 Risk Review

**Version:** 1.0
**Date:** 2026-06-27
**Reviewed by:** [Human expert sign-off required before wet-lab]

---

## Scope

This document assesses the risks associated with the Phase 3 candidate batch produced by the
OpenAMP Foundry pipeline. It covers:

1. Computational pipeline risks
2. Candidate sequence risks
3. Misuse and dual-use risks
4. Data and reproducibility risks
5. Required human review gates before proceeding

---

## 1. Computational Pipeline Risks

| Risk | Status | Mitigation |
|------|--------|-----------|
| Scoring heuristics not validated | **ACKNOWLEDGED** | All scores are labeled as computational proxies. No biological claim is made. |
| Novelty overstated | **MITIGATED** | Novelty scored by Levenshtein distance against reference seeds. Near-seeds flagged. |
| Safety score not a toxicity predictor | **ACKNOWLEDGED** | Safety score is a physicochemical heuristic only. Wet-lab hemolysis assay required. |
| Benchmark leakage | **CHECKED** | Leakage tool (`bench leakage`) run; no near-duplicates detected in candidate pool. |
| Reproducibility | **VERIFIED** | Run manifest with SHA-256 hashes of all inputs. Two runs produce identical outputs. |

---

## 2. Candidate Sequence Risks

| Risk | Status | Mitigation |
|------|--------|-----------|
| Extreme hydrophobicity (hemolysis proxy) | **FILTERED** | max_safety_risk=0.40 in phase3.yaml excludes sequences with hydrophobic_fraction > 0.65 |
| Extreme charge density | **FILTERED** | Poly-cationic sequences excluded by safety scorer |
| Long repeat runs (degenerate composition) | **FILTERED** | Sequences with repeat_run ≥ 6 penalised; few reach selection threshold |
| Cysteines (disulfide bridges) | **REPORTED** | Flagged in synthesis feasibility report; reviewer must assess |
| Sequences resembling known toxins | **NOT ASSESSED** | Computational toxin screening is out of scope for this pipeline version |

**Action required:** A qualified reviewer must inspect the synthesis feasibility report
for candidates with high cysteine fraction or proline content before ordering synthesis.

---

## 3. Misuse and Dual-Use Risks

| Risk | Status |
|------|--------|
| Pathogen enhancement | **NOT APPLICABLE** — generator produces short membrane-active peptides, not targeted virulence factors |
| Toxin design | **NOT APPLICABLE** — pipeline explicitly minimises predicted toxicity; no toxin-optimisation objective |
| Dangerous pathogen targeting | **NOT APPLICABLE** — no pathogen-specific sequences in this pipeline |
| Mass production of harmful agents | **NOT APPLICABLE** — dry-lab output only; no synthesis ordered without human review |

The generator is a conservative substitution explorer over physicochemically balanced
AMP-like templates. It does not optimise for any harmful objective.

---

## 4. Data and Reproducibility Risks

| Risk | Status | Mitigation |
|------|--------|-----------|
| Input data not versioned | **MITIGATED** | SHA-256 of all input files in `outputs/phase3_manifest.json` |
| Config changed after scoring | **MITIGATED** | Config hash in manifest; `configs/phase3.yaml` versioned in git |
| Random seed not fixed | **MITIGATED** | rng_seed=2024 hardcoded in `make generate`; documented in Makefile |
| Pipeline version not tracked | **MITIGATED** | `pipeline_version` field in manifest and evidence certificates |

---

## 5. Required Human Review Gates

The following gates must be completed by qualified humans before any candidate is
synthesised or sent to a lab partner:

| Gate | Required reviewer | Status |
|------|-------------------|--------|
| Peptide sequence review | Peptide chemist or medicinal chemist | **PENDING** |
| Target organism selection | Microbiologist | **PENDING** |
| Safety profiling plan | Safety officer or toxicologist | **PENDING** |
| Synthesis feasibility | Peptide synthesis specialist | **PENDING** |
| Batch release approval | PI or qualified scientific director | **PENDING** |
| CRO/lab partner selection | PI + legal/compliance | **PENDING** |

**No candidate may be synthesised or sent to any external party without all gates above
being cleared. This is non-negotiable.**

---

## 6. Computational Disclaimer (required)

All scores produced by the OpenAMP Foundry pipeline are transparent baseline heuristics
computed from physicochemical properties. They are NOT validated biological predictors.
No antimicrobial activity has been demonstrated in vitro or in vivo. These candidates
are nominated for possible future expert review and assay only.

Do not describe any candidate as an "antibiotic," "drug," "cure," "therapeutic," "safe,"
or "effective" without experimental evidence.

The lab is the judge.
90 changes: 90 additions & 0 deletions docs/SELECTION_RULE.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,90 @@
# Pre-Registered Candidate Selection Rule

**Version:** 1.0
**Locked before:** Phase 3 candidate generation run

This document pre-registers the selection rule used to nominate candidates from the
Phase 3 generation batch. The rule and all thresholds are locked in advance and may not
be changed after the generation run produces scores.

---

## Computational Disclaimer

All scores produced by this foundry are heuristic physicochemical proxies.
They are **not validated biological predictors**.
No antimicrobial activity has been demonstrated in vitro or in vivo.
These rules select candidates for possible future expert review and assay — they do not
predict whether any peptide will be active, safe, or useful.

---

## Locked Scoring Weights

As defined in `configs/phase3.yaml` (separate from the benchmark config `pipeline.yaml`):

| Dimension | Weight |
|-----------------|--------|
| activity | 0.35 |
| safety | 0.30 |
| synthesis | 0.20 |
| novelty | 0.15 |

Ensemble score = weighted sum of the four dimensions.

---

## Pass/Fail Criteria (locked)

A candidate passes all of the following gates, evaluated in order:

| Gate | Threshold | Rationale |
|-----------------------|------------|----------------------------------------------------|
| Length | 8–35 aa | Pipeline filter; shorter too short, longer harder to synthesize |
| Canonical AAs only | No non-standard residues | Synthesis feasibility |
| Ensemble score | ≥ 0.50 | Minimum quality bar across all dimensions |
| Safety score | ≥ 0.60 | max_safety_risk = 0.40; hemolysis/toxicity proxies |
| Novelty score | ≥ 0.05 | Minimum 1-mutation difference from any reference seed |
| Activity score | ≥ 0.40 | Must meet minimum predicted activity signal |

Note on novelty threshold: Phase 3 candidates are generated by 1–3 conservative
substitutions from AMP-like template seeds. A min_novelty of 0.05 ensures every
selected candidate differs by at least one amino acid from the nearest seed, while
allowing the generator to fully explore the near-seed neighborhood. A higher threshold
(e.g. 0.20, used in pipeline.yaml for benchmark purposes) would exclude most or all
near-seed variants and is not appropriate here.

---

## Diversity Selection

After gate filtering, candidates are ranked by ensemble score (descending).
Greedy diverse selection is applied: each candidate added to the batch must have
pairwise normalized Levenshtein similarity < 0.80 with all previously selected candidates.

Target batch size: 50–100 candidates.

---

## What Counts as a Passing Batch

The Phase 3 batch **passes** if:
- ≥ 50 candidates clear all gates above
- ≥ 5 distinct template seeds are represented (diversity of origin)
- No selected candidate has ensemble score < 0.50
- No selected candidate has safety score < 0.60

If fewer than 50 candidates pass, the generation parameters (n_double, n_charge_enhance)
may be increased and the batch re-run. This will be documented transparently.

---

## What This Does NOT Claim

- It does not claim any candidate will show antimicrobial activity.
- It does not claim any candidate is safe for any use.
- It does not constitute a drug development programme.
- It does not replace expert biological evaluation.

The lab is the judge. This rule only determines which computational candidates are
nominated for possible future evaluation.
6 changes: 6 additions & 0 deletions examples/benchmark/active_labels.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
id,sequence,source
BM-POS-001,KWKLFKKIGAVLKVL,benchmark_positive
BM-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,benchmark_positive
BM-POS-003,GLFDIVKKVVGALGSL,benchmark_positive
BM-POS-004,RRWWLRVIAGLLKKVLS,benchmark_positive
BM-POS-005,KLLLKWLKKVLKA,benchmark_positive
14 changes: 14 additions & 0 deletions examples/benchmark/cluster_split_pool.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,14 @@
id,sequence,source
CS-POS-001,KWKLFKKIGAVLKFL,cluster_split
CS-POS-002,KWKLFKRIGAVLKVL,cluster_split
CS-POS-003,GLFDIVKKVVGALGAL,cluster_split
CS-NEG-001,AAAAAAAAAAAA,cluster_split
CS-NEG-002,DEDEDEDEDEDE,cluster_split
CS-NEG-003,GGGGGGGGGGGG,cluster_split
CS-NEG-004,EEEEEEEEEEEE,cluster_split
CS-NEG-005,SSSSSSSSSSSS,cluster_split
CS-NEG-006,PPPPPPPPPPPP,cluster_split
CS-NEG-007,TTTTTTTTTTTT,cluster_split
CS-NEG-008,NNNNNNNNNNNN,cluster_split
CS-NEG-009,QQQQQQQQQQQQ,cluster_split
CS-NEG-010,LLLLLLLLLLL,cluster_split
4 changes: 4 additions & 0 deletions examples/benchmark/cluster_split_refs.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,4 @@
id,sequence,source
CSREF-001,KWKLFKKIGAVLKVL,reference
CSREF-002,GIGKFLHSAKKFGKAFVGEIMNS,reference
CSREF-003,GLFDIVKKVVGALGSL,reference
21 changes: 21 additions & 0 deletions examples/benchmark/mixed_candidates.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
id,sequence,source
BM-POS-001,KWKLFKKIGAVLKVL,benchmark_positive
BM-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,benchmark_positive
BM-POS-003,GLFDIVKKVVGALGSL,benchmark_positive
BM-POS-004,RRWWLRVIAGLLKKVLS,benchmark_positive
BM-POS-005,KLLLKWLKKVLKA,benchmark_positive
BM-NEG-001,AAAAAAAAAAAA,benchmark_negative
BM-NEG-002,GGGGGGGGGGGG,benchmark_negative
BM-NEG-003,DEDEDEDEDEDE,benchmark_negative
BM-NEG-004,SSSSSSSSSSS,benchmark_negative
BM-NEG-005,EEEEEEEEEEEE,benchmark_negative
BM-NEG-006,PPPPPPPPPPPP,benchmark_negative
BM-NEG-007,NNNNNNNNNNN,benchmark_negative
BM-NEG-008,QQQQQQQQQQQQ,benchmark_negative
BM-NEG-009,TTTTTTTTTTTT,benchmark_negative
BM-NEG-010,AAGGAAGGAAGG,benchmark_negative
BM-NEG-011,SSTTSSTTSSTT,benchmark_negative
BM-NEG-012,EDEDEDEDEDED,benchmark_negative
BM-NEG-013,GGGAAAGGGAAA,benchmark_negative
BM-NEG-014,QQNNQQNNQQNN,benchmark_negative
BM-NEG-015,SSSSTTTTSSSS,benchmark_negative
12 changes: 12 additions & 0 deletions examples/benchmark/novelty_pressure_pool.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
id,sequence,source
NOV-DUP-001,KWKLFKKIGAVLKVL,novelty_pressure
NOV-DUP-002,KWKLFKKIGAVLKFL,novelty_pressure
NOV-DUP-003,KWKLFKRIGAVLKVL,novelty_pressure
NOV-NEW-001,RRLKKVLGAVLKVLK,novelty_pressure
NOV-NEW-002,WKWLKKIRGKLLKV,novelty_pressure
NOV-NEW-003,FLKHFKIKAVLKRLK,novelty_pressure
NOV-NEG-001,AAAAAAAAAAAA,novelty_pressure
NOV-NEG-002,DEDEDEDEDEDE,novelty_pressure
NOV-NEG-003,GGGGGGGGGGGG,novelty_pressure
NOV-NEG-004,EEEEEEEEEEEE,novelty_pressure
NOV-NEG-005,SSSSSSSSSSSS,novelty_pressure
4 changes: 4 additions & 0 deletions examples/benchmark/robustness_positives.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,4 @@
id,sequence,source
ROB-POS-001,KWKLFKKIGAVLKVL,robustness
ROB-POS-002,GIGKFLHSAKKFGKAFVGEIMNS,robustness
ROB-POS-003,RRWQWRMKKLG,robustness
Loading
Loading