Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 8 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
.PHONY: demo test lint clean bench-leakage bench-baseline bench-hidden-active generate phase3
.PHONY: demo test lint clean bench-leakage bench-baseline bench-hidden-active generate phase3 pilot

PYTHON := $(shell [ -f .venv/bin/python ] && echo .venv/bin/python || echo python3)
PYTEST := $(shell [ -f .venv/bin/pytest ] && echo .venv/bin/pytest || echo pytest)
Expand Down Expand Up @@ -61,5 +61,12 @@ phase3: generate
--out-json outputs/phase3_batch_pack.json \
--out-md outputs/phase3_batch_pack.md

pilot: phase3
PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli pilot-panel \
--ranked outputs/phase3_ranked.jsonl \
--n 20 \
--out-csv outputs/pilot_panel.csv \
--out-md outputs/pilot_panel.md

clean:
rm -rf outputs/*.jsonl outputs/*.md outputs/*.json outputs/evidence outputs/phase3_evidence .pytest_cache .ruff_cache
81 changes: 81 additions & 0 deletions src/openamp_foundry/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -93,6 +93,36 @@ def build_parser() -> argparse.ArgumentParser:
help="RNG seed for reproducibility (default: 2024)",
)

pilot = sub.add_parser(
"pilot-panel",
help=(
"Select a first-synthesis-wave pilot panel from a ranked JSONL file. "
"Picks n candidates (default 20) maximising ensemble score, minimising scorer "
"disagreement, and ensuring at least one representative per seed template."
),
)
pilot.add_argument(
"--ranked",
required=True,
help="Ranked JSONL file (output of the 'rank' command).",
)
pilot.add_argument(
"--n",
type=int,
default=20,
help="Panel size (default: 20).",
)
pilot.add_argument(
"--out-csv",
required=True,
help="Output CSV path (synthesis-ready format).",
)
pilot.add_argument(
"--out-md",
required=False,
help="Optional output path for human-readable markdown panel.",
)

batch_pack = sub.add_parser(
"batch-pack",
help=(
Expand Down Expand Up @@ -154,6 +184,9 @@ def main(argv: list[str] | None = None) -> int:
if args.command == "generate-batch":
return _run_generate_batch(args)

if args.command == "pilot-panel":
return _run_pilot_panel(args)

if args.command == "batch-pack":
return _run_batch_pack(args)

Expand Down Expand Up @@ -207,6 +240,54 @@ def _run_bench(args: argparse.Namespace) -> int:
return 2


def _run_pilot_panel(args: argparse.Namespace) -> int:
import json as _json
from datetime import datetime, timezone

from openamp_foundry.reports.pilot_panel import write_pilot_csv, write_pilot_markdown
from openamp_foundry.selection.pilot import select_pilot_panel

ranked_path = Path(args.ranked)
if not ranked_path.exists():
print(_json.dumps({"status": "error", "message": f"File not found: {args.ranked}"}))
return 1

candidates = []
with open(ranked_path, encoding="utf-8") as f:
for line in f:
line = line.strip()
if line:
row = _json.loads(line)
if row.get("selected"):
candidates.append(row)

panel = select_pilot_panel(candidates, n=args.n)
generated_at = datetime.now(timezone.utc).isoformat()

write_pilot_csv(panel, args.out_csv)
if args.out_md:
write_pilot_markdown(panel, args.out_md, generated_at=generated_at)

seeds = sorted({c.get("seed", "") for c in panel})
n_consensus = sum(
1 for c in panel if c.get("scores", {}).get("disagreement", 1.0) < 0.20
)
print(_json.dumps({
"status": "ok",
"n_nominees": len(candidates),
"n_panel": len(panel),
"seeds_represented": seeds,
"n_dual_scorer_consensus": n_consensus,
"out_csv": args.out_csv,
"out_md": args.out_md,
"disclaimer": (
"No antimicrobial activity has been demonstrated. "
"Human expert review required before synthesis."
),
}, indent=2))
return 0


def _run_batch_pack(args: argparse.Namespace) -> int:
from openamp_foundry.reports.batch_pack import generate_batch_pack, write_batch_pack_markdown
from openamp_foundry.utils.io import write_json
Expand Down
122 changes: 122 additions & 0 deletions src/openamp_foundry/reports/pilot_panel.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,122 @@
from __future__ import annotations

import csv
from pathlib import Path

_CSV_FIELDS = [
"pilot_rank",
"candidate_id",
"sequence",
"length",
"seed",
"ensemble",
"activity",
"boman_activity",
"disagreement",
"safety",
"synthesis",
"novelty",
"pilot_priority",
]

_DISCLAIMER = (
"All scores are transparent physicochemical heuristics. "
"No antimicrobial activity has been demonstrated in vitro or in vivo. "
"These candidates are hypotheses for possible future expert review and assay only. "
"Human expert review and institutional biosafety sign-off are required before synthesis."
)


def _row(c: dict) -> dict:
scores = c.get("scores", {})
return {
"pilot_rank": c.get("pilot_rank", ""),
"candidate_id": c.get("candidate_id", ""),
"sequence": c.get("sequence", ""),
"length": len(c.get("sequence", "")),
"seed": c.get("seed", ""),
"ensemble": round(scores.get("ensemble", 0.0), 4),
"activity": round(scores.get("activity", 0.0), 4),
"boman_activity": round(scores.get("boman_activity", 0.0), 4),
"disagreement": round(scores.get("disagreement", 0.0), 4),
"safety": round(scores.get("safety", 0.0), 4),
"synthesis": round(scores.get("synthesis", 0.0), 4),
"novelty": round(scores.get("novelty", 0.0), 4),
"pilot_priority": round(c.get("pilot_priority", 0.0), 4),
}


def write_pilot_csv(panel: list[dict], path: str | Path) -> None:
out = Path(path)
out.parent.mkdir(parents=True, exist_ok=True)
with open(out, "w", newline="", encoding="utf-8") as f:
writer = csv.DictWriter(f, fieldnames=_CSV_FIELDS)
writer.writeheader()
for c in panel:
writer.writerow(_row(c))


def write_pilot_markdown(panel: list[dict], path: str | Path, generated_at: str = "") -> None:
out = Path(path)
out.parent.mkdir(parents=True, exist_ok=True)

seeds = sorted({c.get("seed", "") for c in panel})
disagreements = [c.get("scores", {}).get("disagreement", 0.0) for c in panel]
n_consensus = sum(1 for d in disagreements if d < 0.20)

lines = [
"# OpenAMP Foundry — Pilot Synthesis Panel",
"",
"> **Disclaimer:** " + _DISCLAIMER,
"",
]
if generated_at:
lines += [f"Generated: {generated_at}", ""]

lines += [
f"**Panel size:** {len(panel)} candidates ",
f"**Seeds represented:** {len(seeds)} ({', '.join(seeds)}) ",
f"**Dual-scorer consensus (disagreement < 0.20):** {n_consensus}/{len(panel)} ",
"",
"## Selection method",
"",
"Priority score = `ensemble − 0.3 × disagreement` ",
"Rules: one representative per seed (highest priority), then remaining slots filled by priority rank.",
"",
"## Candidates",
"",
"| # | ID | Sequence | Len | Seed | Ensemble | Activity | Boman | Disagree | Safety | Synth |",
"|--:|---|---|--:|---|---:|---:|---:|---:|---:|---:|",
]

for c in panel:
r = _row(c)
lines.append(
f"| {r['pilot_rank']} | {r['candidate_id']} | `{r['sequence']}` | {r['length']} "
f"| {r['seed']} | {r['ensemble']} | {r['activity']} | {r['boman_activity']} "
f"| {r['disagreement']} | {r['safety']} | {r['synthesis']} |"
)

lines += [
"",
"## Next steps",
"",
"1. Human expert reviews this table and removes any sequences with known issues",
"2. Institutional biosafety committee sign-off before ordering synthesis",
"3. Synthesise selected peptides (SPPS, >95% purity, TFA-free if possible)",
"4. MIC assay against target organism(s) in triplicate",
"5. Hemolysis assay at 2× MIC against human erythrocytes",
"6. Serum stability at 37°C, 0/2/4/8h timepoints",
"7. Feed results back into pipeline for second-wave nomination",
"",
"## Reproducibility",
"",
"```bash",
"make phase3 # regenerates the 89 nominees",
"make pilot # selects this panel from the nominees",
"```",
"",
f"Full methodology: `docs/NOMINATION_REPORT.md`",
]

out.write_text("\n".join(lines) + "\n", encoding="utf-8")
70 changes: 70 additions & 0 deletions src/openamp_foundry/selection/pilot.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,70 @@
from __future__ import annotations


def _seed_from_source(source: str) -> str:
"""Extract seed ID from a source string like 'template_mutation_from_SEED-003'."""
prefix = "template_mutation_from_"
if source.startswith(prefix):
return source[len(prefix):]
return source


def _pilot_priority(scores: dict) -> float:
"""Higher is better: reward ensemble score, penalise scorer disagreement."""
ensemble = scores.get("ensemble", 0.0)
disagreement = scores.get("disagreement", 0.5)
return round(ensemble - 0.3 * disagreement, 6)


def select_pilot_panel(
candidates: list[dict],
n: int = 20,
) -> list[dict]:
"""Select an n-candidate first-synthesis-wave panel from a list of nominees.

Selection rules (in order):
1. One representative per distinct seed, chosen by pilot_priority.
2. Remaining slots filled from the rest of the pool, highest priority first.
3. If n < number of seeds, take the top-n seeds by priority.

Each returned dict gets an added `pilot_priority` and `seed` key.
"""
if not candidates:
return []

enriched = []
for c in candidates:
ec = dict(c)
ec["seed"] = _seed_from_source(c.get("source", ""))
ec["pilot_priority"] = _pilot_priority(c.get("scores", {}))
enriched.append(ec)

enriched.sort(key=lambda x: x["pilot_priority"], reverse=True)

selected: list[dict] = []
seen_seeds: set[str] = set()
remainder: list[dict] = []

# Phase 1 — one per seed (best by priority)
for c in enriched:
seed = c["seed"]
if seed not in seen_seeds:
selected.append(c)
seen_seeds.add(seed)
else:
remainder.append(c)
if len(selected) == n:
break

# Phase 2 — fill remaining slots
for c in remainder:
if len(selected) >= n:
break
selected.append(c)

# Final sort: highest priority first
selected.sort(key=lambda x: x["pilot_priority"], reverse=True)
for rank, c in enumerate(selected, start=1):
c["pilot_rank"] = rank

return selected
Loading
Loading