diff --git a/Makefile b/Makefile index 5f6de500..9cb0e1fd 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: demo test lint clean bench-leakage bench-baseline bench-hidden-active generate phase3 +.PHONY: demo test lint clean bench-leakage bench-baseline bench-hidden-active generate phase3 pilot PYTHON := $(shell [ -f .venv/bin/python ] && echo .venv/bin/python || echo python3) PYTEST := $(shell [ -f .venv/bin/pytest ] && echo .venv/bin/pytest || echo pytest) @@ -61,5 +61,12 @@ phase3: generate --out-json outputs/phase3_batch_pack.json \ --out-md outputs/phase3_batch_pack.md +pilot: phase3 + PYTHONPATH=src $(PYTHON) -m openamp_foundry.cli pilot-panel \ + --ranked outputs/phase3_ranked.jsonl \ + --n 20 \ + --out-csv outputs/pilot_panel.csv \ + --out-md outputs/pilot_panel.md + clean: rm -rf outputs/*.jsonl outputs/*.md outputs/*.json outputs/evidence outputs/phase3_evidence .pytest_cache .ruff_cache diff --git a/src/openamp_foundry/cli.py b/src/openamp_foundry/cli.py index e2599712..5ae67472 100644 --- a/src/openamp_foundry/cli.py +++ b/src/openamp_foundry/cli.py @@ -93,6 +93,36 @@ def build_parser() -> argparse.ArgumentParser: help="RNG seed for reproducibility (default: 2024)", ) + pilot = sub.add_parser( + "pilot-panel", + help=( + "Select a first-synthesis-wave pilot panel from a ranked JSONL file. " + "Picks n candidates (default 20) maximising ensemble score, minimising scorer " + "disagreement, and ensuring at least one representative per seed template." + ), + ) + pilot.add_argument( + "--ranked", + required=True, + help="Ranked JSONL file (output of the 'rank' command).", + ) + pilot.add_argument( + "--n", + type=int, + default=20, + help="Panel size (default: 20).", + ) + pilot.add_argument( + "--out-csv", + required=True, + help="Output CSV path (synthesis-ready format).", + ) + pilot.add_argument( + "--out-md", + required=False, + help="Optional output path for human-readable markdown panel.", + ) + batch_pack = sub.add_parser( "batch-pack", help=( @@ -154,6 +184,9 @@ def main(argv: list[str] | None = None) -> int: if args.command == "generate-batch": return _run_generate_batch(args) + if args.command == "pilot-panel": + return _run_pilot_panel(args) + if args.command == "batch-pack": return _run_batch_pack(args) @@ -207,6 +240,54 @@ def _run_bench(args: argparse.Namespace) -> int: return 2 +def _run_pilot_panel(args: argparse.Namespace) -> int: + import json as _json + from datetime import datetime, timezone + + from openamp_foundry.reports.pilot_panel import write_pilot_csv, write_pilot_markdown + from openamp_foundry.selection.pilot import select_pilot_panel + + ranked_path = Path(args.ranked) + if not ranked_path.exists(): + print(_json.dumps({"status": "error", "message": f"File not found: {args.ranked}"})) + return 1 + + candidates = [] + with open(ranked_path, encoding="utf-8") as f: + for line in f: + line = line.strip() + if line: + row = _json.loads(line) + if row.get("selected"): + candidates.append(row) + + panel = select_pilot_panel(candidates, n=args.n) + generated_at = datetime.now(timezone.utc).isoformat() + + write_pilot_csv(panel, args.out_csv) + if args.out_md: + write_pilot_markdown(panel, args.out_md, generated_at=generated_at) + + seeds = sorted({c.get("seed", "") for c in panel}) + n_consensus = sum( + 1 for c in panel if c.get("scores", {}).get("disagreement", 1.0) < 0.20 + ) + print(_json.dumps({ + "status": "ok", + "n_nominees": len(candidates), + "n_panel": len(panel), + "seeds_represented": seeds, + "n_dual_scorer_consensus": n_consensus, + "out_csv": args.out_csv, + "out_md": args.out_md, + "disclaimer": ( + "No antimicrobial activity has been demonstrated. " + "Human expert review required before synthesis." + ), + }, indent=2)) + return 0 + + def _run_batch_pack(args: argparse.Namespace) -> int: from openamp_foundry.reports.batch_pack import generate_batch_pack, write_batch_pack_markdown from openamp_foundry.utils.io import write_json diff --git a/src/openamp_foundry/reports/pilot_panel.py b/src/openamp_foundry/reports/pilot_panel.py new file mode 100644 index 00000000..81106fe9 --- /dev/null +++ b/src/openamp_foundry/reports/pilot_panel.py @@ -0,0 +1,122 @@ +from __future__ import annotations + +import csv +from pathlib import Path + +_CSV_FIELDS = [ + "pilot_rank", + "candidate_id", + "sequence", + "length", + "seed", + "ensemble", + "activity", + "boman_activity", + "disagreement", + "safety", + "synthesis", + "novelty", + "pilot_priority", +] + +_DISCLAIMER = ( + "All scores are transparent physicochemical heuristics. " + "No antimicrobial activity has been demonstrated in vitro or in vivo. " + "These candidates are hypotheses for possible future expert review and assay only. " + "Human expert review and institutional biosafety sign-off are required before synthesis." +) + + +def _row(c: dict) -> dict: + scores = c.get("scores", {}) + return { + "pilot_rank": c.get("pilot_rank", ""), + "candidate_id": c.get("candidate_id", ""), + "sequence": c.get("sequence", ""), + "length": len(c.get("sequence", "")), + "seed": c.get("seed", ""), + "ensemble": round(scores.get("ensemble", 0.0), 4), + "activity": round(scores.get("activity", 0.0), 4), + "boman_activity": round(scores.get("boman_activity", 0.0), 4), + "disagreement": round(scores.get("disagreement", 0.0), 4), + "safety": round(scores.get("safety", 0.0), 4), + "synthesis": round(scores.get("synthesis", 0.0), 4), + "novelty": round(scores.get("novelty", 0.0), 4), + "pilot_priority": round(c.get("pilot_priority", 0.0), 4), + } + + +def write_pilot_csv(panel: list[dict], path: str | Path) -> None: + out = Path(path) + out.parent.mkdir(parents=True, exist_ok=True) + with open(out, "w", newline="", encoding="utf-8") as f: + writer = csv.DictWriter(f, fieldnames=_CSV_FIELDS) + writer.writeheader() + for c in panel: + writer.writerow(_row(c)) + + +def write_pilot_markdown(panel: list[dict], path: str | Path, generated_at: str = "") -> None: + out = Path(path) + out.parent.mkdir(parents=True, exist_ok=True) + + seeds = sorted({c.get("seed", "") for c in panel}) + disagreements = [c.get("scores", {}).get("disagreement", 0.0) for c in panel] + n_consensus = sum(1 for d in disagreements if d < 0.20) + + lines = [ + "# OpenAMP Foundry — Pilot Synthesis Panel", + "", + "> **Disclaimer:** " + _DISCLAIMER, + "", + ] + if generated_at: + lines += [f"Generated: {generated_at}", ""] + + lines += [ + f"**Panel size:** {len(panel)} candidates ", + f"**Seeds represented:** {len(seeds)} ({', '.join(seeds)}) ", + f"**Dual-scorer consensus (disagreement < 0.20):** {n_consensus}/{len(panel)} ", + "", + "## Selection method", + "", + "Priority score = `ensemble − 0.3 × disagreement` ", + "Rules: one representative per seed (highest priority), then remaining slots filled by priority rank.", + "", + "## Candidates", + "", + "| # | ID | Sequence | Len | Seed | Ensemble | Activity | Boman | Disagree | Safety | Synth |", + "|--:|---|---|--:|---|---:|---:|---:|---:|---:|---:|", + ] + + for c in panel: + r = _row(c) + lines.append( + f"| {r['pilot_rank']} | {r['candidate_id']} | `{r['sequence']}` | {r['length']} " + f"| {r['seed']} | {r['ensemble']} | {r['activity']} | {r['boman_activity']} " + f"| {r['disagreement']} | {r['safety']} | {r['synthesis']} |" + ) + + lines += [ + "", + "## Next steps", + "", + "1. Human expert reviews this table and removes any sequences with known issues", + "2. Institutional biosafety committee sign-off before ordering synthesis", + "3. Synthesise selected peptides (SPPS, >95% purity, TFA-free if possible)", + "4. MIC assay against target organism(s) in triplicate", + "5. Hemolysis assay at 2× MIC against human erythrocytes", + "6. Serum stability at 37°C, 0/2/4/8h timepoints", + "7. Feed results back into pipeline for second-wave nomination", + "", + "## Reproducibility", + "", + "```bash", + "make phase3 # regenerates the 89 nominees", + "make pilot # selects this panel from the nominees", + "```", + "", + f"Full methodology: `docs/NOMINATION_REPORT.md`", + ] + + out.write_text("\n".join(lines) + "\n", encoding="utf-8") diff --git a/src/openamp_foundry/selection/pilot.py b/src/openamp_foundry/selection/pilot.py new file mode 100644 index 00000000..6ffa43af --- /dev/null +++ b/src/openamp_foundry/selection/pilot.py @@ -0,0 +1,70 @@ +from __future__ import annotations + + +def _seed_from_source(source: str) -> str: + """Extract seed ID from a source string like 'template_mutation_from_SEED-003'.""" + prefix = "template_mutation_from_" + if source.startswith(prefix): + return source[len(prefix):] + return source + + +def _pilot_priority(scores: dict) -> float: + """Higher is better: reward ensemble score, penalise scorer disagreement.""" + ensemble = scores.get("ensemble", 0.0) + disagreement = scores.get("disagreement", 0.5) + return round(ensemble - 0.3 * disagreement, 6) + + +def select_pilot_panel( + candidates: list[dict], + n: int = 20, +) -> list[dict]: + """Select an n-candidate first-synthesis-wave panel from a list of nominees. + + Selection rules (in order): + 1. One representative per distinct seed, chosen by pilot_priority. + 2. Remaining slots filled from the rest of the pool, highest priority first. + 3. If n < number of seeds, take the top-n seeds by priority. + + Each returned dict gets an added `pilot_priority` and `seed` key. + """ + if not candidates: + return [] + + enriched = [] + for c in candidates: + ec = dict(c) + ec["seed"] = _seed_from_source(c.get("source", "")) + ec["pilot_priority"] = _pilot_priority(c.get("scores", {})) + enriched.append(ec) + + enriched.sort(key=lambda x: x["pilot_priority"], reverse=True) + + selected: list[dict] = [] + seen_seeds: set[str] = set() + remainder: list[dict] = [] + + # Phase 1 — one per seed (best by priority) + for c in enriched: + seed = c["seed"] + if seed not in seen_seeds: + selected.append(c) + seen_seeds.add(seed) + else: + remainder.append(c) + if len(selected) == n: + break + + # Phase 2 — fill remaining slots + for c in remainder: + if len(selected) >= n: + break + selected.append(c) + + # Final sort: highest priority first + selected.sort(key=lambda x: x["pilot_priority"], reverse=True) + for rank, c in enumerate(selected, start=1): + c["pilot_rank"] = rank + + return selected diff --git a/tests/test_pilot_panel.py b/tests/test_pilot_panel.py new file mode 100644 index 00000000..ebad0d8b --- /dev/null +++ b/tests/test_pilot_panel.py @@ -0,0 +1,234 @@ +"""Tests for pilot panel selection and report formatting.""" +from __future__ import annotations + +import csv +import json + +import pytest + +from openamp_foundry.selection.pilot import _pilot_priority, _seed_from_source, select_pilot_panel +from openamp_foundry.reports.pilot_panel import write_pilot_csv, write_pilot_markdown + + +def _make( + cid: str, + seq: str = "KWKLFKKIGAVLKVL", + seed: str = "SEED-001", + ensemble: float = 0.70, + activity: float = 0.72, + boman: float = 0.65, + disagreement: float | None = None, + safety: float = 0.90, + synthesis: float = 0.80, + novelty: float = 0.20, +) -> dict: + if disagreement is None: + disagreement = round(abs(activity - boman), 4) + return { + "candidate_id": cid, + "sequence": seq, + "source": f"template_mutation_from_{seed}", + "selected": True, + "scores": { + "ensemble": ensemble, + "activity": activity, + "boman_activity": boman, + "disagreement": disagreement, + "safety": safety, + "synthesis": synthesis, + "novelty": novelty, + }, + "features": {"length": len(seq)}, + "nearest_reference": None, + "selection_reason": [], + "known_failure_modes": [], + } + + +FIVE_SEEDS = [ + _make("C1", seed="SEED-001", ensemble=0.80, disagreement=0.05), + _make("C2", seed="SEED-002", ensemble=0.75, disagreement=0.10), + _make("C3", seed="SEED-003", ensemble=0.70, disagreement=0.15), + _make("C4", seed="SEED-004", ensemble=0.68, disagreement=0.20), + _make("C5", seed="SEED-005", ensemble=0.65, disagreement=0.25), +] + +EXTRA_SEED1 = [ + _make("C6", seed="SEED-001", ensemble=0.60, disagreement=0.05), + _make("C7", seed="SEED-001", ensemble=0.55, disagreement=0.08), +] + + +class TestSeedFromSource: + def test_standard_prefix(self): + assert _seed_from_source("template_mutation_from_SEED-003") == "SEED-003" + + def test_no_prefix_returns_source_as_is(self): + assert _seed_from_source("SEED-001") == "SEED-001" + + def test_empty_string(self): + assert _seed_from_source("") == "" + + +class TestPilotPriority: + def test_higher_ensemble_higher_priority(self): + p1 = _pilot_priority({"ensemble": 0.80, "disagreement": 0.10}) + p2 = _pilot_priority({"ensemble": 0.70, "disagreement": 0.10}) + assert p1 > p2 + + def test_lower_disagreement_higher_priority(self): + p1 = _pilot_priority({"ensemble": 0.70, "disagreement": 0.05}) + p2 = _pilot_priority({"ensemble": 0.70, "disagreement": 0.30}) + assert p1 > p2 + + def test_missing_disagreement_defaults_to_0_5(self): + p = _pilot_priority({"ensemble": 0.70}) + assert p == pytest.approx(0.70 - 0.3 * 0.5, abs=1e-6) + + +class TestSelectPilotPanel: + def test_empty_input_returns_empty(self): + assert select_pilot_panel([]) == [] + + def test_panel_size_capped_at_n(self): + big = [_make(f"C{i}", seed=f"SEED-{i:03d}") for i in range(50)] + panel = select_pilot_panel(big, n=20) + assert len(panel) == 20 + + def test_panel_size_when_fewer_candidates_than_n(self): + panel = select_pilot_panel(FIVE_SEEDS, n=20) + assert len(panel) == 5 + + def test_each_seed_represented_at_most_once_in_first_pass(self): + candidates = FIVE_SEEDS + EXTRA_SEED1 + panel = select_pilot_panel(candidates, n=5) + seeds = [c["seed"] for c in panel] + # Each of SEED-001..005 appears once in a 5-slot panel + assert len(set(seeds)) == 5 + + def test_extra_slots_filled_from_remainder(self): + candidates = FIVE_SEEDS + EXTRA_SEED1 + panel = select_pilot_panel(candidates, n=7) + assert len(panel) == 7 + # SEED-001 should appear twice (C1 and one of C6/C7) + seed_counts = {} + for c in panel: + seed_counts[c["seed"]] = seed_counts.get(c["seed"], 0) + 1 + assert seed_counts["SEED-001"] == 3 # C1, C6, C7 all in when n=7 + + def test_panel_sorted_by_priority_descending(self): + candidates = FIVE_SEEDS + EXTRA_SEED1 + panel = select_pilot_panel(candidates, n=7) + priorities = [c["pilot_priority"] for c in panel] + assert priorities == sorted(priorities, reverse=True) + + def test_pilot_rank_assigned(self): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + ranks = [c["pilot_rank"] for c in panel] + assert ranks == list(range(1, 6)) + + def test_seed_field_added(self): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + for c in panel: + assert "seed" in c + assert c["seed"].startswith("SEED-") + + def test_best_per_seed_selected_first(self): + # C1 (ensemble=0.80) should beat C6 (0.60) for SEED-001 slot + candidates = EXTRA_SEED1 + FIVE_SEEDS # order shouldn't matter + panel = select_pilot_panel(candidates, n=5) + seed1_entries = [c for c in panel if c["seed"] == "SEED-001"] + assert len(seed1_entries) == 1 + assert seed1_entries[0]["candidate_id"] == "C1" + + def test_high_disagreement_penalised(self): + low_disagree = _make("LOW", seed="SEED-X", ensemble=0.72, disagreement=0.02) + high_disagree = _make("HIGH", seed="SEED-Y", ensemble=0.75, disagreement=0.50) + panel = select_pilot_panel([low_disagree, high_disagree], n=2) + assert panel[0]["candidate_id"] == "LOW" + + def test_n_equals_1(self): + panel = select_pilot_panel(FIVE_SEEDS, n=1) + assert len(panel) == 1 + assert panel[0]["pilot_rank"] == 1 + + +class TestWritePilotCsv: + def test_creates_file(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.csv" + write_pilot_csv(panel, out) + assert out.exists() + + def test_correct_columns(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.csv" + write_pilot_csv(panel, out) + with open(out, newline="") as f: + reader = csv.DictReader(f) + fields = reader.fieldnames + expected = {"pilot_rank", "candidate_id", "sequence", "length", "seed", + "ensemble", "activity", "boman_activity", "disagreement", + "safety", "synthesis", "novelty", "pilot_priority"} + assert set(fields) == expected + + def test_row_count_matches_panel(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.csv" + write_pilot_csv(panel, out) + with open(out, newline="") as f: + rows = list(csv.DictReader(f)) + assert len(rows) == 5 + + def test_length_column_is_sequence_length(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.csv" + write_pilot_csv(panel, out) + with open(out, newline="") as f: + for row in csv.DictReader(f): + assert int(row["length"]) == len(row["sequence"]) + + +class TestWritePilotMarkdown: + def test_creates_file(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.md" + write_pilot_markdown(panel, out) + assert out.exists() + + def test_contains_disclaimer(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.md" + write_pilot_markdown(panel, out) + content = out.read_text() + assert "Disclaimer" in content + assert "demonstrated" in content + + def test_contains_all_candidate_ids(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.md" + write_pilot_markdown(panel, out) + content = out.read_text() + for c in panel: + assert c["candidate_id"] in content + + def test_contains_next_steps(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.md" + write_pilot_markdown(panel, out) + content = out.read_text() + assert "Next steps" in content + assert "MIC" in content + + def test_contains_reproducibility_block(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.md" + write_pilot_markdown(panel, out) + content = out.read_text() + assert "make pilot" in content + + def test_generated_at_included_when_provided(self, tmp_path): + panel = select_pilot_panel(FIVE_SEEDS, n=5) + out = tmp_path / "pilot.md" + write_pilot_markdown(panel, out, generated_at="2026-06-27T00:00:00Z") + assert "2026-06-27" in out.read_text()