forked from uc-derisilab/pairis
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathparams.template.yml
More file actions
106 lines (89 loc) · 4.23 KB
/
Copy pathparams.template.yml
File metadata and controls
106 lines (89 loc) · 4.23 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
# PAIRIS Pipeline Parameters Template
# Copy this file to your project directory and customize
# ===== Project Settings =====
project_name: "my_project"
peptide_fasta: "input/peptides.fasta"
bcr_dir: "input/bcrs"
chain_dir: "input/chains"
# ===== Analysis Parameters =====
# Window sizes for sliding window analysis
# Use [15, 25] for both 15mer and 25mer windows
# Use [15] for only 15mer windows
# Use null for full-length sequences (no sliding window)
window_sizes: [15, 25]
# Number of AlphaFold3 model seeds (diversity in predictions)
num_seeds: 100
# Number of diffusion samples per seed in AF3 folding
num_diffusion_samples: 5
# ===== Folding Backend =====
# Structure-prediction engine: "alphafold3" (default) or "esmfold2".
folding_backend: "alphafold3"
# ESMFold2 options (only used when folding_backend: "esmfold2"):
# esmfold2_model: "fast" -> ESMFold2-Fast, MSA-free. Skips Stage 1 entirely;
# AlphaFold3 is NOT needed. Set run_msa_generation: false.
# esmfold2_model: "full" -> uses MSAs. AF3's data pipeline is the only MSA source,
# so "full" STILL requires AlphaFold3 plus
# run_msa_generation: true (or an existing msa_index.json).
# esmfold2_model: "fast"
# esmfold2_conda_env: "esmfold2" # conda/mamba env for the ESMFold2 folding step
# esmfold2_hf_home: "/path/to/hf_cache" # HF model cache; pre-download on a node with network
# esmfold2_num_loops: 3
# esmfold2_num_sampling_steps: 50
#
# NOTE: ESMFold2 folds num_seeds seeds SERIALLY per complex (keeping the best),
# unlike AF3's single parallel pass. With a large num_seeds (e.g. the default 100),
# raise folding_time accordingly — or lower num_seeds — to avoid wall-clock timeouts.
# ===== Workflow Stages =====
# Control which stages of the pipeline to run
run_msa_generation: true
run_structure_prediction: true
run_rosetta_analysis: true
# ===== Output Directories =====
# Small outputs (MSAs, Rosetta results, reports)
outdir: "results"
# Large AF3 outputs (IMPORTANT: Use shared/group storage location for better quota)
# Replace "my_project" with your actual project name
af3_output_dir: "/path/to/shared_storage/my_project"
# ===== SLURM Settings =====
partition: "preempted"
# ===== Environment Settings =====
# Uncomment and override if your cluster uses different names
# af3_module: "alphafold/3.0.1-23-g792e61e"
# conda_env: "pairis"
# rosetta_conda_env: "rosetta"
# ===== Local Execution Settings =====
# Required for -profile local
# Apptainer/Singularity AF3 container paths
# af3_sif: "/path/to/alphafold.sif"
# af3_model_dir: "/path/to/af3/params"
# af3_db_dir: "/path/to/af3/public_databases"
# GPU configuration
# local_num_gpus: 4
# local_max_parallel_folding: 4
# local_max_parallel_cpu: 8
# Python env for the general steps (input generation, MSA extraction, grouping,
# collation) — a uv venv:
# local_venv: "/path/to/pairis-venv"
#
# NOTE: the local profile is HYBRID. Only the general steps above use local_venv.
# The Rosetta and ESMFold2 folding steps activate *mamba/conda* envs
# (rosetta_conda_env / esmfold2_conda_env) — the same envs as the SLURM profile —
# not uv venvs, because pyrosetta and the ESM fork install via conda/pip envs.
# ===== Setup Guide (local profile) =====
# 1. General venv (input generation, collation, etc.):
# uv venv /path/to/pairis-venv && source /path/to/pairis-venv/bin/activate
# uv pip install polars pandas biopython
#
# 2. Rosetta (only if run_rosetta_analysis: true) — a mamba env, not a venv:
# conda create -n rosetta && conda activate rosetta
# pip install polars pandas biopython pyrosetta # then rosetta_conda_env: "rosetta"
#
# 3. ESMFold2 (only if folding_backend: esmfold2) — see env/requirements-esmfold2.txt;
# set esmfold2_conda_env (default "esmfold2").
#
# 4. Run: nextflow run main.nf -params-file params.yml -profile local
# ===== Notes =====
# 1. Ensure input files exist before running
# 2. BCR FASTA files must have 'heavy'/'light' or 'HC'/'LC' in sequence names
# 3. Create af3_output_dir before running: mkdir -p /path/to/shared_storage/my_project
# 4. Run with: nextflow run main.nf -params-file params.yml