SurfDock / conf /params_example /example.yml
anzhi2710gmailcom's picture
Upload folder using huggingface_hub
10f2621 verified
Raw
History Blame Contribute Delete
5.06 kB
# This file is used to explain the meaning of the SurfDock parameters.
# If you want to learn more about the options available for these parameters, please refer ./utils/parsing.py file
# If the user wants to retrain the model, they can pick the right parameters to optimize in their own way
## General arguments
# CUDA optimization parameter for faster training
cudnn_benchmark: true
# use pin_memory or not in linux system
pin_memory: false
# restart dir for training, which storge the model and optimizer
restart_dir: ~/SurfDock/workdir/project_surface_V3_PDBBind_ema_model_pocket_8A
# restart learning rate
restart_lr: null
## dataset
# training data dir
data_dir: ~/PDBBIND/PDBBind_pocket_8A/
# cache path for training data, if you just use SurfDock, you can ignore this
cache_path: ~/PDBBIND/cache_Surface_PDBBIND_pocket_8A
# esm embedding path,if this is set then the LM embeddings at that path will be used for the receptor features
esm_embeddings_path: ~/PDBBIND/esm_embedding/esm_embedding_pocket_for_train/esm2_3billion_embeddings.pt
# dataset split file
split_test: ~/data/splits/timesplit_test
split_train: ~/data/splits/timesplit_no_lig_overlap_train
split_val: ~/data/splits/timesplit_no_lig_overlap_val
# use rmsd matching or not,use default value is fine
matching: true
# Differential evolution maxiter parameter in matching
matching_maxiter: 20
# the number of workers for dataloader
num_dataloader_workers: 1
# the number of complexes to inference in validation set
num_inference_complexes: 500
# the number of workers for training
num_workers: 1
## model
# if you want to use dynamic max cross, you can set this to true. this parameter can set a different max cross distance for each timestep
dynamic_max_cross: true
# scale the noise by sigma or not
scale_by_sigma: true
# Maximum sigma for rotational component
rot_sigma_max: 1.55
# Minimum sigma for rotational component
rot_sigma_min: 0.03
# Maximum sigma for torsional component
tor_sigma_max: 3.14
# Minimum sigma for torsional component
tor_sigma_min: 0.0314
# Maximum sigma for translational component
tr_sigma_max: 5.0
# Minimum sigma for translational component
tr_sigma_min: 0.1
# the weight of torsional component loss
tor_weight: 0.33
# the weight of translational component loss
tr_weight: 0.33
# the weight of rotational component loss
rot_weight: 0.33
# diffusion model type , default value is surface_score_model,when train a scoring model, you can use mdn_model
model_type: surface_score_model
# diffusion model version, use default value is 3
model_version: version3
# training epochs
n_epochs: 2000
# number of gaussians which used to MDN module for socring module
n_gaussians: 20
# use no batch norm or not
no_batch_norm: false
# use torsion or not ,default value is false
no_torsion: false
# the layer number of the model
num_conv_layers: 6
# the dimension of the scalar embedding in e3nn
ns: 48
# the dimension of the vector embedding in e3nn
nv: 10
# use the second order representation or not
use_second_order_repr: false
# embedding type
embedding_type: sinusoidal
# max number of neighbors for each atom in ligand graph
max_radius: 5.0
batch_size: 12 # batch size
# the radius of the receptor graph
receptor_radius: 15.0
# max number of neighbors for each C-alpha atom(residue graph)
c_alpha_max_neighbors: 24
# Maxximum inter-distance about different node types
cross_max_distance: 80
# cross distance embed dimension
cross_distance_embed_dim: 32
# intra-distance embed dimension
distance_embed_dim: 32
# Size of the embedding of the diffusion time
sigma_embed_dim: 32
# Parameter of the diffusion time embedding
embedding_scale: 1000
# dropout rate for dropout layer in diffusion module
dropout: 0.1
# the dropout rate for scoring module
mdn_dropout: 0.1
# use the ema model or not
use_ema: false
# exponential moving average rate
ema_rate: 0.999
# learning rate ,use default value is fine
lr: 0.001
# the learning rate scheduler
scheduler: plateau
# the weight decay factor, use default value is fine
w_decay: 0.0
# the patience of the learning rate scheduler
scheduler_patience: 50
# earlystop goal use max or min
inference_earlystop_goal: max
# inference earlystop metric
inference_earlystop_metric: valinf_rmsds_lt2
# denoise steps in inference stage
inference_steps: 20
# the early stop goal for training scoring module
mdn_early_stop_patience: 30
# remove hydrogen or not
remove_hs: true
# validation inference frequency,default is 20 epochs
val_inference_freq: 20
# top-N atoms with the smallest distances with surface node for mdn calculate in scoring model
topN: 1
# predict bond type or not in training scoring model stage
bond_type_prediction: true
# atom type prediction or not in training scoring model stage
atom_type_prediction: true
## wandb log
# use wandb to log or not
wandb: true
# wandb dir
wandb_dir: ~/wandb/SurfDock
# the project name in wandb
project: SurfDock_V3_PDBBind_ema_model_pocket_8A
# the run name in wandb
run_name: project_surface_V3_PDBBind_ema_model_pocket_8A
# dir of log files
log_dir: ~/wandb/SurfDock/workdir