File size: 1,244 Bytes
3cd1076 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 | #!/bin/bash
#SBATCH -J duo-lm1b # Job name
#SBATCH -o watch_folder/%x_%j.out # output file (%j expands to jobID)
#SBATCH -N 1 # Total number of nodes requested
#SBATCH --get-user-env # retrieve the users login environment
#SBATCH --mem=64000 # server memory requested (per node)
#SBATCH -t 960:00:00 # Time limit (hh:mm:ss)
#SBATCH --partition=anonymous # Request partition
#SBATCH --constraint="[a5000|a6000|a100|3090]"
#SBATCH --constraint="gpu-mid|gpu-high"
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8 # Type/number of GPUs needed
#SBATCH --open-mode=append # Do not overwrite logs
#SBATCH --requeue # Requeue upon pre-emption
# To enable preemption re-loading, set `hydra.run.dir` or
# `checkpointing.save_dir` explicitly.
python -u -m main \
loader.batch_size=32 \
loader.eval_batch_size=32 \
data=openwebtext-split \
wandb.name=duo-owt \
model=small \
algo=duo \
model.length=1024 \
algo.gumbel_tau_log10_start=-3.0 \
algo.gumbel_tau_log10_end=-3.0 \
algo.gamma_min=-3.55 \
algo.gamma_max=-1.85 \
algo.curriculum_start=0 \
algo.curriculum_end=500000
|