File size: 8,285 Bytes
34bfe21 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 | from datasets.dataset import *
from utils import *
from datetime import datetime
class setting_config:
"""
the config of training setting.
"""
network = 'LGF-VMUNet'
model_config = {
'num_classes': 2,
'input_channels': 3,
'depths': [2,2,2,2],
'depths_decoder': [2,2,2,1],
'drop_path_rate': 0,
'load_ckpt_path': None,
'use_full_scale_skip': True,
}
datasets_name = 'synapse'
input_size_h = 224
input_size_w = 224
if datasets_name == 'synapse':
data_path = './data/Synapse/train_npz/'
datasets = Synapse_dataset
list_dir = './data/Synapse/lists/lists_Synapse/'
test_path = './data/Synapse/test_npz/'
class_names = ['Aorta', 'Gallbladder', 'Kidney(L)', 'Kidney(R)', 'Liver', 'Pancreas', 'Spleen', 'Stomach']
num_classes = 9
model_config['num_classes'] = num_classes
else:
raise Exception('datasets in not right!')
loss_weight = [1, 1]
weights=[0.1, 0.2, 0.3, 0.4]
criterion = AdaptiveHierarchicalLoss()
z_spacing = 1
input_channels = 3
distributed = False
local_rank = -1
num_workers = 32 #16
seed = 42
world_size = None
rank = None
amp = False
batch_size = 32
epochs = 200
work_dir = 'results/' + network + '_' + datasets_name + '_' + datetime.now().strftime('%A_%d_%B_%Y_%Hh_%Mm_%Ss') + '/'
print_interval = 20
val_interval = 5
save_interval = 10
test_weights_path = ''
threshold = 0.5
opt = 'AdamW'
assert opt in ['Adadelta', 'Adagrad', 'Adam', 'AdamW', 'Adamax', 'ASGD', 'RMSprop', 'Rprop', 'SGD'], 'Unsupported optimizer!'
if opt == 'Adadelta':
lr = 0.01 # default: 1.0 β coefficient that scale delta before it is applied to the parameters
rho = 0.9 # default: 0.9 β coefficient used for computing a running average of squared gradients
eps = 1e-6 # default: 1e-6 β term added to the denominator to improve numerical stability
weight_decay = 0.05 # default: 0 β weight decay (L2 penalty)
elif opt == 'Adagrad':
lr = 0.01 # default: 0.01 β learning rate
lr_decay = 0 # default: 0 β learning rate decay
eps = 1e-10 # default: 1e-10 β term added to the denominator to improve numerical stability
weight_decay = 0.05 # default: 0 β weight decay (L2 penalty)
elif opt == 'Adam':
lr = 0.0001 # default: 1e-3 β learning rate
betas = (0.9, 0.999) # default: (0.9, 0.999) β coefficients used for computing running averages of gradient and its square
eps = 1e-8 # default: 1e-8 β term added to the denominator to improve numerical stability
weight_decay = 0.05 # default: 0 β weight decay (L2 penalty)
amsgrad = False # default: False β whether to use the AMSGrad variant of this algorithm from the paper On the Convergence of Adam and Beyond
elif opt == 'AdamW':
lr = 1e-4 # paper: 1e-4
betas = (0.9, 0.999) # default: (0.9, 0.999) β coefficients used for computing running averages of gradient and its square
eps = 1e-8 # default: 1e-8 β term added to the denominator to improve numerical stability
weight_decay = 1e-4 # paper: 1e-4
amsgrad = False # default: False β whether to use the AMSGrad variant of this algorithm from the paper On the Convergence of Adam and Beyond
elif opt == 'Adamax':
lr = 2e-3 # default: 2e-3 β learning rate
betas = (0.9, 0.999) # default: (0.9, 0.999) β coefficients used for computing running averages of gradient and its square
eps = 1e-8 # default: 1e-8 β term added to the denominator to improve numerical stability
weight_decay = 0 # default: 0 β weight decay (L2 penalty)
elif opt == 'ASGD':
lr = 0.01 # default: 1e-2 β learning rate
lambd = 1e-4 # default: 1e-4 β decay term
alpha = 0.75 # default: 0.75 β power for eta update
t0 = 1e6 # default: 1e6 β point at which to start averaging
weight_decay = 0 # default: 0 β weight decay
elif opt == 'RMSprop':
lr = 1e-2 # default: 1e-2 β learning rate
momentum = 0 # default: 0 β momentum factor
alpha = 0.99 # default: 0.99 β smoothing constant
eps = 1e-8 # default: 1e-8 β term added to the denominator to improve numerical stability
centered = False # default: False β if True, compute the centered RMSProp, the gradient is normalized by an estimation of its variance
weight_decay = 0 # default: 0 β weight decay (L2 penalty)
elif opt == 'Rprop':
lr = 1e-2 # default: 1e-2 β learning rate
etas = (0.5, 1.2) # default: (0.5, 1.2) β pair of (etaminus, etaplis), that are multiplicative increase and decrease factors
step_sizes = (1e-6, 50) # default: (1e-6, 50) β a pair of minimal and maximal allowed step sizes
elif opt == 'SGD':
lr = 0.003 # β learning rate
momentum = 0.9 # default: 0 β momentum factor
weight_decay = 0.0001 # default: 0 β weight decay (L2 penalty)
dampening = 0 # default: 0 β dampening for momentum
nesterov = False # default: False β enables Nesterov momentum
sch = 'CosineAnnealingLR'
if sch == 'StepLR':
step_size = epochs // 5 # β Period of learning rate decay.
gamma = 0.5 # β Multiplicative factor of learning rate decay. Default: 0.1
last_epoch = -1 # β The index of last epoch. Default: -1.
elif sch == 'MultiStepLR':
milestones = [60, 120, 150] # β List of epoch indices. Must be increasing.
gamma = 0.1 # β Multiplicative factor of learning rate decay. Default: 0.1.
last_epoch = -1 # β The index of last epoch. Default: -1.
elif sch == 'ExponentialLR':
gamma = 0.99 # β Multiplicative factor of learning rate decay.
last_epoch = -1 # β The index of last epoch. Default: -1.
elif sch == 'CosineAnnealingLR':
T_max = epochs # β Maximum number of iterations. Cosine function period.
eta_min = 0. # β Minimum learning rate. Default: 0.
last_epoch = -1 # β The index of last epoch. Default: -1.
elif sch == 'ReduceLROnPlateau':
mode = 'min' # β One of min, max. In min mode, lr will be reduced when the quantity monitored has stopped decreasing; in max mode it will be reduced when the quantity monitored has stopped increasing. Default: βminβ.
factor = 0.1 # β Factor by which the learning rate will be reduced. new_lr = lr * factor. Default: 0.1.
patience = 10 # β Number of epochs with no improvement after which learning rate will be reduced. For example, if patience = 2, then we will ignore the first 2 epochs with no improvement, and will only decrease the LR after the 3rd epoch if the loss still hasnβt improved then. Default: 10.
threshold = 0.0001 # β Threshold for measuring the new optimum, to only focus on significant changes. Default: 1e-4.
threshold_mode = 'rel' # β One of rel, abs. In rel mode, dynamic_threshold = best * ( 1 + threshold ) in βmaxβ mode or best * ( 1 - threshold ) in min mode. In abs mode, dynamic_threshold = best + threshold in max mode or best - threshold in min mode. Default: βrelβ.
cooldown = 0 # β Number of epochs to wait before resuming normal operation after lr has been reduced. Default: 0.
min_lr = 0 # β A scalar or a list of scalars. A lower bound on the learning rate of all param groups or each group respectively. Default: 0.
eps = 1e-08 # β Minimal decay applied to lr. If the difference between new and old lr is smaller than eps, the update is ignored. Default: 1e-8.
elif sch == 'CosineAnnealingWarmRestarts':
T_0 = 50 # β Number of iterations for the first restart.
T_mult = 2 # β A factor increases T_{i} after a restart. Default: 1.
eta_min = 1e-6 # β Minimum learning rate. Default: 0.
last_epoch = -1 # β The index of last epoch. Default: -1.
elif sch == 'WP_MultiStepLR':
warm_up_epochs = 10
gamma = 0.1
milestones = [125, 225]
elif sch == 'WP_CosineLR':
warm_up_epochs = 20
|