| { |
| "abundance": { |
| "min_abundance": 0.0001, |
| "output_min": -2.0, |
| "output_max": 2.0, |
| "normalize": false, |
| "num_bins": 5, |
| "binning_method": "quantile", |
| "fixed_thresholds": [ |
| 0.0, |
| 0.0001, |
| 0.001, |
| 0.01, |
| 0.1 |
| ], |
| "enable_regression": true, |
| "regression_top_k_bins": 2 |
| }, |
| "model": { |
| "vocab_size": 10000, |
| "embedding_dim": 1536, |
| "hidden_size": 320, |
| "num_hidden_layers": 6, |
| "num_attention_heads": 5, |
| "intermediate_size": 1280, |
| "hidden_dropout_prob": 0.1, |
| "attention_probs_dropout_prob": 0.1, |
| "layer_norm_eps": 1e-12, |
| "max_position_embeddings": 2048, |
| "pad_token_id": -1, |
| "cls_token_id": -1, |
| "mlp_dropout": 0.1 |
| }, |
| "loss": { |
| "lambda_bin": 1.0, |
| "lambda_regression": 0.3, |
| "bin_loss_type": "focal", |
| "regression_loss_type": "mse", |
| "focal_alpha": 0.25, |
| "focal_gamma": 2.0 |
| }, |
| "training": { |
| "mask_prob": 0.15, |
| "mask_ratio": [ |
| 0.8, |
| 0.1, |
| 0.1 |
| ], |
| "train_ratio": 0.9, |
| "val_ratio": 0.1, |
| "batch_size": 512, |
| "num_workers": 0, |
| "learning_rate": 5e-05, |
| "weight_decay": 0.01, |
| "warmup_steps": 1000, |
| "lr_scheduler": "warmup_cosine", |
| "min_lr_ratio": 0.1, |
| "max_epochs": 50, |
| "gradient_clip_val": 1.0, |
| "precision": "bf16", |
| "devices": [ |
| 0 |
| ], |
| "accelerator": "auto" |
| }, |
| "qwen3_alignment": { |
| "enabled": true, |
| "embedding_path": "", |
| "projection_dim": 4096, |
| "temperature": 0.07, |
| "loss_weight": 0.1 |
| }, |
| "ntv3_embedding_path": "../../embeddings/reference_ntv3_embeddings.h5", |
| "data_path": "", |
| "output_dir": "./outputs", |
| "experiment_name": "MGM2-medium", |
| "seed": 42 |
| } |
|
|