File size: 959 Bytes
1afab79
1
{"max_len": 1024, "d_model": 768, "num_layers": 16, "num_heads": 12, "d_ff": 3072, "learning_rate": 0.0001, "batch_size": 72, "gradient_accumulation_steps": 2, "effective_batch_size": 144, "lr_schedule": "fixed", "loss_chunk_size": 32, "pad_token_id": 0, "bos_token_id": 2, "eos_token_id": 3, "corpus_signature": "ba5e40d4dae8dceb", "dataset_case": {"name": "synthetic-textbook-jp", "genre": "textbook", "language": "ja", "dataset_path": "MK0727/SyntheticTextbook-jp", "config_name": "default", "split": "train", "text_column": "rewrite"}, "val_split_modulo": 100, "val_split_index": 0, "validation_cache_path": "models/lambda-160m-midtrained/validation-cache-ba5e40d4dae8dceb-bucket-packing-v1-len1024-samples4608-split100-0.pt", "validation_sample_count": 4608, "packing_version": "bucket-packing-v1", "trained_steps": 10240, "midtraining_source_model": "models/lambda-160m", "midtraining_max_steps": 10240, "shuffle_buffer_size": 10000, "shuffle_seed": 17}