MK0727 commited on
Commit
7b22cdd
·
verified ·
1 Parent(s): 8d8cf7f

Upload lambda-160m pretrained model

Browse files
Files changed (2) hide show
  1. model.pth +1 -1
  2. model_config.json +1 -1
model.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ccf44ef8dd3ef60402ff149195e31321f994ced33109e58b09e5e596196b4e05
3
  size 658115811
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:07fe7a9af8bf2d5daf569846017d28a2c1f566761b5d722e2f128dc8bdde46cc
3
  size 658115811
model_config.json CHANGED
@@ -1 +1 @@
1
- {"max_len": 1024, "d_model": 768, "num_layers": 16, "num_heads": 12, "d_ff": 3072, "learning_rate": 0.0002, "lr_schedule": "warmup_cosine", "lr_warmup_steps": 2000, "min_learning_rate": 2e-05, "min_learning_rate_ratio": 0.1, "loss_chunk_size": 32, "pad_token_id": 0, "bos_token_id": 2, "eos_token_id": 3, "corpus_signature": "551ac72eceb57f5f", "dataset_cases": [{"name": "fineweb2-edu-ja", "genre": "web", "language": "ja", "dataset_path": "hotchpotch/fineweb-2-edu-japanese", "config_name": "default", "split": "train", "text_column": "text", "token_percentage": 30.0, "is_ramped": false, "repeat_on_end": true, "excluded_url_domains": ["wikipedia.org"]}, {"name": "cleanedwiki-jp", "genre": "wiki", "language": "ja", "dataset_path": "MK0727/CleanedWiki-jp", "config_name": "all", "split": "train", "text_column": "text", "token_percentage": 70.0, "is_ramped": true, "repeat_on_end": true, "excluded_url_domains": []}], "mix_cycle_tokens": 100000, "ramp_start_progress": 0.5, "val_split_modulo": 100, "val_split_index": 0, "validation_cache_path": "models/lambda-160m/validation-cache-551ac72eceb57f5f-bos-eos-text-hash-len1024-samples6144-split100-0.pt", "validation_sample_count": 6144, "trained_steps": 40960}
 
1
+ {"max_len": 1024, "d_model": 768, "num_layers": 16, "num_heads": 12, "d_ff": 3072, "learning_rate": 0.0002, "batch_size": 96, "gradient_accumulation_steps": 4, "effective_batch_size": 384, "lr_schedule": "warmup_cosine", "lr_warmup_steps": 2000, "min_learning_rate": 0.0001, "min_learning_rate_ratio": 0.5, "loss_chunk_size": 32, "pad_token_id": 0, "bos_token_id": 2, "eos_token_id": 3, "corpus_signature": "fb2af8e50eb90fad", "dataset_case": {"name": "cleaned-fineweb2-edu-jp", "genre": "web", "language": "ja", "dataset_path": "MK0727/CleanedFineWeb2Edu-jp", "config_name": "default", "split": "train", "text_column": "text"}, "val_split_modulo": 100, "val_split_index": 0, "validation_cache_path": "models/lambda-160m/validation-cache-fb2af8e50eb90fad-bucket-packing-v1-len1024-samples6144-split100-0.pt", "validation_sample_count": 6144, "packing_version": "bucket-packing-v1", "trained_steps": 10240}