| { |
| "artifact_provenance": { |
| "corpus_manifest_sha256": "d9e3cf4778ceec2035e00cf432532ae62bb4c29839b3147a8a2f541daa4c27c3", |
| "corpus_offsets_sha256": "757955f534b0a63d4d46c31e6aa18e4367237303638d6341fb59aae7a9d6ca48", |
| "corpus_source_sha256": "f9926180886410af5560017a3534cdc0cd9e0e2b5b17fbeb52392de62fb1ec31", |
| "corpus_tokens_sha256": "8bf8dd90a22428ef74bbc02d64a3c8916dd94ed15389ca80741c438b0f630ac2", |
| "corpus_words": 8900000, |
| "corpus_words_sha256": "4036ddbe4921b35fb732d9cffb46f968c58ed8463def22bb846a35f5250790c3", |
| "document_order_sha256": null, |
| "factor_manifest_sha256": null, |
| "factorization_manifest_sha256": "52f31760faac4620008f19dbb6533b4b1856831f7e2b94fdb178b3b06803605a", |
| "influence_preflight_sha256": null, |
| "repositories": { |
| "encoded_corpus": { |
| "repo": "miguelcsx/babylm-bbpe16k-512-encoded", |
| "repo_type": "dataset", |
| "revision": "a62a6f274f3d86c05c8dd967a0c7200573ae9ef6" |
| }, |
| "encoded_factorized_null": { |
| "repo": "miguelcsx/factorized-strict-small-null-corpus", |
| "repo_type": "dataset", |
| "revision": "d9eeeb7d68eb31c8f9bc4c9c26088167acc8872f" |
| }, |
| "factorization_priors": { |
| "repo": "miguelcsx/factorized-strict-small-priors", |
| "repo_type": "dataset", |
| "revision": "3ea2f0d08ce87f20ce8986a4a08374ca8cc4f23f" |
| }, |
| "tokenizer": { |
| "repo": "miguelcsx/causal-focus-bbpe16k-tokenizer", |
| "repo_type": "model", |
| "revision": "5c95ae8fe9cc0f745f8d643192775fbceff5fedb" |
| }, |
| "tokenizer_factorized": { |
| "repo": "miguelcsx/factorized-strict-small-bpe16k-tokenizer", |
| "repo_type": "model", |
| "revision": "a1f87b2b439ea988ba2b94b62898ed44780fb5e5" |
| } |
| }, |
| "structured_priors_manifest_sha256": "ee75fe93d2dffd4bc7d86d1a1fd9f1aa1d76a956012dc6357274a252f40d8bbf", |
| "tokenizer_sha256": "2b8d1b3f51c0d8f276a64ea6a69efa50b4d9899780f1b282b4d0dcb7be5bbb0f", |
| "tokenizer_word_start_sha256": "42d9a905c49df4c54dbe3bc93e567c9bbe491ec6fbee5fcbed70e8d1ad533d49" |
| }, |
| "compliance": { |
| "auxiliary_linguistic_training_words": 2000000, |
| "brown_training_words": 1000000, |
| "competition_status": "strict_small_conservative_accounting", |
| "conservative_corpus_accounting_words": 10000000, |
| "conservative_counted_words": 100000000, |
| "corpus_induced_priors": true, |
| "corpus_limit_words": 10000000, |
| "evaluation_holdout_words": 1000000, |
| "exposure_limit_words": 100000000, |
| "external_teacher": false, |
| "generated_words": 0, |
| "induction_corpus_disjoint_from_lm_and_holdout": true, |
| "leaderboard_checkpoint_words": 88000000, |
| "model_exposure_words": 100000000, |
| "model_training_corpus_words": 8900000, |
| "ppmi_induction_words": 1000000, |
| "relational_corpus_fraction": 0.025, |
| "same_run_checkpoint_distribution": false, |
| "signals_derived_from_current_model_input": false, |
| "strict_small_status": "corpus_10m_model_views_100m", |
| "syntax_training_words": 0, |
| "teacher_queries": 0, |
| "tokenizer_training_words": 8900000, |
| "total_exposure_words": 100000000, |
| "track": "strict-small", |
| "unique_corpus_words": 10000000, |
| "within_100m_conservative_budget": true, |
| "within_track_corpus_budget": true, |
| "within_track_exposure_budget": true |
| }, |
| "factor_priors": { |
| "angular_margin": 0.1, |
| "dual": { |
| "enabled": false, |
| "learning_rate": 0.001, |
| "max_weight": 1.0, |
| "targets": [ |
| 1.0, |
| 4.0, |
| 0.1 |
| ] |
| }, |
| "enabled": false, |
| "lambda_conceptual": 0.0, |
| "lambda_lexical": 0.0, |
| "lambda_syntax": 0.0, |
| "max_positions": 256, |
| "radial_margin": 0.02, |
| "radial_weight": 1.0, |
| "randomize": false, |
| "randomize_seed": 42, |
| "syntax_depth_weight": 0.1, |
| "syntax_distance_weight": 0.1, |
| "syntax_window": 32, |
| "temperature": 0.07, |
| "warmup_words": 1000000 |
| }, |
| "factorization": { |
| "context_window": 4, |
| "covariance_weight": 0.005, |
| "cross_covariance_weight": 0.002, |
| "enabled": true, |
| "gradient_budget": 0.1, |
| "gradient_ema_momentum": 0.95, |
| "graph_mode": "null", |
| "loss_enabled": true, |
| "max_gradient_scale": 1000.0, |
| "null_seed": 271828, |
| "pairs_per_microbatch": 128, |
| "ramp_words": 5000000, |
| "regularizer_words": 512, |
| "signal_mode": "induced_priors", |
| "temperature": 0.07, |
| "variance_weight": 0.02 |
| }, |
| "geometry": { |
| "curvature": 1.0, |
| "distance_margin": 0.1, |
| "enabled": false, |
| "lambda_radial": 0.02, |
| "lambda_related": 0.02, |
| "max_tokens": 256, |
| "radial_margin": 0.05, |
| "warmup_words": 5000000 |
| }, |
| "model": { |
| "absolute_positions": false, |
| "attention_dropout": 0.1, |
| "bos_token_id": 1, |
| "cognitive_readout_layer": 0, |
| "cognitive_readout_weight": 0.0, |
| "direct_sum_dims": [], |
| "direct_sum_heads": [], |
| "direct_sum_intermediate_sizes": [], |
| "dropout": 0.1, |
| "eos_token_id": 2, |
| "expert_intermediate_size": null, |
| "experts_per_token": 1, |
| "factor_readout_dims": [ |
| 128, |
| 128, |
| 128 |
| ], |
| "factor_readout_mode": "orthogonal", |
| "factor_readout_reflectors": 384, |
| "factor_readout_seed": 314159, |
| "future_offsets": [], |
| "geometry_curvature": 1.0, |
| "geometry_lexical_dim": 0, |
| "hidden_size": 384, |
| "initializer_range": 0.03227486121839514, |
| "intermediate_size": 1280, |
| "lexical_residual_buckets": 0, |
| "lexical_residual_dim": 0, |
| "lexical_residual_scale": 1.0, |
| "mask_token_id": 4, |
| "max_seq_len": 512, |
| "num_attention_heads": 6, |
| "num_experts": 1, |
| "num_hidden_layers": 12, |
| "pad_token_id": 3, |
| "position_buckets": 32, |
| "recurrent_steps": 1, |
| "residual_mixing": true, |
| "rtd_auxiliary": true, |
| "state_mixer_kernel": 0, |
| "structured_projection_dim": 0, |
| "use_alibi": false, |
| "use_rope": false, |
| "value_gating": true, |
| "vocab_size": 16384 |
| }, |
| "release": "TOLM", |
| "repository": { |
| "commit": "748fa8a55539a43dd27bcf08b42abfa009012223", |
| "tracked_dirty": false |
| }, |
| "structured_priors": { |
| "decay_end_words": 7000000, |
| "enabled": false, |
| "hold_until_words": 2000000, |
| "init_scale": 0.0, |
| "lambda_lexical": 0.0, |
| "lambda_orth": 0.0, |
| "lambda_syntax": 0.0, |
| "lexical_dim": 128, |
| "max_positions": 256, |
| "prior_mode": "contrastive", |
| "randomize": false, |
| "representation": "slices", |
| "require_aligned_artifacts": false, |
| "syntax_dim": 128, |
| "temperature": 0.07, |
| "warmup_words": 100000 |
| }, |
| "training": { |
| "adaptive_masking": { |
| "enabled": false, |
| "max_mask_prob": 4.0, |
| "min_mask_prob": 0.25, |
| "momentum": 0.99 |
| }, |
| "batch_size": 16, |
| "beta1": 0.9, |
| "beta2": 0.98, |
| "causal_fraction": 0.0, |
| "causal_noise_kind": "random", |
| "causal_noise_probability": 0.0, |
| "causal_unit": "segment", |
| "cooldown_fraction": 0.016, |
| "data2vec_layers": 4, |
| "data2vec_weight": 0.0, |
| "device": "xpu:1", |
| "document_curriculum": { |
| "enabled": false |
| }, |
| "ema_decay": 0.9998, |
| "epsilon": 1e-08, |
| "exact_word_checkpoints": true, |
| "exposure_words": 100000000, |
| "final_lr_ratio": 0.1, |
| "frequency_aware_masking": { |
| "enabled": false, |
| "max_mask_prob": 4.0, |
| "min_mask_prob": 0.25, |
| "temperature": 1.0 |
| }, |
| "future_loss_weight": 0.0, |
| "gradient_clip": 2.0, |
| "label_smoothing": 0.0, |
| "learning_progress": { |
| "buckets_per_axis": 4, |
| "enabled": false, |
| "fast_momentum": 0.9, |
| "forgetting_weight": 1.0, |
| "slow_momentum": 0.99, |
| "uniform_floor": 0.3, |
| "window_documents": 4000 |
| }, |
| "learning_rate": 0.0035, |
| "learning_rate_schedule": "cosine", |
| "log_interval": 50, |
| "mask_probability_end": 0.5, |
| "mask_probability_schedule": "uniform", |
| "mask_probability_start": 0.15, |
| "mask_replace_probability": 0.8, |
| "mask_schedule": "complementary", |
| "masked_fraction": 1.0, |
| "masking_unit": "word", |
| "microbatch_tokens": 8192, |
| "mixed_precision": "bf16", |
| "num_workers": 0, |
| "objective_period": 16, |
| "objective_rng_reset_words": [], |
| "objective_sanity_window_steps": 512, |
| "objective_schedule": "coverage", |
| "optimizer": "lamb", |
| "packing_strategy": "dense", |
| "pin_memory": false, |
| "random_replace_probability": 0.1, |
| "recombine_probability": 0.0, |
| "recombine_strategy": "random", |
| "recovery_interval_words": 1000000, |
| "resource_memory": { |
| "enabled": false |
| }, |
| "router_aux_weight": 0.0, |
| "save_steps_words": [ |
| 1000000, |
| 2000000, |
| 3000000, |
| 4000000, |
| 5000000, |
| 6000000, |
| 7000000, |
| 8000000, |
| 9000000, |
| 10000000, |
| 20000000, |
| 30000000, |
| 40000000, |
| 50000000, |
| 60000000, |
| 70000000, |
| 80000000, |
| 88000000, |
| 90000000, |
| 100000000 |
| ], |
| "schedule_total_words": 100000000, |
| "span_max_length": 3, |
| "stable_fraction": 0.9, |
| "telemetry": { |
| "enabled": true, |
| "gradient_checkpoints_words": [ |
| 1000000, |
| 10000000, |
| 40000000, |
| 70000000, |
| 88000000 |
| ], |
| "sampler_trace": true |
| }, |
| "threads_per_process": 12, |
| "tokens_per_update": 16384, |
| "warmup_fraction": 0.016, |
| "weight_decay": 0.1, |
| "z_loss_weight": 0.0001 |
| }, |
| "variant": "factorized_null", |
| "words_seen": 100000000 |
| } |
|
|