File size: 2,244 Bytes
69a4fd2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
{
    "architectures": [
        "InklingForConditionalGeneration"
    ],
    "model_type": "inkling_mm_model",
    "eos_token_id": 200006,
    "text_config": {
        "model_max_length": 1048576,
        "torch_dtype": "bfloat16",
        "hidden_size": 6144,
        "num_hidden_layers": 12,
        "vocab_size": 201024,
        "num_attention_heads": 64,
        "num_key_value_heads": 8,
        "head_dim": 128,
        "d_rel": 16,
        "rel_extent": 1024,
        "q_bias": false,
        "o_bias": false,
        "log_scaling_n_floor": 128000,
        "log_scaling_alpha": 0.1,
        "rms_norm_eps": 1e-06,
        "use_embed_norm": true,
        "local_layer_ids": [
            0,
            1,
            2,
            3,
            4,
            6,
            7,
            8,
            9,
            10
        ],
        "dense_mlp_idx": 2,
        "use_sconv": true,
        "sconv_kernel_size": 4,
        "unpadded_vocab_size": 200058,
        "logits_mup_width_multiplier": 24.0,
        "final_logit_softcapping": null,
        "swa_head_dim": 128,
        "swa_num_attention_heads": 64,
        "swa_num_key_value_heads": 16,
        "sliding_window_size": 512,
        "n_routed_experts": 16,
        "num_experts_per_tok": 6,
        "n_shared_experts": 2,
        "shared_expert_sink": true,
        "dense_intermediate_size": 24576,
        "intermediate_size": 3072,
        "route_scale": 8.0,
        "use_gate_bias": true,
        "gate_activation": "sigmoid",
        "norm_after_topk": true,
        "use_global_scale": true
    },
    "audio_config": {
        "decoder_dmodel": 6144,
        "n_mel_bins": 80,
        "mel_vocab_size": 16,
        "bias": false,
        "dmel_min_value": -7.0,
        "dmel_max_value": 2.0,
        "use_audio_norm": true,
        "audio_mode": "dmel"
    },
    "vision_config": {
        "vision_encoder_type": "hmlp",
        "decoder_dmodel": 6144,
        "patch_size": 40,
        "temporal_patch_size": 2,
        "n_channels": 3,
        "n_layers": 4,
        "use_vision_norm": true
    },
    "mtp_config": {
        "num_nextn_predict_layers": 2,
        "chain_hidden_post_norm": false,
        "local_layer_ids": [
            0
        ]
    }
}