| { |
| "attention_probs_dropout_prob": 0.1, |
| "config_tensor": { |
| "attn": { |
| "fc": { |
| "amp": true, |
| "factorization": "tt", |
| "per_decomp_rank_ratio_limit": 0.75, |
| "ranks": 300, |
| "build_rank_parameters": false, |
| "set_scale_factors": false, |
| "shape": [ |
| 16, |
| 8, |
| 8, |
| 8, |
| 8, |
| 16 |
| ], |
| "tensorized": true |
| }, |
| "sub_layers": { |
| "amp": true, |
| "factorization": "tt", |
| "per_decomp_rank_ratio_limit": 0.75, |
| "ranks": 300, |
| "build_rank_parameters": false, |
| "set_scale_factors": false, |
| "shape": [ |
| 16, |
| 8, |
| 8, |
| 8, |
| 8, |
| 16 |
| ], |
| "tensorized": true |
| } |
| }, |
| "embedding": { |
| "tensorized": false |
| }, |
| "head": { |
| "tensorized": false |
| }, |
| "pff": [ |
| { |
| "amp": true, |
| "factorization": "tt", |
| "per_decomp_rank_ratio_limit": 0.75, |
| "ranks": 300, |
| "build_rank_parameters": false, |
| "set_scale_factors": false, |
| "shape": [ |
| 16, |
| 8, |
| 8, |
| 16, |
| 16, |
| 16 |
| ], |
| "tensorized": true |
| }, |
| { |
| "amp": true, |
| "factorization": "tt", |
| "per_decomp_rank_ratio_limit": 0.75, |
| "ranks": 300, |
| "build_rank_parameters": false, |
| "set_scale_factors": false, |
| "shape": [ |
| 16, |
| 16, |
| 16, |
| 8, |
| 8, |
| 16 |
| ], |
| "tensorized": true |
| } |
| ], |
| "pooler": { |
| "tensorized": false |
| } |
| }, |
| "hidden_act": "gelu", |
| "hidden_dropout_prob": 0.1, |
| "hidden_size": 1024, |
| "initializer_range": 0.02, |
| "intermediate_size": 4096, |
| "max_position_embeddings": 512, |
| "num_attention_heads": 16, |
| "num_hidden_layers": 24, |
| "output_all_encoded_layers": false, |
| "type_vocab_size": 2, |
| "vocab_size": 30528 |
| } |
|
|