File size: 1,379 Bytes
3516d2e
 
 
 
 
f201b74
 
3516d2e
f201b74
 
 
3516d2e
f201b74
 
 
 
 
 
 
 
 
3516d2e
5973e3a
 
9ef095a
3516d2e
5973e3a
9ef095a
3516d2e
5973e3a
f201b74
5973e3a
3516d2e
 
5973e3a
3516d2e
 
5973e3a
 
9ef095a
5973e3a
 
f201b74
9ef095a
3516d2e
 
f201b74
 
3516d2e
 
5973e3a
f201b74
3516d2e
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
{
  "_name_or_path": "/models/phi-2",
  "architectures": [
    "PhiForCausalLM"
  ],
  "attention_dropout": 0.0,
  "bos_token_id": 50256,
  "embd_pdrop": 0.0,
  "eos_token_id": 50256,
  "hidden_act": "gelu_new",
  "hidden_size": 2560,
  "initializer_range": 0.02,
  "intermediate_size": 10240,
  "layer_norm_eps": 1e-05,
  "max_position_embeddings": 2048,
  "model_type": "phi",
  "num_attention_heads": 32,
  "num_hidden_layers": 32,
  "num_key_value_heads": 32,
  "partial_rotary_factor": 0.4,
  "qk_layernorm": false,
  "quantization_config": {
    "amp": true,
    "autoround_version": "0.3.1.dev",
    "backend": "auto_round:gptq:exllamav2",
    "bits": 4,
    "data_type": "int",
    "dataset": "NeelNanda/pile-10k",
    "enable_minmax_tuning": true,
    "enable_norm_bias_tuning": false,
    "enable_quanted_input": true,
    "gradient_accumulate_steps": 1,
    "group_size": 128,
    "iters": 1000,
    "low_gpu_mem_usage": false,
    "lr": 0.001,
    "minmax_lr": 0.001,
    "nsamples": 512,
    "quant_block_list": null,
    "quant_method": "intel/auto-round",
    "scale_dtype": "torch.float16",
    "seqlen": 2048,
    "sym": true,
    "train_bs": 8
  },
  "resid_pdrop": 0.1,
  "rope_scaling": null,
  "rope_theta": 10000.0,
  "tie_word_embeddings": false,
  "torch_dtype": "float16",
  "transformers_version": "4.44.2",
  "use_cache": true,
  "vocab_size": 51200
}