rahuldhole commited on
Commit
7e58652
·
verified ·
1 Parent(s): c1e5946

Update fine-tuned adapter weights

Browse files
README.md CHANGED
@@ -19,6 +19,6 @@ Tiny LLM is a fine-tuned language model by Rahul Dhole, built on top of Qwen2.5-
19
  ## Training
20
 
21
  - **Method**: LoRA (r=8, alpha=32)
22
- - **Epochs**: 50
23
  - **Learning Rate**: 0.001
24
  - **Data**: data/dummy_train.jsonl
 
19
  ## Training
20
 
21
  - **Method**: LoRA (r=8, alpha=32)
22
+ - **Epochs**: 10
23
  - **Learning Rate**: 0.001
24
  - **Data**: data/dummy_train.jsonl
adapter_config.json CHANGED
@@ -1,9 +1,12 @@
1
  {
 
2
  "alpha_pattern": {},
 
3
  "auto_mapping": null,
4
  "base_model_name_or_path": "Qwen/Qwen2.5-0.5B-Instruct",
5
  "bias": "none",
6
  "corda_config": null,
 
7
  "eva_config": null,
8
  "exclude_modules": null,
9
  "fan_in_fan_out": false,
@@ -20,6 +23,7 @@
20
  "megatron_core": "megatron.core",
21
  "modules_to_save": null,
22
  "peft_type": "LORA",
 
23
  "qalora_group_size": 16,
24
  "r": 8,
25
  "rank_pattern": {},
 
1
  {
2
+ "alora_invocation_tokens": null,
3
  "alpha_pattern": {},
4
+ "arrow_config": null,
5
  "auto_mapping": null,
6
  "base_model_name_or_path": "Qwen/Qwen2.5-0.5B-Instruct",
7
  "bias": "none",
8
  "corda_config": null,
9
+ "ensure_weight_tying": false,
10
  "eva_config": null,
11
  "exclude_modules": null,
12
  "fan_in_fan_out": false,
 
23
  "megatron_core": "megatron.core",
24
  "modules_to_save": null,
25
  "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
  "qalora_group_size": 16,
28
  "r": 8,
29
  "rank_pattern": {},
adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c7d417c1213718029a1c4cc61520bcaf45d48f39e0e983825ba1245978ac1879
3
  size 2175168
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cfe9aab1757f20a25a8f7137e31901ced052a80c2a98858bd0cd10546184dd73
3
  size 2175168
tokenizer.json CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
- size 11421896
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8
3
+ size 11421892
tokenizer_config.json CHANGED
@@ -1,185 +1,11 @@
1
  {
2
- "add_bos_token": false,
3
  "add_prefix_space": false,
4
- "added_tokens_decoder": {
5
- "151643": {
6
- "content": "<|endoftext|>",
7
- "lstrip": false,
8
- "normalized": false,
9
- "rstrip": false,
10
- "single_word": false,
11
- "special": true
12
- },
13
- "151644": {
14
- "content": "<|im_start|>",
15
- "lstrip": false,
16
- "normalized": false,
17
- "rstrip": false,
18
- "single_word": false,
19
- "special": true
20
- },
21
- "151645": {
22
- "content": "<|im_end|>",
23
- "lstrip": false,
24
- "normalized": false,
25
- "rstrip": false,
26
- "single_word": false,
27
- "special": true
28
- },
29
- "151646": {
30
- "content": "<|object_ref_start|>",
31
- "lstrip": false,
32
- "normalized": false,
33
- "rstrip": false,
34
- "single_word": false,
35
- "special": true
36
- },
37
- "151647": {
38
- "content": "<|object_ref_end|>",
39
- "lstrip": false,
40
- "normalized": false,
41
- "rstrip": false,
42
- "single_word": false,
43
- "special": true
44
- },
45
- "151648": {
46
- "content": "<|box_start|>",
47
- "lstrip": false,
48
- "normalized": false,
49
- "rstrip": false,
50
- "single_word": false,
51
- "special": true
52
- },
53
- "151649": {
54
- "content": "<|box_end|>",
55
- "lstrip": false,
56
- "normalized": false,
57
- "rstrip": false,
58
- "single_word": false,
59
- "special": true
60
- },
61
- "151650": {
62
- "content": "<|quad_start|>",
63
- "lstrip": false,
64
- "normalized": false,
65
- "rstrip": false,
66
- "single_word": false,
67
- "special": true
68
- },
69
- "151651": {
70
- "content": "<|quad_end|>",
71
- "lstrip": false,
72
- "normalized": false,
73
- "rstrip": false,
74
- "single_word": false,
75
- "special": true
76
- },
77
- "151652": {
78
- "content": "<|vision_start|>",
79
- "lstrip": false,
80
- "normalized": false,
81
- "rstrip": false,
82
- "single_word": false,
83
- "special": true
84
- },
85
- "151653": {
86
- "content": "<|vision_end|>",
87
- "lstrip": false,
88
- "normalized": false,
89
- "rstrip": false,
90
- "single_word": false,
91
- "special": true
92
- },
93
- "151654": {
94
- "content": "<|vision_pad|>",
95
- "lstrip": false,
96
- "normalized": false,
97
- "rstrip": false,
98
- "single_word": false,
99
- "special": true
100
- },
101
- "151655": {
102
- "content": "<|image_pad|>",
103
- "lstrip": false,
104
- "normalized": false,
105
- "rstrip": false,
106
- "single_word": false,
107
- "special": true
108
- },
109
- "151656": {
110
- "content": "<|video_pad|>",
111
- "lstrip": false,
112
- "normalized": false,
113
- "rstrip": false,
114
- "single_word": false,
115
- "special": true
116
- },
117
- "151657": {
118
- "content": "<tool_call>",
119
- "lstrip": false,
120
- "normalized": false,
121
- "rstrip": false,
122
- "single_word": false,
123
- "special": false
124
- },
125
- "151658": {
126
- "content": "</tool_call>",
127
- "lstrip": false,
128
- "normalized": false,
129
- "rstrip": false,
130
- "single_word": false,
131
- "special": false
132
- },
133
- "151659": {
134
- "content": "<|fim_prefix|>",
135
- "lstrip": false,
136
- "normalized": false,
137
- "rstrip": false,
138
- "single_word": false,
139
- "special": false
140
- },
141
- "151660": {
142
- "content": "<|fim_middle|>",
143
- "lstrip": false,
144
- "normalized": false,
145
- "rstrip": false,
146
- "single_word": false,
147
- "special": false
148
- },
149
- "151661": {
150
- "content": "<|fim_suffix|>",
151
- "lstrip": false,
152
- "normalized": false,
153
- "rstrip": false,
154
- "single_word": false,
155
- "special": false
156
- },
157
- "151662": {
158
- "content": "<|fim_pad|>",
159
- "lstrip": false,
160
- "normalized": false,
161
- "rstrip": false,
162
- "single_word": false,
163
- "special": false
164
- },
165
- "151663": {
166
- "content": "<|repo_name|>",
167
- "lstrip": false,
168
- "normalized": false,
169
- "rstrip": false,
170
- "single_word": false,
171
- "special": false
172
- },
173
- "151664": {
174
- "content": "<|file_sep|>",
175
- "lstrip": false,
176
- "normalized": false,
177
- "rstrip": false,
178
- "single_word": false,
179
- "special": false
180
- }
181
- },
182
- "additional_special_tokens": [
183
  "<|im_start|>",
184
  "<|im_end|>",
185
  "<|object_ref_start|>",
@@ -194,11 +20,7 @@
194
  "<|image_pad|>",
195
  "<|video_pad|>"
196
  ],
197
- "bos_token": null,
198
- "clean_up_tokenization_spaces": false,
199
- "eos_token": "<|im_end|>",
200
- "errors": "replace",
201
- "extra_special_tokens": {},
202
  "model_max_length": 131072,
203
  "pad_token": "<|endoftext|>",
204
  "split_special_tokens": false,
 
1
  {
 
2
  "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": null,
5
+ "clean_up_tokenization_spaces": false,
6
+ "eos_token": "<|im_end|>",
7
+ "errors": "replace",
8
+ "extra_special_tokens": [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9
  "<|im_start|>",
10
  "<|im_end|>",
11
  "<|object_ref_start|>",
 
20
  "<|image_pad|>",
21
  "<|video_pad|>"
22
  ],
23
+ "is_local": false,
 
 
 
 
24
  "model_max_length": 131072,
25
  "pad_token": "<|endoftext|>",
26
  "split_special_tokens": false,
train_config.yaml CHANGED
@@ -1,35 +1,30 @@
1
- model:
2
- name: Qwen/Qwen2.5-0.5B-Instruct
3
- dtype: float16
4
-
5
- # White-label metadata
6
- metadata:
7
- model_name: Tiny LLM
8
- author: Rahul Dhole
9
- base_model: Qwen/Qwen2.5-0.5B-Instruct
10
- license: apache-2.0
11
- description: >
12
- Tiny LLM is a fine-tuned language model by Rahul Dhole,
13
- built on top of Qwen2.5-0.5B-Instruct using LoRA/PEFT.
14
-
15
  lora:
16
- r: 8
17
  alpha: 32
18
  dropout: 0.1
 
19
  target_modules:
20
- - q_proj
21
- - v_proj
 
 
 
 
 
22
 
 
 
 
 
 
 
 
 
23
  training:
24
- epochs: 50
25
  batch_size: 1
 
26
  gradient_accumulation_steps: 4
27
  learning_rate: 0.001
28
- max_length: 512
29
  seed: 42
30
-
31
- data:
32
- path: data/dummy_train.jsonl
33
-
34
- output:
35
- dir: outputs/qwen-fine-tuned
 
1
+ data:
2
+ path: data/dummy_train.jsonl
 
 
 
 
 
 
 
 
 
 
 
 
3
  lora:
 
4
  alpha: 32
5
  dropout: 0.1
6
+ r: 8
7
  target_modules:
8
+ - q_proj
9
+ - v_proj
10
+ metadata:
11
+ author: Rahul Dhole
12
+ base_model: Qwen/Qwen2.5-0.5B-Instruct
13
+ description: 'Tiny LLM is a fine-tuned language model by Rahul Dhole, built on top
14
+ of Qwen2.5-0.5B-Instruct using LoRA/PEFT.
15
 
16
+ '
17
+ license: apache-2.0
18
+ model_name: Tiny LLM
19
+ model:
20
+ dtype: float32
21
+ name: Qwen/Qwen2.5-0.5B-Instruct
22
+ output:
23
+ dir: outputs/qwen-fine-tuned
24
  training:
 
25
  batch_size: 1
26
+ epochs: 10
27
  gradient_accumulation_steps: 4
28
  learning_rate: 0.001
29
+ max_length: 256
30
  seed: 42
 
 
 
 
 
 
training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d0d13eb4ff4f11267cc1f04ea4d7767ec2600a26e55bdf0620baf0a10c8d0b0e
3
- size 6225
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:72969f98bd8768293347d4066f905a1c856b38761560138780eea84b87347511
3
+ size 5585