leonMW commited on
Commit
5b27081
·
verified ·
1 Parent(s): d9ae97c

Training in progress, epoch 1

Browse files
README.md CHANGED
@@ -1,16 +1,17 @@
1
  ---
 
2
  library_name: transformers
3
  model_name: DeepSeek-R1-Distill-Qwen-7B-S
4
  tags:
5
  - generated_from_trainer
6
- - grpo
7
  - trl
 
8
  licence: license
9
  ---
10
 
11
  # Model Card for DeepSeek-R1-Distill-Qwen-7B-S
12
 
13
- This model is a fine-tuned version of [None](https://huggingface.co/None).
14
  It has been trained using [TRL](https://github.com/huggingface/trl).
15
 
16
  ## Quick start
@@ -26,7 +27,7 @@ print(output["generated_text"])
26
 
27
  ## Training procedure
28
 
29
- [<img src="https://raw.githubusercontent.com/wandb/assets/main/wandb-github-badge-28.svg" alt="Visualize in Weights & Biases" width="150" height="24"/>](https://wandb.ai/leonwenderoth-tu-darmstadt/huggingface/runs/3b318nnm)
30
 
31
 
32
  This model was trained with GRPO, a method introduced in [DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models](https://huggingface.co/papers/2402.03300).
@@ -34,10 +35,10 @@ This model was trained with GRPO, a method introduced in [DeepSeekMath: Pushing
34
  ### Framework versions
35
 
36
  - TRL: 0.23.0
37
- - Transformers: 4.56.1
38
  - Pytorch: 2.7.1
39
- - Datasets: 4.1.0
40
- - Tokenizers: 0.22.0
41
 
42
  ## Citations
43
 
 
1
  ---
2
+ base_model: deepseek-ai/DeepSeek-R1-Distill-Qwen-7B
3
  library_name: transformers
4
  model_name: DeepSeek-R1-Distill-Qwen-7B-S
5
  tags:
6
  - generated_from_trainer
 
7
  - trl
8
+ - grpo
9
  licence: license
10
  ---
11
 
12
  # Model Card for DeepSeek-R1-Distill-Qwen-7B-S
13
 
14
+ This model is a fine-tuned version of [deepseek-ai/DeepSeek-R1-Distill-Qwen-7B](https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-7B).
15
  It has been trained using [TRL](https://github.com/huggingface/trl).
16
 
17
  ## Quick start
 
27
 
28
  ## Training procedure
29
 
30
+ [<img src="https://raw.githubusercontent.com/wandb/assets/main/wandb-github-badge-28.svg" alt="Visualize in Weights & Biases" width="150" height="24"/>](https://wandb.ai/leonwenderoth-tu-darmstadt/huggingface/runs/6e7rbl0y)
31
 
32
 
33
  This model was trained with GRPO, a method introduced in [DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models](https://huggingface.co/papers/2402.03300).
 
35
  ### Framework versions
36
 
37
  - TRL: 0.23.0
38
+ - Transformers: 4.56.2
39
  - Pytorch: 2.7.1
40
+ - Datasets: 4.1.1
41
+ - Tokenizers: 0.22.1
42
 
43
  ## Citations
44
 
config.json CHANGED
@@ -52,7 +52,7 @@
52
  "rope_theta": 10000,
53
  "sliding_window": null,
54
  "tie_word_embeddings": false,
55
- "transformers_version": "4.56.1",
56
  "use_cache": true,
57
  "use_mrope": false,
58
  "use_sliding_window": false,
 
52
  "rope_theta": 10000,
53
  "sliding_window": null,
54
  "tie_word_embeddings": false,
55
+ "transformers_version": "4.56.2",
56
  "use_cache": true,
57
  "use_mrope": false,
58
  "use_sliding_window": false,
model-00001-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:44e1554944693ab2c2ee2206298f672ad187461af7273c75a2846dfcfdafc5ce
3
  size 4877660776
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f92e283c68a5aaaec340154b19c5eb47b1af2763d5f8373fb828f2d9e99c090f
3
  size 4877660776
model-00002-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f8e7d92b415e20cf4860a58c03441f1d7c36094f14f85e6eb7811b04a6cd4384
3
  size 4932751008
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8071dbd73b597cf546bcc617e1fe0ea63e345ad0ca74131e9edd534f4a7300da
3
  size 4932751008
model-00003-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d6c4440cef95ba260c50cdadcf504608b5be3a974ba61880ebd20438ebb5bc7a
3
  size 4330865200
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e8a082cc033a19c7e743abbe2ebd4d4b964c0a464bf7373434f6c70fc0529581
3
  size 4330865200
model-00004-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:73fe97f70c9c92587afcc3726d3718965c64ac523630704c96617a0551c539e2
3
  size 1089994880
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:73725f36c100c9544e595c52f70ea3b8b1f2bd2702e6dd05a0cc7cce5cacc4d1
3
  size 1089994880
tokenizer_config.json CHANGED
@@ -185,12 +185,8 @@
185
  "eos_token": "<|end▁of▁sentence|>",
186
  "extra_special_tokens": {},
187
  "legacy": true,
188
- "max_length": null,
189
  "model_max_length": 16384,
190
- "pad_to_multiple_of": null,
191
  "pad_token": "<|end▁of▁sentence|>",
192
- "pad_token_type_id": 0,
193
- "padding_side": "left",
194
  "sp_model_kwargs": {},
195
  "tokenizer_class": "LlamaTokenizerFast",
196
  "unk_token": null,
 
185
  "eos_token": "<|end▁of▁sentence|>",
186
  "extra_special_tokens": {},
187
  "legacy": true,
 
188
  "model_max_length": 16384,
 
189
  "pad_token": "<|end▁of▁sentence|>",
 
 
190
  "sp_model_kwargs": {},
191
  "tokenizer_class": "LlamaTokenizerFast",
192
  "unk_token": null,
training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:311ddb11cdb2270d9b5f9713aaf543b5aeaa254ca31d6d738d23ad0aaf61d89f
3
  size 11665
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b4c22616f29b6acb0c167ad6cfb8098e01151b4a513cbf44c28513543f5a598c
3
  size 11665