NathanHerr commited on
Commit
04da0f8
·
verified ·
1 Parent(s): 571127f

Add qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +4 -0
  2. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/README.md +207 -0
  3. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/adapter_config.json +46 -0
  4. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/adapter_model.safetensors +3 -0
  5. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/added_tokens.json +24 -0
  6. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/chat_template.jinja +54 -0
  7. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/README.md +207 -0
  8. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/adapter_config.json +46 -0
  9. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/adapter_model.safetensors +3 -0
  10. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/added_tokens.json +24 -0
  11. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/chat_template.jinja +54 -0
  12. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/merges.txt +0 -0
  13. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/optimizer.pt +3 -0
  14. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/rng_state_0.pth +3 -0
  15. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/rng_state_1.pth +3 -0
  16. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/rng_state_2.pth +3 -0
  17. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/rng_state_3.pth +3 -0
  18. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/scheduler.pt +3 -0
  19. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/special_tokens_map.json +31 -0
  20. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/tokenizer.json +3 -0
  21. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/tokenizer_config.json +207 -0
  22. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/trainer_state.json +749 -0
  23. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/training_args.bin +3 -0
  24. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/vocab.json +0 -0
  25. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/README.md +207 -0
  26. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/adapter_config.json +46 -0
  27. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/adapter_model.safetensors +3 -0
  28. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/added_tokens.json +24 -0
  29. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/chat_template.jinja +54 -0
  30. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/merges.txt +0 -0
  31. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/optimizer.pt +3 -0
  32. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/rng_state_0.pth +3 -0
  33. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/rng_state_1.pth +3 -0
  34. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/rng_state_2.pth +3 -0
  35. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/rng_state_3.pth +3 -0
  36. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/scheduler.pt +3 -0
  37. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/special_tokens_map.json +31 -0
  38. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/tokenizer.json +3 -0
  39. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/tokenizer_config.json +207 -0
  40. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/trainer_state.json +1457 -0
  41. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/training_args.bin +3 -0
  42. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/vocab.json +0 -0
  43. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/README.md +207 -0
  44. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/adapter_config.json +46 -0
  45. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/adapter_model.safetensors +3 -0
  46. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/added_tokens.json +24 -0
  47. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/chat_template.jinja +54 -0
  48. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/merges.txt +0 -0
  49. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/optimizer.pt +3 -0
  50. qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/rng_state_0.pth +3 -0
.gitattributes CHANGED
@@ -172,3 +172,7 @@ qwen2.5-coder-7B-v4-if-grpo-oci-v3-balanced-derivedscratch-updated-b16g16-ng16-l
172
  qwen2.5-coder-7B-v4-if-grpo-oci-v3-balanced-derivedscratch-updated-b16g16-ng16-lr1e-05-v3_delta_prefill-cwup-w50-c2i0.25/checkpoint-750/tokenizer.json filter=lfs diff=lfs merge=lfs -text
173
  qwen2.5-coder-7B-v4-if-grpo-oci-v3-balanced-derivedscratch-updated-b16g16-ng16-lr1e-05-v3_delta_prefill-cwup-w50-c2i0.25/rewarded_completions.jsonl filter=lfs diff=lfs merge=lfs -text
174
  qwen2.5-coder-7B-v4-if-grpo-oci-v3-balanced-derivedscratch-updated-b16g16-ng16-lr1e-05-v3_delta_prefill-cwup-w50-c2i0.25-checkpoint-2750/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
172
  qwen2.5-coder-7B-v4-if-grpo-oci-v3-balanced-derivedscratch-updated-b16g16-ng16-lr1e-05-v3_delta_prefill-cwup-w50-c2i0.25/checkpoint-750/tokenizer.json filter=lfs diff=lfs merge=lfs -text
173
  qwen2.5-coder-7B-v4-if-grpo-oci-v3-balanced-derivedscratch-updated-b16g16-ng16-lr1e-05-v3_delta_prefill-cwup-w50-c2i0.25/rewarded_completions.jsonl filter=lfs diff=lfs merge=lfs -text
174
  qwen2.5-coder-7B-v4-if-grpo-oci-v3-balanced-derivedscratch-updated-b16g16-ng16-lr1e-05-v3_delta_prefill-cwup-w50-c2i0.25-checkpoint-2750/tokenizer.json filter=lfs diff=lfs merge=lfs -text
175
+ qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
176
+ qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
177
+ qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/tokenizer.json filter=lfs diff=lfs merge=lfs -text
178
+ qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/tokenizer.json filter=lfs diff=lfs merge=lfs -text
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/README.md ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen2.5-Coder-7B-Instruct
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen2.5-Coder-7B-Instruct
7
+ - lora
8
+ - transformers
9
+ ---
10
+
11
+ # Model Card for Model ID
12
+
13
+ <!-- Provide a quick summary of what the model is/does. -->
14
+
15
+
16
+
17
+ ## Model Details
18
+
19
+ ### Model Description
20
+
21
+ <!-- Provide a longer summary of what this model is. -->
22
+
23
+
24
+
25
+ - **Developed by:** [More Information Needed]
26
+ - **Funded by [optional]:** [More Information Needed]
27
+ - **Shared by [optional]:** [More Information Needed]
28
+ - **Model type:** [More Information Needed]
29
+ - **Language(s) (NLP):** [More Information Needed]
30
+ - **License:** [More Information Needed]
31
+ - **Finetuned from model [optional]:** [More Information Needed]
32
+
33
+ ### Model Sources [optional]
34
+
35
+ <!-- Provide the basic links for the model. -->
36
+
37
+ - **Repository:** [More Information Needed]
38
+ - **Paper [optional]:** [More Information Needed]
39
+ - **Demo [optional]:** [More Information Needed]
40
+
41
+ ## Uses
42
+
43
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
44
+
45
+ ### Direct Use
46
+
47
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
48
+
49
+ [More Information Needed]
50
+
51
+ ### Downstream Use [optional]
52
+
53
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
54
+
55
+ [More Information Needed]
56
+
57
+ ### Out-of-Scope Use
58
+
59
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
60
+
61
+ [More Information Needed]
62
+
63
+ ## Bias, Risks, and Limitations
64
+
65
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
66
+
67
+ [More Information Needed]
68
+
69
+ ### Recommendations
70
+
71
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
72
+
73
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
74
+
75
+ ## How to Get Started with the Model
76
+
77
+ Use the code below to get started with the model.
78
+
79
+ [More Information Needed]
80
+
81
+ ## Training Details
82
+
83
+ ### Training Data
84
+
85
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
86
+
87
+ [More Information Needed]
88
+
89
+ ### Training Procedure
90
+
91
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
92
+
93
+ #### Preprocessing [optional]
94
+
95
+ [More Information Needed]
96
+
97
+
98
+ #### Training Hyperparameters
99
+
100
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
101
+
102
+ #### Speeds, Sizes, Times [optional]
103
+
104
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
105
+
106
+ [More Information Needed]
107
+
108
+ ## Evaluation
109
+
110
+ <!-- This section describes the evaluation protocols and provides the results. -->
111
+
112
+ ### Testing Data, Factors & Metrics
113
+
114
+ #### Testing Data
115
+
116
+ <!-- This should link to a Dataset Card if possible. -->
117
+
118
+ [More Information Needed]
119
+
120
+ #### Factors
121
+
122
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
123
+
124
+ [More Information Needed]
125
+
126
+ #### Metrics
127
+
128
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
129
+
130
+ [More Information Needed]
131
+
132
+ ### Results
133
+
134
+ [More Information Needed]
135
+
136
+ #### Summary
137
+
138
+
139
+
140
+ ## Model Examination [optional]
141
+
142
+ <!-- Relevant interpretability work for the model goes here -->
143
+
144
+ [More Information Needed]
145
+
146
+ ## Environmental Impact
147
+
148
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
149
+
150
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
151
+
152
+ - **Hardware Type:** [More Information Needed]
153
+ - **Hours used:** [More Information Needed]
154
+ - **Cloud Provider:** [More Information Needed]
155
+ - **Compute Region:** [More Information Needed]
156
+ - **Carbon Emitted:** [More Information Needed]
157
+
158
+ ## Technical Specifications [optional]
159
+
160
+ ### Model Architecture and Objective
161
+
162
+ [More Information Needed]
163
+
164
+ ### Compute Infrastructure
165
+
166
+ [More Information Needed]
167
+
168
+ #### Hardware
169
+
170
+ [More Information Needed]
171
+
172
+ #### Software
173
+
174
+ [More Information Needed]
175
+
176
+ ## Citation [optional]
177
+
178
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
179
+
180
+ **BibTeX:**
181
+
182
+ [More Information Needed]
183
+
184
+ **APA:**
185
+
186
+ [More Information Needed]
187
+
188
+ ## Glossary [optional]
189
+
190
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
191
+
192
+ [More Information Needed]
193
+
194
+ ## More Information [optional]
195
+
196
+ [More Information Needed]
197
+
198
+ ## Model Card Authors [optional]
199
+
200
+ [More Information Needed]
201
+
202
+ ## Model Card Contact
203
+
204
+ [More Information Needed]
205
+ ### Framework versions
206
+
207
+ - PEFT 0.18.0
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-Coder-7B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.0",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "down_proj",
34
+ "o_proj",
35
+ "k_proj",
36
+ "gate_proj",
37
+ "q_proj",
38
+ "v_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:612b030ce67864735881485d454ff41babafd018d7b4f86b95ea3b1b64db944f
3
+ size 161533192
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/README.md ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen2.5-Coder-7B-Instruct
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen2.5-Coder-7B-Instruct
7
+ - lora
8
+ - transformers
9
+ ---
10
+
11
+ # Model Card for Model ID
12
+
13
+ <!-- Provide a quick summary of what the model is/does. -->
14
+
15
+
16
+
17
+ ## Model Details
18
+
19
+ ### Model Description
20
+
21
+ <!-- Provide a longer summary of what this model is. -->
22
+
23
+
24
+
25
+ - **Developed by:** [More Information Needed]
26
+ - **Funded by [optional]:** [More Information Needed]
27
+ - **Shared by [optional]:** [More Information Needed]
28
+ - **Model type:** [More Information Needed]
29
+ - **Language(s) (NLP):** [More Information Needed]
30
+ - **License:** [More Information Needed]
31
+ - **Finetuned from model [optional]:** [More Information Needed]
32
+
33
+ ### Model Sources [optional]
34
+
35
+ <!-- Provide the basic links for the model. -->
36
+
37
+ - **Repository:** [More Information Needed]
38
+ - **Paper [optional]:** [More Information Needed]
39
+ - **Demo [optional]:** [More Information Needed]
40
+
41
+ ## Uses
42
+
43
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
44
+
45
+ ### Direct Use
46
+
47
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
48
+
49
+ [More Information Needed]
50
+
51
+ ### Downstream Use [optional]
52
+
53
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
54
+
55
+ [More Information Needed]
56
+
57
+ ### Out-of-Scope Use
58
+
59
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
60
+
61
+ [More Information Needed]
62
+
63
+ ## Bias, Risks, and Limitations
64
+
65
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
66
+
67
+ [More Information Needed]
68
+
69
+ ### Recommendations
70
+
71
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
72
+
73
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
74
+
75
+ ## How to Get Started with the Model
76
+
77
+ Use the code below to get started with the model.
78
+
79
+ [More Information Needed]
80
+
81
+ ## Training Details
82
+
83
+ ### Training Data
84
+
85
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
86
+
87
+ [More Information Needed]
88
+
89
+ ### Training Procedure
90
+
91
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
92
+
93
+ #### Preprocessing [optional]
94
+
95
+ [More Information Needed]
96
+
97
+
98
+ #### Training Hyperparameters
99
+
100
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
101
+
102
+ #### Speeds, Sizes, Times [optional]
103
+
104
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
105
+
106
+ [More Information Needed]
107
+
108
+ ## Evaluation
109
+
110
+ <!-- This section describes the evaluation protocols and provides the results. -->
111
+
112
+ ### Testing Data, Factors & Metrics
113
+
114
+ #### Testing Data
115
+
116
+ <!-- This should link to a Dataset Card if possible. -->
117
+
118
+ [More Information Needed]
119
+
120
+ #### Factors
121
+
122
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
123
+
124
+ [More Information Needed]
125
+
126
+ #### Metrics
127
+
128
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
129
+
130
+ [More Information Needed]
131
+
132
+ ### Results
133
+
134
+ [More Information Needed]
135
+
136
+ #### Summary
137
+
138
+
139
+
140
+ ## Model Examination [optional]
141
+
142
+ <!-- Relevant interpretability work for the model goes here -->
143
+
144
+ [More Information Needed]
145
+
146
+ ## Environmental Impact
147
+
148
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
149
+
150
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
151
+
152
+ - **Hardware Type:** [More Information Needed]
153
+ - **Hours used:** [More Information Needed]
154
+ - **Cloud Provider:** [More Information Needed]
155
+ - **Compute Region:** [More Information Needed]
156
+ - **Carbon Emitted:** [More Information Needed]
157
+
158
+ ## Technical Specifications [optional]
159
+
160
+ ### Model Architecture and Objective
161
+
162
+ [More Information Needed]
163
+
164
+ ### Compute Infrastructure
165
+
166
+ [More Information Needed]
167
+
168
+ #### Hardware
169
+
170
+ [More Information Needed]
171
+
172
+ #### Software
173
+
174
+ [More Information Needed]
175
+
176
+ ## Citation [optional]
177
+
178
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
179
+
180
+ **BibTeX:**
181
+
182
+ [More Information Needed]
183
+
184
+ **APA:**
185
+
186
+ [More Information Needed]
187
+
188
+ ## Glossary [optional]
189
+
190
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
191
+
192
+ [More Information Needed]
193
+
194
+ ## More Information [optional]
195
+
196
+ [More Information Needed]
197
+
198
+ ## Model Card Authors [optional]
199
+
200
+ [More Information Needed]
201
+
202
+ ## Model Card Contact
203
+
204
+ [More Information Needed]
205
+ ### Framework versions
206
+
207
+ - PEFT 0.18.0
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-Coder-7B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.0",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "down_proj",
34
+ "o_proj",
35
+ "k_proj",
36
+ "gate_proj",
37
+ "q_proj",
38
+ "v_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:48f8cd88f920d16215085e0d42bcd477c88d74c2b24526ed27219340b13575cc
3
+ size 161533192
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d3fb0cb47abd9bedc5ac1b1a230560a406e0b3eb3d8653109d60a013bf89ad39
3
+ size 323296891
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3351812c4159024f3fde05c12e79368e9b898e08ec0d655abfcb1219fa9251fc
3
+ size 15429
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/rng_state_1.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fb896a1096df8e9408c583245defe5c138a5a1572616310a0e0460e8def4c7cf
3
+ size 15429
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/rng_state_2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:33411d486ac95c675064862213b96c73d60599c7ed152c868c354eff8c6ec21d
3
+ size 15429
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/rng_state_3.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2ddcaf3db2ad21e36b43cb7bd66891d249fe07b12c84574b2df36df48f1d2e30
3
+ size 15429
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5c167a2d01592ceab1aae383137685e548e2c682643130fdc4d4f2ff5a122c81
3
+ size 1465
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|im_end|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
+ size 11421896
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/tokenizer_config.json ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ }
181
+ },
182
+ "additional_special_tokens": [
183
+ "<|im_start|>",
184
+ "<|im_end|>",
185
+ "<|object_ref_start|>",
186
+ "<|object_ref_end|>",
187
+ "<|box_start|>",
188
+ "<|box_end|>",
189
+ "<|quad_start|>",
190
+ "<|quad_end|>",
191
+ "<|vision_start|>",
192
+ "<|vision_end|>",
193
+ "<|vision_pad|>",
194
+ "<|image_pad|>",
195
+ "<|video_pad|>"
196
+ ],
197
+ "bos_token": null,
198
+ "clean_up_tokenization_spaces": false,
199
+ "eos_token": "<|im_end|>",
200
+ "errors": "replace",
201
+ "extra_special_tokens": {},
202
+ "model_max_length": 32768,
203
+ "pad_token": "<|endoftext|>",
204
+ "split_special_tokens": false,
205
+ "tokenizer_class": "Qwen2Tokenizer",
206
+ "unk_token": null
207
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/trainer_state.json ADDED
@@ -0,0 +1,749 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.8366450533361222,
6
+ "eval_steps": 1000,
7
+ "global_step": 1000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0008366450533361222,
14
+ "grad_norm": 0.43044784665107727,
15
+ "learning_rate": 0.0,
16
+ "loss": 0.971,
17
+ "step": 1
18
+ },
19
+ {
20
+ "epoch": 0.00836645053336122,
21
+ "grad_norm": 0.47508111596107483,
22
+ "learning_rate": 1.875e-07,
23
+ "loss": 1.025,
24
+ "step": 10
25
+ },
26
+ {
27
+ "epoch": 0.01673290106672244,
28
+ "grad_norm": 0.42676985263824463,
29
+ "learning_rate": 3.958333333333333e-07,
30
+ "loss": 0.9742,
31
+ "step": 20
32
+ },
33
+ {
34
+ "epoch": 0.025099351600083666,
35
+ "grad_norm": 0.46747535467147827,
36
+ "learning_rate": 6.041666666666666e-07,
37
+ "loss": 1.0238,
38
+ "step": 30
39
+ },
40
+ {
41
+ "epoch": 0.03346580213344488,
42
+ "grad_norm": 0.45301714539527893,
43
+ "learning_rate": 8.125e-07,
44
+ "loss": 1.0189,
45
+ "step": 40
46
+ },
47
+ {
48
+ "epoch": 0.04183225266680611,
49
+ "grad_norm": 0.49705836176872253,
50
+ "learning_rate": 9.999995509192137e-07,
51
+ "loss": 1.0155,
52
+ "step": 50
53
+ },
54
+ {
55
+ "epoch": 0.05019870320016733,
56
+ "grad_norm": 0.4859037697315216,
57
+ "learning_rate": 9.999456622009543e-07,
58
+ "loss": 0.9877,
59
+ "step": 60
60
+ },
61
+ {
62
+ "epoch": 0.05856515373352855,
63
+ "grad_norm": 0.47822538018226624,
64
+ "learning_rate": 9.998019684171583e-07,
65
+ "loss": 0.9869,
66
+ "step": 70
67
+ },
68
+ {
69
+ "epoch": 0.06693160426688977,
70
+ "grad_norm": 0.46603235602378845,
71
+ "learning_rate": 9.995684953794905e-07,
72
+ "loss": 0.9779,
73
+ "step": 80
74
+ },
75
+ {
76
+ "epoch": 0.075298054800251,
77
+ "grad_norm": 0.49305635690689087,
78
+ "learning_rate": 9.99245285026631e-07,
79
+ "loss": 0.9917,
80
+ "step": 90
81
+ },
82
+ {
83
+ "epoch": 0.08366450533361222,
84
+ "grad_norm": 0.5048590302467346,
85
+ "learning_rate": 9.988323954167438e-07,
86
+ "loss": 0.9838,
87
+ "step": 100
88
+ },
89
+ {
90
+ "epoch": 0.09203095586697344,
91
+ "grad_norm": 0.5521052479743958,
92
+ "learning_rate": 9.983299007170452e-07,
93
+ "loss": 1.0057,
94
+ "step": 110
95
+ },
96
+ {
97
+ "epoch": 0.10039740640033466,
98
+ "grad_norm": 0.5106943845748901,
99
+ "learning_rate": 9.97737891190484e-07,
100
+ "loss": 0.9732,
101
+ "step": 120
102
+ },
103
+ {
104
+ "epoch": 0.10876385693369588,
105
+ "grad_norm": 0.5572932362556458,
106
+ "learning_rate": 9.970564731795256e-07,
107
+ "loss": 0.9497,
108
+ "step": 130
109
+ },
110
+ {
111
+ "epoch": 0.1171303074670571,
112
+ "grad_norm": 0.5039992928504944,
113
+ "learning_rate": 9.962857690870505e-07,
114
+ "loss": 0.9497,
115
+ "step": 140
116
+ },
117
+ {
118
+ "epoch": 0.12549675800041832,
119
+ "grad_norm": 0.4970445930957794,
120
+ "learning_rate": 9.95425917354367e-07,
121
+ "loss": 0.918,
122
+ "step": 150
123
+ },
124
+ {
125
+ "epoch": 0.13386320853377953,
126
+ "grad_norm": 0.555020809173584,
127
+ "learning_rate": 9.944770724363427e-07,
128
+ "loss": 0.9078,
129
+ "step": 160
130
+ },
131
+ {
132
+ "epoch": 0.14222965906714077,
133
+ "grad_norm": 0.557834267616272,
134
+ "learning_rate": 9.934394047736607e-07,
135
+ "loss": 0.9089,
136
+ "step": 170
137
+ },
138
+ {
139
+ "epoch": 0.150596109600502,
140
+ "grad_norm": 0.5249358415603638,
141
+ "learning_rate": 9.923131007622025e-07,
142
+ "loss": 0.8769,
143
+ "step": 180
144
+ },
145
+ {
146
+ "epoch": 0.1589625601338632,
147
+ "grad_norm": 0.5492634177207947,
148
+ "learning_rate": 9.910983627195665e-07,
149
+ "loss": 0.8527,
150
+ "step": 190
151
+ },
152
+ {
153
+ "epoch": 0.16732901066722444,
154
+ "grad_norm": 0.542378842830658,
155
+ "learning_rate": 9.897954088487244e-07,
156
+ "loss": 0.8595,
157
+ "step": 200
158
+ },
159
+ {
160
+ "epoch": 0.17569546120058566,
161
+ "grad_norm": 0.5153322219848633,
162
+ "learning_rate": 9.884044731988276e-07,
163
+ "loss": 0.8369,
164
+ "step": 210
165
+ },
166
+ {
167
+ "epoch": 0.18406191173394687,
168
+ "grad_norm": 0.5284155011177063,
169
+ "learning_rate": 9.869258056231638e-07,
170
+ "loss": 0.811,
171
+ "step": 220
172
+ },
173
+ {
174
+ "epoch": 0.19242836226730808,
175
+ "grad_norm": 0.5867093205451965,
176
+ "learning_rate": 9.85359671734275e-07,
177
+ "loss": 0.794,
178
+ "step": 230
179
+ },
180
+ {
181
+ "epoch": 0.20079481280066933,
182
+ "grad_norm": 0.5008439421653748,
183
+ "learning_rate": 9.837063528562478e-07,
184
+ "loss": 0.7535,
185
+ "step": 240
186
+ },
187
+ {
188
+ "epoch": 0.20916126333403054,
189
+ "grad_norm": 0.5109360814094543,
190
+ "learning_rate": 9.819661459741772e-07,
191
+ "loss": 0.7542,
192
+ "step": 250
193
+ },
194
+ {
195
+ "epoch": 0.21752771386739175,
196
+ "grad_norm": 0.5287544131278992,
197
+ "learning_rate": 9.80139363680821e-07,
198
+ "loss": 0.7186,
199
+ "step": 260
200
+ },
201
+ {
202
+ "epoch": 0.22589416440075297,
203
+ "grad_norm": 0.5497297048568726,
204
+ "learning_rate": 9.782263341204475e-07,
205
+ "loss": 0.7025,
206
+ "step": 270
207
+ },
208
+ {
209
+ "epoch": 0.2342606149341142,
210
+ "grad_norm": 0.49839189648628235,
211
+ "learning_rate": 9.762274009298916e-07,
212
+ "loss": 0.6918,
213
+ "step": 280
214
+ },
215
+ {
216
+ "epoch": 0.24262706546747542,
217
+ "grad_norm": 0.5106262564659119,
218
+ "learning_rate": 9.741429231768277e-07,
219
+ "loss": 0.6795,
220
+ "step": 290
221
+ },
222
+ {
223
+ "epoch": 0.25099351600083664,
224
+ "grad_norm": 0.4450535476207733,
225
+ "learning_rate": 9.7197327529527e-07,
226
+ "loss": 0.6618,
227
+ "step": 300
228
+ },
229
+ {
230
+ "epoch": 0.25935996653419785,
231
+ "grad_norm": 0.44939276576042175,
232
+ "learning_rate": 9.697188470183136e-07,
233
+ "loss": 0.6448,
234
+ "step": 310
235
+ },
236
+ {
237
+ "epoch": 0.26772641706755906,
238
+ "grad_norm": 0.4323934018611908,
239
+ "learning_rate": 9.673800433081258e-07,
240
+ "loss": 0.6341,
241
+ "step": 320
242
+ },
243
+ {
244
+ "epoch": 0.27609286760092033,
245
+ "grad_norm": 0.40285560488700867,
246
+ "learning_rate": 9.649572842832048e-07,
247
+ "loss": 0.6171,
248
+ "step": 330
249
+ },
250
+ {
251
+ "epoch": 0.28445931813428155,
252
+ "grad_norm": 0.35840362310409546,
253
+ "learning_rate": 9.624510051429115e-07,
254
+ "loss": 0.6044,
255
+ "step": 340
256
+ },
257
+ {
258
+ "epoch": 0.29282576866764276,
259
+ "grad_norm": 0.34704065322875977,
260
+ "learning_rate": 9.598616560892975e-07,
261
+ "loss": 0.6028,
262
+ "step": 350
263
+ },
264
+ {
265
+ "epoch": 0.301192219201004,
266
+ "grad_norm": 0.3540046513080597,
267
+ "learning_rate": 9.571897022462327e-07,
268
+ "loss": 0.5803,
269
+ "step": 360
270
+ },
271
+ {
272
+ "epoch": 0.3095586697343652,
273
+ "grad_norm": 0.35438406467437744,
274
+ "learning_rate": 9.544356235758578e-07,
275
+ "loss": 0.5843,
276
+ "step": 370
277
+ },
278
+ {
279
+ "epoch": 0.3179251202677264,
280
+ "grad_norm": 0.3275851905345917,
281
+ "learning_rate": 9.515999147923665e-07,
282
+ "loss": 0.5793,
283
+ "step": 380
284
+ },
285
+ {
286
+ "epoch": 0.3262915708010876,
287
+ "grad_norm": 0.30544060468673706,
288
+ "learning_rate": 9.486830852731427e-07,
289
+ "loss": 0.5894,
290
+ "step": 390
291
+ },
292
+ {
293
+ "epoch": 0.3346580213344489,
294
+ "grad_norm": 0.30233773589134216,
295
+ "learning_rate": 9.456856589672584e-07,
296
+ "loss": 0.5592,
297
+ "step": 400
298
+ },
299
+ {
300
+ "epoch": 0.3430244718678101,
301
+ "grad_norm": 0.3126496970653534,
302
+ "learning_rate": 9.426081743013597e-07,
303
+ "loss": 0.561,
304
+ "step": 410
305
+ },
306
+ {
307
+ "epoch": 0.3513909224011713,
308
+ "grad_norm": 0.29356780648231506,
309
+ "learning_rate": 9.394511840829473e-07,
310
+ "loss": 0.5506,
311
+ "step": 420
312
+ },
313
+ {
314
+ "epoch": 0.3597573729345325,
315
+ "grad_norm": 0.2923426330089569,
316
+ "learning_rate": 9.362152554010769e-07,
317
+ "loss": 0.5475,
318
+ "step": 430
319
+ },
320
+ {
321
+ "epoch": 0.36812382346789374,
322
+ "grad_norm": 0.3088696002960205,
323
+ "learning_rate": 9.329009695244927e-07,
324
+ "loss": 0.5252,
325
+ "step": 440
326
+ },
327
+ {
328
+ "epoch": 0.37649027400125495,
329
+ "grad_norm": 0.2797798216342926,
330
+ "learning_rate": 9.29508921797215e-07,
331
+ "loss": 0.5401,
332
+ "step": 450
333
+ },
334
+ {
335
+ "epoch": 0.38485672453461617,
336
+ "grad_norm": 0.2643727958202362,
337
+ "learning_rate": 9.260397215315982e-07,
338
+ "loss": 0.5256,
339
+ "step": 460
340
+ },
341
+ {
342
+ "epoch": 0.39322317506797744,
343
+ "grad_norm": 0.2582707703113556,
344
+ "learning_rate": 9.224939918988799e-07,
345
+ "loss": 0.5276,
346
+ "step": 470
347
+ },
348
+ {
349
+ "epoch": 0.40158962560133865,
350
+ "grad_norm": 0.23798155784606934,
351
+ "learning_rate": 9.188723698172419e-07,
352
+ "loss": 0.5129,
353
+ "step": 480
354
+ },
355
+ {
356
+ "epoch": 0.40995607613469986,
357
+ "grad_norm": 0.2367415726184845,
358
+ "learning_rate": 9.151755058373999e-07,
359
+ "loss": 0.5229,
360
+ "step": 490
361
+ },
362
+ {
363
+ "epoch": 0.4183225266680611,
364
+ "grad_norm": 0.21267226338386536,
365
+ "learning_rate": 9.114040640257457e-07,
366
+ "loss": 0.5239,
367
+ "step": 500
368
+ },
369
+ {
370
+ "epoch": 0.4266889772014223,
371
+ "grad_norm": 0.2113737016916275,
372
+ "learning_rate": 9.07558721845061e-07,
373
+ "loss": 0.5004,
374
+ "step": 510
375
+ },
376
+ {
377
+ "epoch": 0.4350554277347835,
378
+ "grad_norm": 0.19743551313877106,
379
+ "learning_rate": 9.036401700328253e-07,
380
+ "loss": 0.51,
381
+ "step": 520
382
+ },
383
+ {
384
+ "epoch": 0.4434218782681447,
385
+ "grad_norm": 0.17708665132522583,
386
+ "learning_rate": 8.996491124771386e-07,
387
+ "loss": 0.4916,
388
+ "step": 530
389
+ },
390
+ {
391
+ "epoch": 0.45178832880150593,
392
+ "grad_norm": 0.19286687672138214,
393
+ "learning_rate": 8.955862660902827e-07,
394
+ "loss": 0.4972,
395
+ "step": 540
396
+ },
397
+ {
398
+ "epoch": 0.4601547793348672,
399
+ "grad_norm": 0.18794295191764832,
400
+ "learning_rate": 8.914523606799416e-07,
401
+ "loss": 0.4958,
402
+ "step": 550
403
+ },
404
+ {
405
+ "epoch": 0.4685212298682284,
406
+ "grad_norm": 0.17703424394130707,
407
+ "learning_rate": 8.872481388181076e-07,
408
+ "loss": 0.5041,
409
+ "step": 560
410
+ },
411
+ {
412
+ "epoch": 0.47688768040158963,
413
+ "grad_norm": 0.1806008517742157,
414
+ "learning_rate": 8.829743557076924e-07,
415
+ "loss": 0.4786,
416
+ "step": 570
417
+ },
418
+ {
419
+ "epoch": 0.48525413093495084,
420
+ "grad_norm": 0.17116817831993103,
421
+ "learning_rate": 8.786317790468707e-07,
422
+ "loss": 0.4771,
423
+ "step": 580
424
+ },
425
+ {
426
+ "epoch": 0.49362058146831206,
427
+ "grad_norm": 0.16165761649608612,
428
+ "learning_rate": 8.742211888911788e-07,
429
+ "loss": 0.4831,
430
+ "step": 590
431
+ },
432
+ {
433
+ "epoch": 0.5019870320016733,
434
+ "grad_norm": 0.16249604523181915,
435
+ "learning_rate": 8.697433775133934e-07,
436
+ "loss": 0.4703,
437
+ "step": 600
438
+ },
439
+ {
440
+ "epoch": 0.5103534825350345,
441
+ "grad_norm": 0.16176247596740723,
442
+ "learning_rate": 8.651991492612158e-07,
443
+ "loss": 0.4898,
444
+ "step": 610
445
+ },
446
+ {
447
+ "epoch": 0.5187199330683957,
448
+ "grad_norm": 0.17271846532821655,
449
+ "learning_rate": 8.605893204127876e-07,
450
+ "loss": 0.4783,
451
+ "step": 620
452
+ },
453
+ {
454
+ "epoch": 0.527086383601757,
455
+ "grad_norm": 0.1579618752002716,
456
+ "learning_rate": 8.559147190300629e-07,
457
+ "loss": 0.4685,
458
+ "step": 630
459
+ },
460
+ {
461
+ "epoch": 0.5354528341351181,
462
+ "grad_norm": 0.14920172095298767,
463
+ "learning_rate": 8.511761848100628e-07,
464
+ "loss": 0.468,
465
+ "step": 640
466
+ },
467
+ {
468
+ "epoch": 0.5438192846684794,
469
+ "grad_norm": 0.15946857631206512,
470
+ "learning_rate": 8.463745689340426e-07,
471
+ "loss": 0.4669,
472
+ "step": 650
473
+ },
474
+ {
475
+ "epoch": 0.5521857352018407,
476
+ "grad_norm": 0.14744576811790466,
477
+ "learning_rate": 8.415107339145932e-07,
478
+ "loss": 0.4709,
479
+ "step": 660
480
+ },
481
+ {
482
+ "epoch": 0.5605521857352018,
483
+ "grad_norm": 0.13913576304912567,
484
+ "learning_rate": 8.365855534407088e-07,
485
+ "loss": 0.4711,
486
+ "step": 670
487
+ },
488
+ {
489
+ "epoch": 0.5689186362685631,
490
+ "grad_norm": 0.14849849045276642,
491
+ "learning_rate": 8.315999122208458e-07,
492
+ "loss": 0.4739,
493
+ "step": 680
494
+ },
495
+ {
496
+ "epoch": 0.5772850868019243,
497
+ "grad_norm": 0.1468994915485382,
498
+ "learning_rate": 8.265547058240038e-07,
499
+ "loss": 0.47,
500
+ "step": 690
501
+ },
502
+ {
503
+ "epoch": 0.5856515373352855,
504
+ "grad_norm": 0.1443009227514267,
505
+ "learning_rate": 8.214508405188542e-07,
506
+ "loss": 0.4706,
507
+ "step": 700
508
+ },
509
+ {
510
+ "epoch": 0.5940179878686467,
511
+ "grad_norm": 0.13816598057746887,
512
+ "learning_rate": 8.162892331109481e-07,
513
+ "loss": 0.4625,
514
+ "step": 710
515
+ },
516
+ {
517
+ "epoch": 0.602384438402008,
518
+ "grad_norm": 0.13011474907398224,
519
+ "learning_rate": 8.110708107780299e-07,
520
+ "loss": 0.4527,
521
+ "step": 720
522
+ },
523
+ {
524
+ "epoch": 0.6107508889353692,
525
+ "grad_norm": 0.13332195580005646,
526
+ "learning_rate": 8.057965109034896e-07,
527
+ "loss": 0.4527,
528
+ "step": 730
529
+ },
530
+ {
531
+ "epoch": 0.6191173394687304,
532
+ "grad_norm": 0.12983644008636475,
533
+ "learning_rate": 8.004672809079807e-07,
534
+ "loss": 0.4486,
535
+ "step": 740
536
+ },
537
+ {
538
+ "epoch": 0.6274837900020916,
539
+ "grad_norm": 0.13085156679153442,
540
+ "learning_rate": 7.950840780792347e-07,
541
+ "loss": 0.4516,
542
+ "step": 750
543
+ },
544
+ {
545
+ "epoch": 0.6358502405354528,
546
+ "grad_norm": 0.12932556867599487,
547
+ "learning_rate": 7.896478694001047e-07,
548
+ "loss": 0.4515,
549
+ "step": 760
550
+ },
551
+ {
552
+ "epoch": 0.6442166910688141,
553
+ "grad_norm": 0.13092581927776337,
554
+ "learning_rate": 7.841596313748666e-07,
555
+ "loss": 0.4475,
556
+ "step": 770
557
+ },
558
+ {
559
+ "epoch": 0.6525831416021752,
560
+ "grad_norm": 0.1348915696144104,
561
+ "learning_rate": 7.786203498538093e-07,
562
+ "loss": 0.4471,
563
+ "step": 780
564
+ },
565
+ {
566
+ "epoch": 0.6609495921355365,
567
+ "grad_norm": 0.12818686664104462,
568
+ "learning_rate": 7.730310198561469e-07,
569
+ "loss": 0.4443,
570
+ "step": 790
571
+ },
572
+ {
573
+ "epoch": 0.6693160426688978,
574
+ "grad_norm": 0.12480778992176056,
575
+ "learning_rate": 7.673926453912845e-07,
576
+ "loss": 0.4536,
577
+ "step": 800
578
+ },
579
+ {
580
+ "epoch": 0.6776824932022589,
581
+ "grad_norm": 0.12339136004447937,
582
+ "learning_rate": 7.617062392784671e-07,
583
+ "loss": 0.4452,
584
+ "step": 810
585
+ },
586
+ {
587
+ "epoch": 0.6860489437356202,
588
+ "grad_norm": 0.1215905249118805,
589
+ "learning_rate": 7.559728229648488e-07,
590
+ "loss": 0.4456,
591
+ "step": 820
592
+ },
593
+ {
594
+ "epoch": 0.6944153942689814,
595
+ "grad_norm": 0.12470731884241104,
596
+ "learning_rate": 7.501934263420088e-07,
597
+ "loss": 0.4511,
598
+ "step": 830
599
+ },
600
+ {
601
+ "epoch": 0.7027818448023426,
602
+ "grad_norm": 0.12183189392089844,
603
+ "learning_rate": 7.443690875609541e-07,
604
+ "loss": 0.4497,
605
+ "step": 840
606
+ },
607
+ {
608
+ "epoch": 0.7111482953357038,
609
+ "grad_norm": 0.1178760975599289,
610
+ "learning_rate": 7.385008528456356e-07,
611
+ "loss": 0.438,
612
+ "step": 850
613
+ },
614
+ {
615
+ "epoch": 0.719514745869065,
616
+ "grad_norm": 0.11445871740579605,
617
+ "learning_rate": 7.325897763050154e-07,
618
+ "loss": 0.4468,
619
+ "step": 860
620
+ },
621
+ {
622
+ "epoch": 0.7278811964024263,
623
+ "grad_norm": 0.11578909307718277,
624
+ "learning_rate": 7.266369197437181e-07,
625
+ "loss": 0.4388,
626
+ "step": 870
627
+ },
628
+ {
629
+ "epoch": 0.7362476469357875,
630
+ "grad_norm": 0.11665735393762589,
631
+ "learning_rate": 7.206433524712988e-07,
632
+ "loss": 0.4401,
633
+ "step": 880
634
+ },
635
+ {
636
+ "epoch": 0.7446140974691488,
637
+ "grad_norm": 0.11469507217407227,
638
+ "learning_rate": 7.146101511101635e-07,
639
+ "loss": 0.449,
640
+ "step": 890
641
+ },
642
+ {
643
+ "epoch": 0.7529805480025099,
644
+ "grad_norm": 0.11232412606477737,
645
+ "learning_rate": 7.085383994021756e-07,
646
+ "loss": 0.4355,
647
+ "step": 900
648
+ },
649
+ {
650
+ "epoch": 0.7613469985358712,
651
+ "grad_norm": 0.10072898119688034,
652
+ "learning_rate": 7.024291880139842e-07,
653
+ "loss": 0.4418,
654
+ "step": 910
655
+ },
656
+ {
657
+ "epoch": 0.7697134490692323,
658
+ "grad_norm": 0.11294315755367279,
659
+ "learning_rate": 6.962836143411077e-07,
660
+ "loss": 0.4369,
661
+ "step": 920
662
+ },
663
+ {
664
+ "epoch": 0.7780798996025936,
665
+ "grad_norm": 0.10808352380990982,
666
+ "learning_rate": 6.901027823108087e-07,
667
+ "loss": 0.4479,
668
+ "step": 930
669
+ },
670
+ {
671
+ "epoch": 0.7864463501359549,
672
+ "grad_norm": 0.10845589637756348,
673
+ "learning_rate": 6.838878021837968e-07,
674
+ "loss": 0.431,
675
+ "step": 940
676
+ },
677
+ {
678
+ "epoch": 0.794812800669316,
679
+ "grad_norm": 0.1148146539926529,
680
+ "learning_rate": 6.776397903547918e-07,
681
+ "loss": 0.4241,
682
+ "step": 950
683
+ },
684
+ {
685
+ "epoch": 0.8031792512026773,
686
+ "grad_norm": 0.10050572454929352,
687
+ "learning_rate": 6.713598691519873e-07,
688
+ "loss": 0.4326,
689
+ "step": 960
690
+ },
691
+ {
692
+ "epoch": 0.8115457017360385,
693
+ "grad_norm": 0.10299045592546463,
694
+ "learning_rate": 6.650491666354458e-07,
695
+ "loss": 0.4332,
696
+ "step": 970
697
+ },
698
+ {
699
+ "epoch": 0.8199121522693997,
700
+ "grad_norm": 0.1057298332452774,
701
+ "learning_rate": 6.587088163944676e-07,
702
+ "loss": 0.4271,
703
+ "step": 980
704
+ },
705
+ {
706
+ "epoch": 0.8282786028027609,
707
+ "grad_norm": 0.1022634506225586,
708
+ "learning_rate": 6.523399573439621e-07,
709
+ "loss": 0.4296,
710
+ "step": 990
711
+ },
712
+ {
713
+ "epoch": 0.8366450533361222,
714
+ "grad_norm": 0.10467051714658737,
715
+ "learning_rate": 6.459437335198675e-07,
716
+ "loss": 0.4348,
717
+ "step": 1000
718
+ },
719
+ {
720
+ "epoch": 0.8366450533361222,
721
+ "eval_loss": 0.3649196922779083,
722
+ "eval_runtime": 2.5049,
723
+ "eval_samples_per_second": 79.045,
724
+ "eval_steps_per_second": 2.795,
725
+ "step": 1000
726
+ }
727
+ ],
728
+ "logging_steps": 10,
729
+ "max_steps": 2392,
730
+ "num_input_tokens_seen": 0,
731
+ "num_train_epochs": 2,
732
+ "save_steps": 1000,
733
+ "stateful_callbacks": {
734
+ "TrainerControl": {
735
+ "args": {
736
+ "should_epoch_stop": false,
737
+ "should_evaluate": false,
738
+ "should_log": false,
739
+ "should_save": true,
740
+ "should_training_stop": false
741
+ },
742
+ "attributes": {}
743
+ }
744
+ },
745
+ "total_flos": 1.037929133414495e+19,
746
+ "train_batch_size": 8,
747
+ "trial_name": null,
748
+ "trial_params": null
749
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fec14acee5ae660b18bff7cbae869950b6c5260e6a5f1f23db0eac1c537c6c4f
3
+ size 6417
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-1000/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/README.md ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen2.5-Coder-7B-Instruct
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen2.5-Coder-7B-Instruct
7
+ - lora
8
+ - transformers
9
+ ---
10
+
11
+ # Model Card for Model ID
12
+
13
+ <!-- Provide a quick summary of what the model is/does. -->
14
+
15
+
16
+
17
+ ## Model Details
18
+
19
+ ### Model Description
20
+
21
+ <!-- Provide a longer summary of what this model is. -->
22
+
23
+
24
+
25
+ - **Developed by:** [More Information Needed]
26
+ - **Funded by [optional]:** [More Information Needed]
27
+ - **Shared by [optional]:** [More Information Needed]
28
+ - **Model type:** [More Information Needed]
29
+ - **Language(s) (NLP):** [More Information Needed]
30
+ - **License:** [More Information Needed]
31
+ - **Finetuned from model [optional]:** [More Information Needed]
32
+
33
+ ### Model Sources [optional]
34
+
35
+ <!-- Provide the basic links for the model. -->
36
+
37
+ - **Repository:** [More Information Needed]
38
+ - **Paper [optional]:** [More Information Needed]
39
+ - **Demo [optional]:** [More Information Needed]
40
+
41
+ ## Uses
42
+
43
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
44
+
45
+ ### Direct Use
46
+
47
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
48
+
49
+ [More Information Needed]
50
+
51
+ ### Downstream Use [optional]
52
+
53
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
54
+
55
+ [More Information Needed]
56
+
57
+ ### Out-of-Scope Use
58
+
59
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
60
+
61
+ [More Information Needed]
62
+
63
+ ## Bias, Risks, and Limitations
64
+
65
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
66
+
67
+ [More Information Needed]
68
+
69
+ ### Recommendations
70
+
71
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
72
+
73
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
74
+
75
+ ## How to Get Started with the Model
76
+
77
+ Use the code below to get started with the model.
78
+
79
+ [More Information Needed]
80
+
81
+ ## Training Details
82
+
83
+ ### Training Data
84
+
85
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
86
+
87
+ [More Information Needed]
88
+
89
+ ### Training Procedure
90
+
91
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
92
+
93
+ #### Preprocessing [optional]
94
+
95
+ [More Information Needed]
96
+
97
+
98
+ #### Training Hyperparameters
99
+
100
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
101
+
102
+ #### Speeds, Sizes, Times [optional]
103
+
104
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
105
+
106
+ [More Information Needed]
107
+
108
+ ## Evaluation
109
+
110
+ <!-- This section describes the evaluation protocols and provides the results. -->
111
+
112
+ ### Testing Data, Factors & Metrics
113
+
114
+ #### Testing Data
115
+
116
+ <!-- This should link to a Dataset Card if possible. -->
117
+
118
+ [More Information Needed]
119
+
120
+ #### Factors
121
+
122
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
123
+
124
+ [More Information Needed]
125
+
126
+ #### Metrics
127
+
128
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
129
+
130
+ [More Information Needed]
131
+
132
+ ### Results
133
+
134
+ [More Information Needed]
135
+
136
+ #### Summary
137
+
138
+
139
+
140
+ ## Model Examination [optional]
141
+
142
+ <!-- Relevant interpretability work for the model goes here -->
143
+
144
+ [More Information Needed]
145
+
146
+ ## Environmental Impact
147
+
148
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
149
+
150
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
151
+
152
+ - **Hardware Type:** [More Information Needed]
153
+ - **Hours used:** [More Information Needed]
154
+ - **Cloud Provider:** [More Information Needed]
155
+ - **Compute Region:** [More Information Needed]
156
+ - **Carbon Emitted:** [More Information Needed]
157
+
158
+ ## Technical Specifications [optional]
159
+
160
+ ### Model Architecture and Objective
161
+
162
+ [More Information Needed]
163
+
164
+ ### Compute Infrastructure
165
+
166
+ [More Information Needed]
167
+
168
+ #### Hardware
169
+
170
+ [More Information Needed]
171
+
172
+ #### Software
173
+
174
+ [More Information Needed]
175
+
176
+ ## Citation [optional]
177
+
178
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
179
+
180
+ **BibTeX:**
181
+
182
+ [More Information Needed]
183
+
184
+ **APA:**
185
+
186
+ [More Information Needed]
187
+
188
+ ## Glossary [optional]
189
+
190
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
191
+
192
+ [More Information Needed]
193
+
194
+ ## More Information [optional]
195
+
196
+ [More Information Needed]
197
+
198
+ ## Model Card Authors [optional]
199
+
200
+ [More Information Needed]
201
+
202
+ ## Model Card Contact
203
+
204
+ [More Information Needed]
205
+ ### Framework versions
206
+
207
+ - PEFT 0.18.0
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-Coder-7B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.0",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "down_proj",
34
+ "o_proj",
35
+ "k_proj",
36
+ "gate_proj",
37
+ "q_proj",
38
+ "v_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3aa9e87fd78e2f4c9fb3e836de7160dc681d9738f8de777a29d855ed7aba88a3
3
+ size 161533192
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a21e8717e2c618402d5d7873ee26842b4c2feb0eaaa9870aaca81f87f28c21b6
3
+ size 323296891
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1492c39c82776763a089e674886f0ed863a7166b2f75a0d47ee7bf98370f1118
3
+ size 15429
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/rng_state_1.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8875bd971154fca07bc7028fcee980311970172a7305d713a5b80e0cff6d6923
3
+ size 15429
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/rng_state_2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:52dd6aaa943a8a63c90400cdbdbfd4d450cf8a2585a0255c38aa7bb755c46445
3
+ size 15429
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/rng_state_3.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e8d5d3f9a3c68075d4b5d37072c54ab34d1c65d77c91b5b5182cfc6ef849ea81
3
+ size 15429
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bd7176dfe2567a3dd43eae0fe45647d4fa466e69fa8d9566ae8b63fde93d0477
3
+ size 1465
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|im_end|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
+ size 11421896
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/tokenizer_config.json ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ }
181
+ },
182
+ "additional_special_tokens": [
183
+ "<|im_start|>",
184
+ "<|im_end|>",
185
+ "<|object_ref_start|>",
186
+ "<|object_ref_end|>",
187
+ "<|box_start|>",
188
+ "<|box_end|>",
189
+ "<|quad_start|>",
190
+ "<|quad_end|>",
191
+ "<|vision_start|>",
192
+ "<|vision_end|>",
193
+ "<|vision_pad|>",
194
+ "<|image_pad|>",
195
+ "<|video_pad|>"
196
+ ],
197
+ "bos_token": null,
198
+ "clean_up_tokenization_spaces": false,
199
+ "eos_token": "<|im_end|>",
200
+ "errors": "replace",
201
+ "extra_special_tokens": {},
202
+ "model_max_length": 32768,
203
+ "pad_token": "<|endoftext|>",
204
+ "split_special_tokens": false,
205
+ "tokenizer_class": "Qwen2Tokenizer",
206
+ "unk_token": null
207
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/trainer_state.json ADDED
@@ -0,0 +1,1457 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 1.672662622882242,
6
+ "eval_steps": 1000,
7
+ "global_step": 2000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0008366450533361222,
14
+ "grad_norm": 0.43044784665107727,
15
+ "learning_rate": 0.0,
16
+ "loss": 0.971,
17
+ "step": 1
18
+ },
19
+ {
20
+ "epoch": 0.00836645053336122,
21
+ "grad_norm": 0.47508111596107483,
22
+ "learning_rate": 1.875e-07,
23
+ "loss": 1.025,
24
+ "step": 10
25
+ },
26
+ {
27
+ "epoch": 0.01673290106672244,
28
+ "grad_norm": 0.42676985263824463,
29
+ "learning_rate": 3.958333333333333e-07,
30
+ "loss": 0.9742,
31
+ "step": 20
32
+ },
33
+ {
34
+ "epoch": 0.025099351600083666,
35
+ "grad_norm": 0.46747535467147827,
36
+ "learning_rate": 6.041666666666666e-07,
37
+ "loss": 1.0238,
38
+ "step": 30
39
+ },
40
+ {
41
+ "epoch": 0.03346580213344488,
42
+ "grad_norm": 0.45301714539527893,
43
+ "learning_rate": 8.125e-07,
44
+ "loss": 1.0189,
45
+ "step": 40
46
+ },
47
+ {
48
+ "epoch": 0.04183225266680611,
49
+ "grad_norm": 0.49705836176872253,
50
+ "learning_rate": 9.999995509192137e-07,
51
+ "loss": 1.0155,
52
+ "step": 50
53
+ },
54
+ {
55
+ "epoch": 0.05019870320016733,
56
+ "grad_norm": 0.4859037697315216,
57
+ "learning_rate": 9.999456622009543e-07,
58
+ "loss": 0.9877,
59
+ "step": 60
60
+ },
61
+ {
62
+ "epoch": 0.05856515373352855,
63
+ "grad_norm": 0.47822538018226624,
64
+ "learning_rate": 9.998019684171583e-07,
65
+ "loss": 0.9869,
66
+ "step": 70
67
+ },
68
+ {
69
+ "epoch": 0.06693160426688977,
70
+ "grad_norm": 0.46603235602378845,
71
+ "learning_rate": 9.995684953794905e-07,
72
+ "loss": 0.9779,
73
+ "step": 80
74
+ },
75
+ {
76
+ "epoch": 0.075298054800251,
77
+ "grad_norm": 0.49305635690689087,
78
+ "learning_rate": 9.99245285026631e-07,
79
+ "loss": 0.9917,
80
+ "step": 90
81
+ },
82
+ {
83
+ "epoch": 0.08366450533361222,
84
+ "grad_norm": 0.5048590302467346,
85
+ "learning_rate": 9.988323954167438e-07,
86
+ "loss": 0.9838,
87
+ "step": 100
88
+ },
89
+ {
90
+ "epoch": 0.09203095586697344,
91
+ "grad_norm": 0.5521052479743958,
92
+ "learning_rate": 9.983299007170452e-07,
93
+ "loss": 1.0057,
94
+ "step": 110
95
+ },
96
+ {
97
+ "epoch": 0.10039740640033466,
98
+ "grad_norm": 0.5106943845748901,
99
+ "learning_rate": 9.97737891190484e-07,
100
+ "loss": 0.9732,
101
+ "step": 120
102
+ },
103
+ {
104
+ "epoch": 0.10876385693369588,
105
+ "grad_norm": 0.5572932362556458,
106
+ "learning_rate": 9.970564731795256e-07,
107
+ "loss": 0.9497,
108
+ "step": 130
109
+ },
110
+ {
111
+ "epoch": 0.1171303074670571,
112
+ "grad_norm": 0.5039992928504944,
113
+ "learning_rate": 9.962857690870505e-07,
114
+ "loss": 0.9497,
115
+ "step": 140
116
+ },
117
+ {
118
+ "epoch": 0.12549675800041832,
119
+ "grad_norm": 0.4970445930957794,
120
+ "learning_rate": 9.95425917354367e-07,
121
+ "loss": 0.918,
122
+ "step": 150
123
+ },
124
+ {
125
+ "epoch": 0.13386320853377953,
126
+ "grad_norm": 0.555020809173584,
127
+ "learning_rate": 9.944770724363427e-07,
128
+ "loss": 0.9078,
129
+ "step": 160
130
+ },
131
+ {
132
+ "epoch": 0.14222965906714077,
133
+ "grad_norm": 0.557834267616272,
134
+ "learning_rate": 9.934394047736607e-07,
135
+ "loss": 0.9089,
136
+ "step": 170
137
+ },
138
+ {
139
+ "epoch": 0.150596109600502,
140
+ "grad_norm": 0.5249358415603638,
141
+ "learning_rate": 9.923131007622025e-07,
142
+ "loss": 0.8769,
143
+ "step": 180
144
+ },
145
+ {
146
+ "epoch": 0.1589625601338632,
147
+ "grad_norm": 0.5492634177207947,
148
+ "learning_rate": 9.910983627195665e-07,
149
+ "loss": 0.8527,
150
+ "step": 190
151
+ },
152
+ {
153
+ "epoch": 0.16732901066722444,
154
+ "grad_norm": 0.542378842830658,
155
+ "learning_rate": 9.897954088487244e-07,
156
+ "loss": 0.8595,
157
+ "step": 200
158
+ },
159
+ {
160
+ "epoch": 0.17569546120058566,
161
+ "grad_norm": 0.5153322219848633,
162
+ "learning_rate": 9.884044731988276e-07,
163
+ "loss": 0.8369,
164
+ "step": 210
165
+ },
166
+ {
167
+ "epoch": 0.18406191173394687,
168
+ "grad_norm": 0.5284155011177063,
169
+ "learning_rate": 9.869258056231638e-07,
170
+ "loss": 0.811,
171
+ "step": 220
172
+ },
173
+ {
174
+ "epoch": 0.19242836226730808,
175
+ "grad_norm": 0.5867093205451965,
176
+ "learning_rate": 9.85359671734275e-07,
177
+ "loss": 0.794,
178
+ "step": 230
179
+ },
180
+ {
181
+ "epoch": 0.20079481280066933,
182
+ "grad_norm": 0.5008439421653748,
183
+ "learning_rate": 9.837063528562478e-07,
184
+ "loss": 0.7535,
185
+ "step": 240
186
+ },
187
+ {
188
+ "epoch": 0.20916126333403054,
189
+ "grad_norm": 0.5109360814094543,
190
+ "learning_rate": 9.819661459741772e-07,
191
+ "loss": 0.7542,
192
+ "step": 250
193
+ },
194
+ {
195
+ "epoch": 0.21752771386739175,
196
+ "grad_norm": 0.5287544131278992,
197
+ "learning_rate": 9.80139363680821e-07,
198
+ "loss": 0.7186,
199
+ "step": 260
200
+ },
201
+ {
202
+ "epoch": 0.22589416440075297,
203
+ "grad_norm": 0.5497297048568726,
204
+ "learning_rate": 9.782263341204475e-07,
205
+ "loss": 0.7025,
206
+ "step": 270
207
+ },
208
+ {
209
+ "epoch": 0.2342606149341142,
210
+ "grad_norm": 0.49839189648628235,
211
+ "learning_rate": 9.762274009298916e-07,
212
+ "loss": 0.6918,
213
+ "step": 280
214
+ },
215
+ {
216
+ "epoch": 0.24262706546747542,
217
+ "grad_norm": 0.5106262564659119,
218
+ "learning_rate": 9.741429231768277e-07,
219
+ "loss": 0.6795,
220
+ "step": 290
221
+ },
222
+ {
223
+ "epoch": 0.25099351600083664,
224
+ "grad_norm": 0.4450535476207733,
225
+ "learning_rate": 9.7197327529527e-07,
226
+ "loss": 0.6618,
227
+ "step": 300
228
+ },
229
+ {
230
+ "epoch": 0.25935996653419785,
231
+ "grad_norm": 0.44939276576042175,
232
+ "learning_rate": 9.697188470183136e-07,
233
+ "loss": 0.6448,
234
+ "step": 310
235
+ },
236
+ {
237
+ "epoch": 0.26772641706755906,
238
+ "grad_norm": 0.4323934018611908,
239
+ "learning_rate": 9.673800433081258e-07,
240
+ "loss": 0.6341,
241
+ "step": 320
242
+ },
243
+ {
244
+ "epoch": 0.27609286760092033,
245
+ "grad_norm": 0.40285560488700867,
246
+ "learning_rate": 9.649572842832048e-07,
247
+ "loss": 0.6171,
248
+ "step": 330
249
+ },
250
+ {
251
+ "epoch": 0.28445931813428155,
252
+ "grad_norm": 0.35840362310409546,
253
+ "learning_rate": 9.624510051429115e-07,
254
+ "loss": 0.6044,
255
+ "step": 340
256
+ },
257
+ {
258
+ "epoch": 0.29282576866764276,
259
+ "grad_norm": 0.34704065322875977,
260
+ "learning_rate": 9.598616560892975e-07,
261
+ "loss": 0.6028,
262
+ "step": 350
263
+ },
264
+ {
265
+ "epoch": 0.301192219201004,
266
+ "grad_norm": 0.3540046513080597,
267
+ "learning_rate": 9.571897022462327e-07,
268
+ "loss": 0.5803,
269
+ "step": 360
270
+ },
271
+ {
272
+ "epoch": 0.3095586697343652,
273
+ "grad_norm": 0.35438406467437744,
274
+ "learning_rate": 9.544356235758578e-07,
275
+ "loss": 0.5843,
276
+ "step": 370
277
+ },
278
+ {
279
+ "epoch": 0.3179251202677264,
280
+ "grad_norm": 0.3275851905345917,
281
+ "learning_rate": 9.515999147923665e-07,
282
+ "loss": 0.5793,
283
+ "step": 380
284
+ },
285
+ {
286
+ "epoch": 0.3262915708010876,
287
+ "grad_norm": 0.30544060468673706,
288
+ "learning_rate": 9.486830852731427e-07,
289
+ "loss": 0.5894,
290
+ "step": 390
291
+ },
292
+ {
293
+ "epoch": 0.3346580213344489,
294
+ "grad_norm": 0.30233773589134216,
295
+ "learning_rate": 9.456856589672584e-07,
296
+ "loss": 0.5592,
297
+ "step": 400
298
+ },
299
+ {
300
+ "epoch": 0.3430244718678101,
301
+ "grad_norm": 0.3126496970653534,
302
+ "learning_rate": 9.426081743013597e-07,
303
+ "loss": 0.561,
304
+ "step": 410
305
+ },
306
+ {
307
+ "epoch": 0.3513909224011713,
308
+ "grad_norm": 0.29356780648231506,
309
+ "learning_rate": 9.394511840829473e-07,
310
+ "loss": 0.5506,
311
+ "step": 420
312
+ },
313
+ {
314
+ "epoch": 0.3597573729345325,
315
+ "grad_norm": 0.2923426330089569,
316
+ "learning_rate": 9.362152554010769e-07,
317
+ "loss": 0.5475,
318
+ "step": 430
319
+ },
320
+ {
321
+ "epoch": 0.36812382346789374,
322
+ "grad_norm": 0.3088696002960205,
323
+ "learning_rate": 9.329009695244927e-07,
324
+ "loss": 0.5252,
325
+ "step": 440
326
+ },
327
+ {
328
+ "epoch": 0.37649027400125495,
329
+ "grad_norm": 0.2797798216342926,
330
+ "learning_rate": 9.29508921797215e-07,
331
+ "loss": 0.5401,
332
+ "step": 450
333
+ },
334
+ {
335
+ "epoch": 0.38485672453461617,
336
+ "grad_norm": 0.2643727958202362,
337
+ "learning_rate": 9.260397215315982e-07,
338
+ "loss": 0.5256,
339
+ "step": 460
340
+ },
341
+ {
342
+ "epoch": 0.39322317506797744,
343
+ "grad_norm": 0.2582707703113556,
344
+ "learning_rate": 9.224939918988799e-07,
345
+ "loss": 0.5276,
346
+ "step": 470
347
+ },
348
+ {
349
+ "epoch": 0.40158962560133865,
350
+ "grad_norm": 0.23798155784606934,
351
+ "learning_rate": 9.188723698172419e-07,
352
+ "loss": 0.5129,
353
+ "step": 480
354
+ },
355
+ {
356
+ "epoch": 0.40995607613469986,
357
+ "grad_norm": 0.2367415726184845,
358
+ "learning_rate": 9.151755058373999e-07,
359
+ "loss": 0.5229,
360
+ "step": 490
361
+ },
362
+ {
363
+ "epoch": 0.4183225266680611,
364
+ "grad_norm": 0.21267226338386536,
365
+ "learning_rate": 9.114040640257457e-07,
366
+ "loss": 0.5239,
367
+ "step": 500
368
+ },
369
+ {
370
+ "epoch": 0.4266889772014223,
371
+ "grad_norm": 0.2113737016916275,
372
+ "learning_rate": 9.07558721845061e-07,
373
+ "loss": 0.5004,
374
+ "step": 510
375
+ },
376
+ {
377
+ "epoch": 0.4350554277347835,
378
+ "grad_norm": 0.19743551313877106,
379
+ "learning_rate": 9.036401700328253e-07,
380
+ "loss": 0.51,
381
+ "step": 520
382
+ },
383
+ {
384
+ "epoch": 0.4434218782681447,
385
+ "grad_norm": 0.17708665132522583,
386
+ "learning_rate": 8.996491124771386e-07,
387
+ "loss": 0.4916,
388
+ "step": 530
389
+ },
390
+ {
391
+ "epoch": 0.45178832880150593,
392
+ "grad_norm": 0.19286687672138214,
393
+ "learning_rate": 8.955862660902827e-07,
394
+ "loss": 0.4972,
395
+ "step": 540
396
+ },
397
+ {
398
+ "epoch": 0.4601547793348672,
399
+ "grad_norm": 0.18794295191764832,
400
+ "learning_rate": 8.914523606799416e-07,
401
+ "loss": 0.4958,
402
+ "step": 550
403
+ },
404
+ {
405
+ "epoch": 0.4685212298682284,
406
+ "grad_norm": 0.17703424394130707,
407
+ "learning_rate": 8.872481388181076e-07,
408
+ "loss": 0.5041,
409
+ "step": 560
410
+ },
411
+ {
412
+ "epoch": 0.47688768040158963,
413
+ "grad_norm": 0.1806008517742157,
414
+ "learning_rate": 8.829743557076924e-07,
415
+ "loss": 0.4786,
416
+ "step": 570
417
+ },
418
+ {
419
+ "epoch": 0.48525413093495084,
420
+ "grad_norm": 0.17116817831993103,
421
+ "learning_rate": 8.786317790468707e-07,
422
+ "loss": 0.4771,
423
+ "step": 580
424
+ },
425
+ {
426
+ "epoch": 0.49362058146831206,
427
+ "grad_norm": 0.16165761649608612,
428
+ "learning_rate": 8.742211888911788e-07,
429
+ "loss": 0.4831,
430
+ "step": 590
431
+ },
432
+ {
433
+ "epoch": 0.5019870320016733,
434
+ "grad_norm": 0.16249604523181915,
435
+ "learning_rate": 8.697433775133934e-07,
436
+ "loss": 0.4703,
437
+ "step": 600
438
+ },
439
+ {
440
+ "epoch": 0.5103534825350345,
441
+ "grad_norm": 0.16176247596740723,
442
+ "learning_rate": 8.651991492612158e-07,
443
+ "loss": 0.4898,
444
+ "step": 610
445
+ },
446
+ {
447
+ "epoch": 0.5187199330683957,
448
+ "grad_norm": 0.17271846532821655,
449
+ "learning_rate": 8.605893204127876e-07,
450
+ "loss": 0.4783,
451
+ "step": 620
452
+ },
453
+ {
454
+ "epoch": 0.527086383601757,
455
+ "grad_norm": 0.1579618752002716,
456
+ "learning_rate": 8.559147190300629e-07,
457
+ "loss": 0.4685,
458
+ "step": 630
459
+ },
460
+ {
461
+ "epoch": 0.5354528341351181,
462
+ "grad_norm": 0.14920172095298767,
463
+ "learning_rate": 8.511761848100628e-07,
464
+ "loss": 0.468,
465
+ "step": 640
466
+ },
467
+ {
468
+ "epoch": 0.5438192846684794,
469
+ "grad_norm": 0.15946857631206512,
470
+ "learning_rate": 8.463745689340426e-07,
471
+ "loss": 0.4669,
472
+ "step": 650
473
+ },
474
+ {
475
+ "epoch": 0.5521857352018407,
476
+ "grad_norm": 0.14744576811790466,
477
+ "learning_rate": 8.415107339145932e-07,
478
+ "loss": 0.4709,
479
+ "step": 660
480
+ },
481
+ {
482
+ "epoch": 0.5605521857352018,
483
+ "grad_norm": 0.13913576304912567,
484
+ "learning_rate": 8.365855534407088e-07,
485
+ "loss": 0.4711,
486
+ "step": 670
487
+ },
488
+ {
489
+ "epoch": 0.5689186362685631,
490
+ "grad_norm": 0.14849849045276642,
491
+ "learning_rate": 8.315999122208458e-07,
492
+ "loss": 0.4739,
493
+ "step": 680
494
+ },
495
+ {
496
+ "epoch": 0.5772850868019243,
497
+ "grad_norm": 0.1468994915485382,
498
+ "learning_rate": 8.265547058240038e-07,
499
+ "loss": 0.47,
500
+ "step": 690
501
+ },
502
+ {
503
+ "epoch": 0.5856515373352855,
504
+ "grad_norm": 0.1443009227514267,
505
+ "learning_rate": 8.214508405188542e-07,
506
+ "loss": 0.4706,
507
+ "step": 700
508
+ },
509
+ {
510
+ "epoch": 0.5940179878686467,
511
+ "grad_norm": 0.13816598057746887,
512
+ "learning_rate": 8.162892331109481e-07,
513
+ "loss": 0.4625,
514
+ "step": 710
515
+ },
516
+ {
517
+ "epoch": 0.602384438402008,
518
+ "grad_norm": 0.13011474907398224,
519
+ "learning_rate": 8.110708107780299e-07,
520
+ "loss": 0.4527,
521
+ "step": 720
522
+ },
523
+ {
524
+ "epoch": 0.6107508889353692,
525
+ "grad_norm": 0.13332195580005646,
526
+ "learning_rate": 8.057965109034896e-07,
527
+ "loss": 0.4527,
528
+ "step": 730
529
+ },
530
+ {
531
+ "epoch": 0.6191173394687304,
532
+ "grad_norm": 0.12983644008636475,
533
+ "learning_rate": 8.004672809079807e-07,
534
+ "loss": 0.4486,
535
+ "step": 740
536
+ },
537
+ {
538
+ "epoch": 0.6274837900020916,
539
+ "grad_norm": 0.13085156679153442,
540
+ "learning_rate": 7.950840780792347e-07,
541
+ "loss": 0.4516,
542
+ "step": 750
543
+ },
544
+ {
545
+ "epoch": 0.6358502405354528,
546
+ "grad_norm": 0.12932556867599487,
547
+ "learning_rate": 7.896478694001047e-07,
548
+ "loss": 0.4515,
549
+ "step": 760
550
+ },
551
+ {
552
+ "epoch": 0.6442166910688141,
553
+ "grad_norm": 0.13092581927776337,
554
+ "learning_rate": 7.841596313748666e-07,
555
+ "loss": 0.4475,
556
+ "step": 770
557
+ },
558
+ {
559
+ "epoch": 0.6525831416021752,
560
+ "grad_norm": 0.1348915696144104,
561
+ "learning_rate": 7.786203498538093e-07,
562
+ "loss": 0.4471,
563
+ "step": 780
564
+ },
565
+ {
566
+ "epoch": 0.6609495921355365,
567
+ "grad_norm": 0.12818686664104462,
568
+ "learning_rate": 7.730310198561469e-07,
569
+ "loss": 0.4443,
570
+ "step": 790
571
+ },
572
+ {
573
+ "epoch": 0.6693160426688978,
574
+ "grad_norm": 0.12480778992176056,
575
+ "learning_rate": 7.673926453912845e-07,
576
+ "loss": 0.4536,
577
+ "step": 800
578
+ },
579
+ {
580
+ "epoch": 0.6776824932022589,
581
+ "grad_norm": 0.12339136004447937,
582
+ "learning_rate": 7.617062392784671e-07,
583
+ "loss": 0.4452,
584
+ "step": 810
585
+ },
586
+ {
587
+ "epoch": 0.6860489437356202,
588
+ "grad_norm": 0.1215905249118805,
589
+ "learning_rate": 7.559728229648488e-07,
590
+ "loss": 0.4456,
591
+ "step": 820
592
+ },
593
+ {
594
+ "epoch": 0.6944153942689814,
595
+ "grad_norm": 0.12470731884241104,
596
+ "learning_rate": 7.501934263420088e-07,
597
+ "loss": 0.4511,
598
+ "step": 830
599
+ },
600
+ {
601
+ "epoch": 0.7027818448023426,
602
+ "grad_norm": 0.12183189392089844,
603
+ "learning_rate": 7.443690875609541e-07,
604
+ "loss": 0.4497,
605
+ "step": 840
606
+ },
607
+ {
608
+ "epoch": 0.7111482953357038,
609
+ "grad_norm": 0.1178760975599289,
610
+ "learning_rate": 7.385008528456356e-07,
611
+ "loss": 0.438,
612
+ "step": 850
613
+ },
614
+ {
615
+ "epoch": 0.719514745869065,
616
+ "grad_norm": 0.11445871740579605,
617
+ "learning_rate": 7.325897763050154e-07,
618
+ "loss": 0.4468,
619
+ "step": 860
620
+ },
621
+ {
622
+ "epoch": 0.7278811964024263,
623
+ "grad_norm": 0.11578909307718277,
624
+ "learning_rate": 7.266369197437181e-07,
625
+ "loss": 0.4388,
626
+ "step": 870
627
+ },
628
+ {
629
+ "epoch": 0.7362476469357875,
630
+ "grad_norm": 0.11665735393762589,
631
+ "learning_rate": 7.206433524712988e-07,
632
+ "loss": 0.4401,
633
+ "step": 880
634
+ },
635
+ {
636
+ "epoch": 0.7446140974691488,
637
+ "grad_norm": 0.11469507217407227,
638
+ "learning_rate": 7.146101511101635e-07,
639
+ "loss": 0.449,
640
+ "step": 890
641
+ },
642
+ {
643
+ "epoch": 0.7529805480025099,
644
+ "grad_norm": 0.11232412606477737,
645
+ "learning_rate": 7.085383994021756e-07,
646
+ "loss": 0.4355,
647
+ "step": 900
648
+ },
649
+ {
650
+ "epoch": 0.7613469985358712,
651
+ "grad_norm": 0.10072898119688034,
652
+ "learning_rate": 7.024291880139842e-07,
653
+ "loss": 0.4418,
654
+ "step": 910
655
+ },
656
+ {
657
+ "epoch": 0.7697134490692323,
658
+ "grad_norm": 0.11294315755367279,
659
+ "learning_rate": 6.962836143411077e-07,
660
+ "loss": 0.4369,
661
+ "step": 920
662
+ },
663
+ {
664
+ "epoch": 0.7780798996025936,
665
+ "grad_norm": 0.10808352380990982,
666
+ "learning_rate": 6.901027823108087e-07,
667
+ "loss": 0.4479,
668
+ "step": 930
669
+ },
670
+ {
671
+ "epoch": 0.7864463501359549,
672
+ "grad_norm": 0.10845589637756348,
673
+ "learning_rate": 6.838878021837968e-07,
674
+ "loss": 0.431,
675
+ "step": 940
676
+ },
677
+ {
678
+ "epoch": 0.794812800669316,
679
+ "grad_norm": 0.1148146539926529,
680
+ "learning_rate": 6.776397903547918e-07,
681
+ "loss": 0.4241,
682
+ "step": 950
683
+ },
684
+ {
685
+ "epoch": 0.8031792512026773,
686
+ "grad_norm": 0.10050572454929352,
687
+ "learning_rate": 6.713598691519873e-07,
688
+ "loss": 0.4326,
689
+ "step": 960
690
+ },
691
+ {
692
+ "epoch": 0.8115457017360385,
693
+ "grad_norm": 0.10299045592546463,
694
+ "learning_rate": 6.650491666354458e-07,
695
+ "loss": 0.4332,
696
+ "step": 970
697
+ },
698
+ {
699
+ "epoch": 0.8199121522693997,
700
+ "grad_norm": 0.1057298332452774,
701
+ "learning_rate": 6.587088163944676e-07,
702
+ "loss": 0.4271,
703
+ "step": 980
704
+ },
705
+ {
706
+ "epoch": 0.8282786028027609,
707
+ "grad_norm": 0.1022634506225586,
708
+ "learning_rate": 6.523399573439621e-07,
709
+ "loss": 0.4296,
710
+ "step": 990
711
+ },
712
+ {
713
+ "epoch": 0.8366450533361222,
714
+ "grad_norm": 0.10467051714658737,
715
+ "learning_rate": 6.459437335198675e-07,
716
+ "loss": 0.4348,
717
+ "step": 1000
718
+ },
719
+ {
720
+ "epoch": 0.8366450533361222,
721
+ "eval_loss": 0.3649196922779083,
722
+ "eval_runtime": 2.5049,
723
+ "eval_samples_per_second": 79.045,
724
+ "eval_steps_per_second": 2.795,
725
+ "step": 1000
726
+ },
727
+ {
728
+ "epoch": 0.8450115038694834,
729
+ "grad_norm": 0.10084307193756104,
730
+ "learning_rate": 6.395212938736459e-07,
731
+ "loss": 0.4326,
732
+ "step": 1010
733
+ },
734
+ {
735
+ "epoch": 0.8533779544028446,
736
+ "grad_norm": 0.09792770445346832,
737
+ "learning_rate": 6.330737920658989e-07,
738
+ "loss": 0.4227,
739
+ "step": 1020
740
+ },
741
+ {
742
+ "epoch": 0.8617444049362059,
743
+ "grad_norm": 0.10365421324968338,
744
+ "learning_rate": 6.266023862591349e-07,
745
+ "loss": 0.4272,
746
+ "step": 1030
747
+ },
748
+ {
749
+ "epoch": 0.870110855469567,
750
+ "grad_norm": 0.10585031658411026,
751
+ "learning_rate": 6.201082389097301e-07,
752
+ "loss": 0.4277,
753
+ "step": 1040
754
+ },
755
+ {
756
+ "epoch": 0.8784773060029283,
757
+ "grad_norm": 0.10123966634273529,
758
+ "learning_rate": 6.135925165591159e-07,
759
+ "loss": 0.423,
760
+ "step": 1050
761
+ },
762
+ {
763
+ "epoch": 0.8868437565362894,
764
+ "grad_norm": 0.10199949890375137,
765
+ "learning_rate": 6.070563896242328e-07,
766
+ "loss": 0.4323,
767
+ "step": 1060
768
+ },
769
+ {
770
+ "epoch": 0.8952102070696507,
771
+ "grad_norm": 0.09743819385766983,
772
+ "learning_rate": 6.00501032187291e-07,
773
+ "loss": 0.4271,
774
+ "step": 1070
775
+ },
776
+ {
777
+ "epoch": 0.9035766576030119,
778
+ "grad_norm": 0.10138524323701859,
779
+ "learning_rate": 5.939276217848683e-07,
780
+ "loss": 0.4263,
781
+ "step": 1080
782
+ },
783
+ {
784
+ "epoch": 0.9119431081363731,
785
+ "grad_norm": 0.10496041923761368,
786
+ "learning_rate": 5.873373391963905e-07,
787
+ "loss": 0.4162,
788
+ "step": 1090
789
+ },
790
+ {
791
+ "epoch": 0.9203095586697344,
792
+ "grad_norm": 0.09408185631036758,
793
+ "learning_rate": 5.807313682320283e-07,
794
+ "loss": 0.4272,
795
+ "step": 1100
796
+ },
797
+ {
798
+ "epoch": 0.9286760092030956,
799
+ "grad_norm": 0.10358374565839767,
800
+ "learning_rate": 5.741108955200502e-07,
801
+ "loss": 0.4253,
802
+ "step": 1110
803
+ },
804
+ {
805
+ "epoch": 0.9370424597364568,
806
+ "grad_norm": 0.09181825816631317,
807
+ "learning_rate": 5.674771102936684e-07,
808
+ "loss": 0.4337,
809
+ "step": 1120
810
+ },
811
+ {
812
+ "epoch": 0.945408910269818,
813
+ "grad_norm": 0.09469820559024811,
814
+ "learning_rate": 5.608312041774174e-07,
815
+ "loss": 0.4279,
816
+ "step": 1130
817
+ },
818
+ {
819
+ "epoch": 0.9537753608031793,
820
+ "grad_norm": 0.09452180564403534,
821
+ "learning_rate": 5.541743709731028e-07,
822
+ "loss": 0.4205,
823
+ "step": 1140
824
+ },
825
+ {
826
+ "epoch": 0.9621418113365404,
827
+ "grad_norm": 0.10470899194478989,
828
+ "learning_rate": 5.475078064453596e-07,
829
+ "loss": 0.4277,
830
+ "step": 1150
831
+ },
832
+ {
833
+ "epoch": 0.9705082618699017,
834
+ "grad_norm": 0.09374157339334488,
835
+ "learning_rate": 5.408327081068568e-07,
836
+ "loss": 0.4185,
837
+ "step": 1160
838
+ },
839
+ {
840
+ "epoch": 0.978874712403263,
841
+ "grad_norm": 0.10022414475679398,
842
+ "learning_rate": 5.34150275003189e-07,
843
+ "loss": 0.4168,
844
+ "step": 1170
845
+ },
846
+ {
847
+ "epoch": 0.9872411629366241,
848
+ "grad_norm": 0.08453768491744995,
849
+ "learning_rate": 5.274617074974917e-07,
850
+ "loss": 0.4167,
851
+ "step": 1180
852
+ },
853
+ {
854
+ "epoch": 0.9956076134699854,
855
+ "grad_norm": 0.09851043671369553,
856
+ "learning_rate": 5.207682070548215e-07,
857
+ "loss": 0.4221,
858
+ "step": 1190
859
+ },
860
+ {
861
+ "epoch": 1.0033465802133446,
862
+ "grad_norm": 0.10138513892889023,
863
+ "learning_rate": 5.140709760263363e-07,
864
+ "loss": 0.415,
865
+ "step": 1200
866
+ },
867
+ {
868
+ "epoch": 1.0117130307467057,
869
+ "grad_norm": 0.09353633224964142,
870
+ "learning_rate": 5.073712174333181e-07,
871
+ "loss": 0.4192,
872
+ "step": 1210
873
+ },
874
+ {
875
+ "epoch": 1.0200794812800669,
876
+ "grad_norm": 0.0947510227560997,
877
+ "learning_rate": 5.006701347510745e-07,
878
+ "loss": 0.4264,
879
+ "step": 1220
880
+ },
881
+ {
882
+ "epoch": 1.0284459318134282,
883
+ "grad_norm": 0.09594232589006424,
884
+ "learning_rate": 4.939689316927584e-07,
885
+ "loss": 0.4111,
886
+ "step": 1230
887
+ },
888
+ {
889
+ "epoch": 1.0368123823467894,
890
+ "grad_norm": 0.09330349415540695,
891
+ "learning_rate": 4.872688119931462e-07,
892
+ "loss": 0.4211,
893
+ "step": 1240
894
+ },
895
+ {
896
+ "epoch": 1.0451788328801506,
897
+ "grad_norm": 0.09639482200145721,
898
+ "learning_rate": 4.805709791924105e-07,
899
+ "loss": 0.4153,
900
+ "step": 1250
901
+ },
902
+ {
903
+ "epoch": 1.0535452834135117,
904
+ "grad_norm": 0.0985507071018219,
905
+ "learning_rate": 4.7387663641992974e-07,
906
+ "loss": 0.4133,
907
+ "step": 1260
908
+ },
909
+ {
910
+ "epoch": 1.061911733946873,
911
+ "grad_norm": 0.08969943225383759,
912
+ "learning_rate": 4.6718698617816916e-07,
913
+ "loss": 0.4199,
914
+ "step": 1270
915
+ },
916
+ {
917
+ "epoch": 1.0702781844802343,
918
+ "grad_norm": 0.09307977557182312,
919
+ "learning_rate": 4.6050323012667673e-07,
920
+ "loss": 0.4188,
921
+ "step": 1280
922
+ },
923
+ {
924
+ "epoch": 1.0786446350135954,
925
+ "grad_norm": 0.09855697304010391,
926
+ "learning_rate": 4.5382656886622896e-07,
927
+ "loss": 0.4199,
928
+ "step": 1290
929
+ },
930
+ {
931
+ "epoch": 1.0870110855469568,
932
+ "grad_norm": 0.09455449134111404,
933
+ "learning_rate": 4.471582017231672e-07,
934
+ "loss": 0.4101,
935
+ "step": 1300
936
+ },
937
+ {
938
+ "epoch": 1.095377536080318,
939
+ "grad_norm": 0.09281763434410095,
940
+ "learning_rate": 4.4049932653396396e-07,
941
+ "loss": 0.4156,
942
+ "step": 1310
943
+ },
944
+ {
945
+ "epoch": 1.1037439866136791,
946
+ "grad_norm": 0.09138793498277664,
947
+ "learning_rate": 4.338511394300548e-07,
948
+ "loss": 0.416,
949
+ "step": 1320
950
+ },
951
+ {
952
+ "epoch": 1.1121104371470403,
953
+ "grad_norm": 0.09749738872051239,
954
+ "learning_rate": 4.272148346229788e-07,
955
+ "loss": 0.4255,
956
+ "step": 1330
957
+ },
958
+ {
959
+ "epoch": 1.1204768876804017,
960
+ "grad_norm": 0.08886326104402542,
961
+ "learning_rate": 4.205916041898619e-07,
962
+ "loss": 0.3988,
963
+ "step": 1340
964
+ },
965
+ {
966
+ "epoch": 1.1288433382137628,
967
+ "grad_norm": 0.09860669076442719,
968
+ "learning_rate": 4.139826378592844e-07,
969
+ "loss": 0.4116,
970
+ "step": 1350
971
+ },
972
+ {
973
+ "epoch": 1.137209788747124,
974
+ "grad_norm": 0.09252766519784927,
975
+ "learning_rate": 4.0738912279757145e-07,
976
+ "loss": 0.4147,
977
+ "step": 1360
978
+ },
979
+ {
980
+ "epoch": 1.1455762392804854,
981
+ "grad_norm": 0.08980321139097214,
982
+ "learning_rate": 4.008122433955419e-07,
983
+ "loss": 0.4114,
984
+ "step": 1370
985
+ },
986
+ {
987
+ "epoch": 1.1539426898138465,
988
+ "grad_norm": 0.09422625601291656,
989
+ "learning_rate": 3.9425318105575687e-07,
990
+ "loss": 0.4198,
991
+ "step": 1380
992
+ },
993
+ {
994
+ "epoch": 1.1623091403472077,
995
+ "grad_norm": 0.08835183084011078,
996
+ "learning_rate": 3.877131139803045e-07,
997
+ "loss": 0.4204,
998
+ "step": 1390
999
+ },
1000
+ {
1001
+ "epoch": 1.1706755908805688,
1002
+ "grad_norm": 0.09362850338220596,
1003
+ "learning_rate": 3.811932169591608e-07,
1004
+ "loss": 0.4111,
1005
+ "step": 1400
1006
+ },
1007
+ {
1008
+ "epoch": 1.1790420414139302,
1009
+ "grad_norm": 0.09230010956525803,
1010
+ "learning_rate": 3.7469466115916125e-07,
1011
+ "loss": 0.4192,
1012
+ "step": 1410
1013
+ },
1014
+ {
1015
+ "epoch": 1.1874084919472914,
1016
+ "grad_norm": 0.08950341492891312,
1017
+ "learning_rate": 3.682186139136257e-07,
1018
+ "loss": 0.4103,
1019
+ "step": 1420
1020
+ },
1021
+ {
1022
+ "epoch": 1.1957749424806525,
1023
+ "grad_norm": 0.10122744739055634,
1024
+ "learning_rate": 3.617662385126701e-07,
1025
+ "loss": 0.4131,
1026
+ "step": 1430
1027
+ },
1028
+ {
1029
+ "epoch": 1.2041413930140137,
1030
+ "grad_norm": 0.09493398666381836,
1031
+ "learning_rate": 3.5533869399424476e-07,
1032
+ "loss": 0.42,
1033
+ "step": 1440
1034
+ },
1035
+ {
1036
+ "epoch": 1.212507843547375,
1037
+ "grad_norm": 0.0915885791182518,
1038
+ "learning_rate": 3.4893713493593793e-07,
1039
+ "loss": 0.4044,
1040
+ "step": 1450
1041
+ },
1042
+ {
1043
+ "epoch": 1.2208742940807362,
1044
+ "grad_norm": 0.10119744390249252,
1045
+ "learning_rate": 3.4256271124757784e-07,
1046
+ "loss": 0.409,
1047
+ "step": 1460
1048
+ },
1049
+ {
1050
+ "epoch": 1.2292407446140974,
1051
+ "grad_norm": 0.09402301162481308,
1052
+ "learning_rate": 3.362165679646766e-07,
1053
+ "loss": 0.4188,
1054
+ "step": 1470
1055
+ },
1056
+ {
1057
+ "epoch": 1.2376071951474588,
1058
+ "grad_norm": 0.08955765515565872,
1059
+ "learning_rate": 3.29899845042746e-07,
1060
+ "loss": 0.4203,
1061
+ "step": 1480
1062
+ },
1063
+ {
1064
+ "epoch": 1.24597364568082,
1065
+ "grad_norm": 0.08440003544092178,
1066
+ "learning_rate": 3.236136771525292e-07,
1067
+ "loss": 0.4002,
1068
+ "step": 1490
1069
+ },
1070
+ {
1071
+ "epoch": 1.254340096214181,
1072
+ "grad_norm": 0.09873976558446884,
1073
+ "learning_rate": 3.1735919347617933e-07,
1074
+ "loss": 0.4146,
1075
+ "step": 1500
1076
+ },
1077
+ {
1078
+ "epoch": 1.2627065467475425,
1079
+ "grad_norm": 0.08994744718074799,
1080
+ "learning_rate": 3.111375175044254e-07,
1081
+ "loss": 0.4128,
1082
+ "step": 1510
1083
+ },
1084
+ {
1085
+ "epoch": 1.2710729972809036,
1086
+ "grad_norm": 0.09055975079536438,
1087
+ "learning_rate": 3.049497668347598e-07,
1088
+ "loss": 0.4111,
1089
+ "step": 1520
1090
+ },
1091
+ {
1092
+ "epoch": 1.2794394478142648,
1093
+ "grad_norm": 0.09734614938497543,
1094
+ "learning_rate": 2.9879705297068457e-07,
1095
+ "loss": 0.4018,
1096
+ "step": 1530
1097
+ },
1098
+ {
1099
+ "epoch": 1.2878058983476262,
1100
+ "grad_norm": 0.08868604898452759,
1101
+ "learning_rate": 2.9268048112205273e-07,
1102
+ "loss": 0.4022,
1103
+ "step": 1540
1104
+ },
1105
+ {
1106
+ "epoch": 1.2961723488809873,
1107
+ "grad_norm": 0.09322873502969742,
1108
+ "learning_rate": 2.866011500065394e-07,
1109
+ "loss": 0.4066,
1110
+ "step": 1550
1111
+ },
1112
+ {
1113
+ "epoch": 1.3045387994143485,
1114
+ "grad_norm": 0.09405604004859924,
1115
+ "learning_rate": 2.805601516522802e-07,
1116
+ "loss": 0.4026,
1117
+ "step": 1560
1118
+ },
1119
+ {
1120
+ "epoch": 1.3129052499477096,
1121
+ "grad_norm": 0.09636805951595306,
1122
+ "learning_rate": 2.7455857120170946e-07,
1123
+ "loss": 0.4175,
1124
+ "step": 1570
1125
+ },
1126
+ {
1127
+ "epoch": 1.3212717004810708,
1128
+ "grad_norm": 0.09669621288776398,
1129
+ "learning_rate": 2.685974867166376e-07,
1130
+ "loss": 0.4059,
1131
+ "step": 1580
1132
+ },
1133
+ {
1134
+ "epoch": 1.3296381510144322,
1135
+ "grad_norm": 0.08673422038555145,
1136
+ "learning_rate": 2.6267796898459905e-07,
1137
+ "loss": 0.4099,
1138
+ "step": 1590
1139
+ },
1140
+ {
1141
+ "epoch": 1.3380046015477933,
1142
+ "grad_norm": 0.08576668798923492,
1143
+ "learning_rate": 2.568010813265066e-07,
1144
+ "loss": 0.4029,
1145
+ "step": 1600
1146
+ },
1147
+ {
1148
+ "epoch": 1.3463710520811545,
1149
+ "grad_norm": 0.09726343303918839,
1150
+ "learning_rate": 2.509678794056473e-07,
1151
+ "loss": 0.4078,
1152
+ "step": 1610
1153
+ },
1154
+ {
1155
+ "epoch": 1.3547375026145159,
1156
+ "grad_norm": 0.09295405447483063,
1157
+ "learning_rate": 2.4517941103805535e-07,
1158
+ "loss": 0.4068,
1159
+ "step": 1620
1160
+ },
1161
+ {
1162
+ "epoch": 1.363103953147877,
1163
+ "grad_norm": 0.09591178596019745,
1164
+ "learning_rate": 2.3943671600429105e-07,
1165
+ "loss": 0.4163,
1166
+ "step": 1630
1167
+ },
1168
+ {
1169
+ "epoch": 1.3714704036812382,
1170
+ "grad_norm": 0.08477117121219635,
1171
+ "learning_rate": 2.3374082586266785e-07,
1172
+ "loss": 0.416,
1173
+ "step": 1640
1174
+ },
1175
+ {
1176
+ "epoch": 1.3798368542145996,
1177
+ "grad_norm": 0.1042475700378418,
1178
+ "learning_rate": 2.2809276376395187e-07,
1179
+ "loss": 0.4112,
1180
+ "step": 1650
1181
+ },
1182
+ {
1183
+ "epoch": 1.3882033047479607,
1184
+ "grad_norm": 0.09027962386608124,
1185
+ "learning_rate": 2.2249354426757495e-07,
1186
+ "loss": 0.3947,
1187
+ "step": 1660
1188
+ },
1189
+ {
1190
+ "epoch": 1.3965697552813219,
1191
+ "grad_norm": 0.08813817799091339,
1192
+ "learning_rate": 2.169441731593893e-07,
1193
+ "loss": 0.3996,
1194
+ "step": 1670
1195
+ },
1196
+ {
1197
+ "epoch": 1.4049362058146833,
1198
+ "grad_norm": 0.09734838455915451,
1199
+ "learning_rate": 2.1144564727099788e-07,
1200
+ "loss": 0.4048,
1201
+ "step": 1680
1202
+ },
1203
+ {
1204
+ "epoch": 1.4133026563480444,
1205
+ "grad_norm": 0.09102274477481842,
1206
+ "learning_rate": 2.0599895430069525e-07,
1207
+ "loss": 0.3962,
1208
+ "step": 1690
1209
+ },
1210
+ {
1211
+ "epoch": 1.4216691068814056,
1212
+ "grad_norm": 0.08596834540367126,
1213
+ "learning_rate": 2.0060507263604671e-07,
1214
+ "loss": 0.4021,
1215
+ "step": 1700
1216
+ },
1217
+ {
1218
+ "epoch": 1.4300355574147667,
1219
+ "grad_norm": 0.093462735414505,
1220
+ "learning_rate": 1.9526497117814045e-07,
1221
+ "loss": 0.4011,
1222
+ "step": 1710
1223
+ },
1224
+ {
1225
+ "epoch": 1.438402007948128,
1226
+ "grad_norm": 0.08627881109714508,
1227
+ "learning_rate": 1.8997960916754563e-07,
1228
+ "loss": 0.4048,
1229
+ "step": 1720
1230
+ },
1231
+ {
1232
+ "epoch": 1.4467684584814893,
1233
+ "grad_norm": 0.08983384817838669,
1234
+ "learning_rate": 1.8474993601200316e-07,
1235
+ "loss": 0.409,
1236
+ "step": 1730
1237
+ },
1238
+ {
1239
+ "epoch": 1.4551349090148504,
1240
+ "grad_norm": 0.09563502669334412,
1241
+ "learning_rate": 1.7957689111588448e-07,
1242
+ "loss": 0.4116,
1243
+ "step": 1740
1244
+ },
1245
+ {
1246
+ "epoch": 1.4635013595482116,
1247
+ "grad_norm": 0.08999122679233551,
1248
+ "learning_rate": 1.7446140371144597e-07,
1249
+ "loss": 0.4063,
1250
+ "step": 1750
1251
+ },
1252
+ {
1253
+ "epoch": 1.471867810081573,
1254
+ "grad_norm": 0.08469448238611221,
1255
+ "learning_rate": 1.6940439269191238e-07,
1256
+ "loss": 0.4054,
1257
+ "step": 1760
1258
+ },
1259
+ {
1260
+ "epoch": 1.4802342606149341,
1261
+ "grad_norm": 0.08758175373077393,
1262
+ "learning_rate": 1.644067664464152e-07,
1263
+ "loss": 0.4081,
1264
+ "step": 1770
1265
+ },
1266
+ {
1267
+ "epoch": 1.4886007111482953,
1268
+ "grad_norm": 0.09469179809093475,
1269
+ "learning_rate": 1.594694226968203e-07,
1270
+ "loss": 0.4079,
1271
+ "step": 1780
1272
+ },
1273
+ {
1274
+ "epoch": 1.4969671616816567,
1275
+ "grad_norm": 0.08764393627643585,
1276
+ "learning_rate": 1.5459324833646898e-07,
1277
+ "loss": 0.3982,
1278
+ "step": 1790
1279
+ },
1280
+ {
1281
+ "epoch": 1.5053336122150178,
1282
+ "grad_norm": 0.08189354091882706,
1283
+ "learning_rate": 1.4977911927086688e-07,
1284
+ "loss": 0.3991,
1285
+ "step": 1800
1286
+ },
1287
+ {
1288
+ "epoch": 1.513700062748379,
1289
+ "grad_norm": 0.08130005747079849,
1290
+ "learning_rate": 1.450279002603451e-07,
1291
+ "loss": 0.3984,
1292
+ "step": 1810
1293
+ },
1294
+ {
1295
+ "epoch": 1.5220665132817404,
1296
+ "grad_norm": 0.09187156707048416,
1297
+ "learning_rate": 1.4034044476472278e-07,
1298
+ "loss": 0.3983,
1299
+ "step": 1820
1300
+ },
1301
+ {
1302
+ "epoch": 1.5304329638151013,
1303
+ "grad_norm": 0.09889491647481918,
1304
+ "learning_rate": 1.3571759479000156e-07,
1305
+ "loss": 0.4132,
1306
+ "step": 1830
1307
+ },
1308
+ {
1309
+ "epoch": 1.5387994143484627,
1310
+ "grad_norm": 0.08915228396654129,
1311
+ "learning_rate": 1.3116018073711571e-07,
1312
+ "loss": 0.3998,
1313
+ "step": 1840
1314
+ },
1315
+ {
1316
+ "epoch": 1.5471658648818238,
1317
+ "grad_norm": 0.0829881951212883,
1318
+ "learning_rate": 1.2666902125276734e-07,
1319
+ "loss": 0.4054,
1320
+ "step": 1850
1321
+ },
1322
+ {
1323
+ "epoch": 1.555532315415185,
1324
+ "grad_norm": 0.08255288749933243,
1325
+ "learning_rate": 1.2224492308237383e-07,
1326
+ "loss": 0.4018,
1327
+ "step": 1860
1328
+ },
1329
+ {
1330
+ "epoch": 1.5638987659485464,
1331
+ "grad_norm": 0.09240256994962692,
1332
+ "learning_rate": 1.1788868092515175e-07,
1333
+ "loss": 0.4025,
1334
+ "step": 1870
1335
+ },
1336
+ {
1337
+ "epoch": 1.5722652164819075,
1338
+ "grad_norm": 0.07964866608381271,
1339
+ "learning_rate": 1.1360107729136586e-07,
1340
+ "loss": 0.4014,
1341
+ "step": 1880
1342
+ },
1343
+ {
1344
+ "epoch": 1.5806316670152687,
1345
+ "grad_norm": 0.08687260001897812,
1346
+ "learning_rate": 1.0938288236176646e-07,
1347
+ "loss": 0.4148,
1348
+ "step": 1890
1349
+ },
1350
+ {
1351
+ "epoch": 1.58899811754863,
1352
+ "grad_norm": 0.1075129359960556,
1353
+ "learning_rate": 1.0523485384924291e-07,
1354
+ "loss": 0.416,
1355
+ "step": 1900
1356
+ },
1357
+ {
1358
+ "epoch": 1.5973645680819912,
1359
+ "grad_norm": 0.08477263152599335,
1360
+ "learning_rate": 1.0115773686271484e-07,
1361
+ "loss": 0.3995,
1362
+ "step": 1910
1363
+ },
1364
+ {
1365
+ "epoch": 1.6057310186153524,
1366
+ "grad_norm": 0.09497664868831635,
1367
+ "learning_rate": 9.715226377328988e-08,
1368
+ "loss": 0.4084,
1369
+ "step": 1920
1370
+ },
1371
+ {
1372
+ "epoch": 1.6140974691487138,
1373
+ "grad_norm": 0.08730168640613556,
1374
+ "learning_rate": 9.321915408270653e-08,
1375
+ "loss": 0.4071,
1376
+ "step": 1930
1377
+ },
1378
+ {
1379
+ "epoch": 1.622463919682075,
1380
+ "grad_norm": 0.08849070221185684,
1381
+ "learning_rate": 8.935911429409166e-08,
1382
+ "loss": 0.4079,
1383
+ "step": 1940
1384
+ },
1385
+ {
1386
+ "epoch": 1.630830370215436,
1387
+ "grad_norm": 0.0913858711719513,
1388
+ "learning_rate": 8.557283778505098e-08,
1389
+ "loss": 0.4163,
1390
+ "step": 1950
1391
+ },
1392
+ {
1393
+ "epoch": 1.6391968207487975,
1394
+ "grad_norm": 0.08180033415555954,
1395
+ "learning_rate": 8.186100468311763e-08,
1396
+ "loss": 0.4013,
1397
+ "step": 1960
1398
+ },
1399
+ {
1400
+ "epoch": 1.6475632712821584,
1401
+ "grad_norm": 0.09243475645780563,
1402
+ "learning_rate": 7.822428174358165e-08,
1403
+ "loss": 0.4119,
1404
+ "step": 1970
1405
+ },
1406
+ {
1407
+ "epoch": 1.6559297218155198,
1408
+ "grad_norm": 0.09416525810956955,
1409
+ "learning_rate": 7.466332222972083e-08,
1410
+ "loss": 0.4002,
1411
+ "step": 1980
1412
+ },
1413
+ {
1414
+ "epoch": 1.664296172348881,
1415
+ "grad_norm": 0.0888829380273819,
1416
+ "learning_rate": 7.117876579545478e-08,
1417
+ "loss": 0.391,
1418
+ "step": 1990
1419
+ },
1420
+ {
1421
+ "epoch": 1.672662622882242,
1422
+ "grad_norm": 0.0871991366147995,
1423
+ "learning_rate": 6.777123837044468e-08,
1424
+ "loss": 0.4049,
1425
+ "step": 2000
1426
+ },
1427
+ {
1428
+ "epoch": 1.672662622882242,
1429
+ "eval_loss": 0.3422872722148895,
1430
+ "eval_runtime": 2.5054,
1431
+ "eval_samples_per_second": 79.028,
1432
+ "eval_steps_per_second": 2.794,
1433
+ "step": 2000
1434
+ }
1435
+ ],
1436
+ "logging_steps": 10,
1437
+ "max_steps": 2392,
1438
+ "num_input_tokens_seen": 0,
1439
+ "num_train_epochs": 2,
1440
+ "save_steps": 1000,
1441
+ "stateful_callbacks": {
1442
+ "TrainerControl": {
1443
+ "args": {
1444
+ "should_epoch_stop": false,
1445
+ "should_evaluate": false,
1446
+ "should_log": false,
1447
+ "should_save": true,
1448
+ "should_training_stop": false
1449
+ },
1450
+ "attributes": {}
1451
+ }
1452
+ },
1453
+ "total_flos": 2.074237676024968e+19,
1454
+ "train_batch_size": 8,
1455
+ "trial_name": null,
1456
+ "trial_params": null
1457
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fec14acee5ae660b18bff7cbae869950b6c5260e6a5f1f23db0eac1c537c6c4f
3
+ size 6417
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2000/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/README.md ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen2.5-Coder-7B-Instruct
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen2.5-Coder-7B-Instruct
7
+ - lora
8
+ - transformers
9
+ ---
10
+
11
+ # Model Card for Model ID
12
+
13
+ <!-- Provide a quick summary of what the model is/does. -->
14
+
15
+
16
+
17
+ ## Model Details
18
+
19
+ ### Model Description
20
+
21
+ <!-- Provide a longer summary of what this model is. -->
22
+
23
+
24
+
25
+ - **Developed by:** [More Information Needed]
26
+ - **Funded by [optional]:** [More Information Needed]
27
+ - **Shared by [optional]:** [More Information Needed]
28
+ - **Model type:** [More Information Needed]
29
+ - **Language(s) (NLP):** [More Information Needed]
30
+ - **License:** [More Information Needed]
31
+ - **Finetuned from model [optional]:** [More Information Needed]
32
+
33
+ ### Model Sources [optional]
34
+
35
+ <!-- Provide the basic links for the model. -->
36
+
37
+ - **Repository:** [More Information Needed]
38
+ - **Paper [optional]:** [More Information Needed]
39
+ - **Demo [optional]:** [More Information Needed]
40
+
41
+ ## Uses
42
+
43
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
44
+
45
+ ### Direct Use
46
+
47
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
48
+
49
+ [More Information Needed]
50
+
51
+ ### Downstream Use [optional]
52
+
53
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
54
+
55
+ [More Information Needed]
56
+
57
+ ### Out-of-Scope Use
58
+
59
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
60
+
61
+ [More Information Needed]
62
+
63
+ ## Bias, Risks, and Limitations
64
+
65
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
66
+
67
+ [More Information Needed]
68
+
69
+ ### Recommendations
70
+
71
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
72
+
73
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
74
+
75
+ ## How to Get Started with the Model
76
+
77
+ Use the code below to get started with the model.
78
+
79
+ [More Information Needed]
80
+
81
+ ## Training Details
82
+
83
+ ### Training Data
84
+
85
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
86
+
87
+ [More Information Needed]
88
+
89
+ ### Training Procedure
90
+
91
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
92
+
93
+ #### Preprocessing [optional]
94
+
95
+ [More Information Needed]
96
+
97
+
98
+ #### Training Hyperparameters
99
+
100
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
101
+
102
+ #### Speeds, Sizes, Times [optional]
103
+
104
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
105
+
106
+ [More Information Needed]
107
+
108
+ ## Evaluation
109
+
110
+ <!-- This section describes the evaluation protocols and provides the results. -->
111
+
112
+ ### Testing Data, Factors & Metrics
113
+
114
+ #### Testing Data
115
+
116
+ <!-- This should link to a Dataset Card if possible. -->
117
+
118
+ [More Information Needed]
119
+
120
+ #### Factors
121
+
122
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
123
+
124
+ [More Information Needed]
125
+
126
+ #### Metrics
127
+
128
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
129
+
130
+ [More Information Needed]
131
+
132
+ ### Results
133
+
134
+ [More Information Needed]
135
+
136
+ #### Summary
137
+
138
+
139
+
140
+ ## Model Examination [optional]
141
+
142
+ <!-- Relevant interpretability work for the model goes here -->
143
+
144
+ [More Information Needed]
145
+
146
+ ## Environmental Impact
147
+
148
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
149
+
150
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
151
+
152
+ - **Hardware Type:** [More Information Needed]
153
+ - **Hours used:** [More Information Needed]
154
+ - **Cloud Provider:** [More Information Needed]
155
+ - **Compute Region:** [More Information Needed]
156
+ - **Carbon Emitted:** [More Information Needed]
157
+
158
+ ## Technical Specifications [optional]
159
+
160
+ ### Model Architecture and Objective
161
+
162
+ [More Information Needed]
163
+
164
+ ### Compute Infrastructure
165
+
166
+ [More Information Needed]
167
+
168
+ #### Hardware
169
+
170
+ [More Information Needed]
171
+
172
+ #### Software
173
+
174
+ [More Information Needed]
175
+
176
+ ## Citation [optional]
177
+
178
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
179
+
180
+ **BibTeX:**
181
+
182
+ [More Information Needed]
183
+
184
+ **APA:**
185
+
186
+ [More Information Needed]
187
+
188
+ ## Glossary [optional]
189
+
190
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
191
+
192
+ [More Information Needed]
193
+
194
+ ## More Information [optional]
195
+
196
+ [More Information Needed]
197
+
198
+ ## Model Card Authors [optional]
199
+
200
+ [More Information Needed]
201
+
202
+ ## Model Card Contact
203
+
204
+ [More Information Needed]
205
+ ### Framework versions
206
+
207
+ - PEFT 0.18.0
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-Coder-7B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.0",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "down_proj",
34
+ "o_proj",
35
+ "k_proj",
36
+ "gate_proj",
37
+ "q_proj",
38
+ "v_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:612b030ce67864735881485d454ff41babafd018d7b4f86b95ea3b1b64db944f
3
+ size 161533192
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:96da046db41d63f15bb2fd908804aa2bad020ff00c2b0c0e7dcafc6402de31fd
3
+ size 323296891
qwen2.5-coder-7B-v4-if-llmmuts-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-v4-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts_iim_system/checkpoint-2392/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ed58848f8759bd5b65fb2745d6958d9301569ca2d56ccbcdbc4ca3091a8550f4
3
+ size 15429