NathanHerr commited on
Commit
52f0cdc
·
verified ·
1 Parent(s): 21bd3e4

Add qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +7 -0
  2. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/README.md +207 -0
  3. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/adapter_config.json +46 -0
  4. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/adapter_model.safetensors +3 -0
  5. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/added_tokens.json +24 -0
  6. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/chat_template.jinja +54 -0
  7. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/README.md +207 -0
  8. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/adapter_config.json +46 -0
  9. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/adapter_model.safetensors +3 -0
  10. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/added_tokens.json +24 -0
  11. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/chat_template.jinja +54 -0
  12. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/merges.txt +0 -0
  13. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/optimizer.pt +3 -0
  14. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/rng_state_0.pth +3 -0
  15. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/rng_state_1.pth +3 -0
  16. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/rng_state_2.pth +3 -0
  17. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/rng_state_3.pth +3 -0
  18. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/scheduler.pt +3 -0
  19. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/special_tokens_map.json +31 -0
  20. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/tokenizer.json +3 -0
  21. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/tokenizer_config.json +207 -0
  22. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/trainer_state.json +749 -0
  23. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/training_args.bin +3 -0
  24. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/vocab.json +0 -0
  25. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/README.md +207 -0
  26. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/adapter_config.json +46 -0
  27. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/adapter_model.safetensors +3 -0
  28. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/added_tokens.json +24 -0
  29. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/chat_template.jinja +54 -0
  30. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/merges.txt +0 -0
  31. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/optimizer.pt +3 -0
  32. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/rng_state_0.pth +3 -0
  33. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/rng_state_1.pth +3 -0
  34. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/rng_state_2.pth +3 -0
  35. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/rng_state_3.pth +3 -0
  36. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/scheduler.pt +3 -0
  37. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/special_tokens_map.json +31 -0
  38. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/tokenizer.json +3 -0
  39. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/tokenizer_config.json +207 -0
  40. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/trainer_state.json +1457 -0
  41. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/training_args.bin +3 -0
  42. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/vocab.json +0 -0
  43. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/README.md +207 -0
  44. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/adapter_config.json +46 -0
  45. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/adapter_model.safetensors +3 -0
  46. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/added_tokens.json +24 -0
  47. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/chat_template.jinja +54 -0
  48. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/merges.txt +0 -0
  49. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/optimizer.pt +3 -0
  50. qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/rng_state_0.pth +3 -0
.gitattributes CHANGED
@@ -34,3 +34,10 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_all_iim_system_checkpoint-5842/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_all_iim_system_checkpoint-5842/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
38
+ qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
39
+ qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
40
+ qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-4000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
41
+ qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-5000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
42
+ qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-5842/tokenizer.json filter=lfs diff=lfs merge=lfs -text
43
+ qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/tokenizer.json filter=lfs diff=lfs merge=lfs -text
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/README.md ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen2.5-Coder-7B-Instruct
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen2.5-Coder-7B-Instruct
7
+ - lora
8
+ - transformers
9
+ ---
10
+
11
+ # Model Card for Model ID
12
+
13
+ <!-- Provide a quick summary of what the model is/does. -->
14
+
15
+
16
+
17
+ ## Model Details
18
+
19
+ ### Model Description
20
+
21
+ <!-- Provide a longer summary of what this model is. -->
22
+
23
+
24
+
25
+ - **Developed by:** [More Information Needed]
26
+ - **Funded by [optional]:** [More Information Needed]
27
+ - **Shared by [optional]:** [More Information Needed]
28
+ - **Model type:** [More Information Needed]
29
+ - **Language(s) (NLP):** [More Information Needed]
30
+ - **License:** [More Information Needed]
31
+ - **Finetuned from model [optional]:** [More Information Needed]
32
+
33
+ ### Model Sources [optional]
34
+
35
+ <!-- Provide the basic links for the model. -->
36
+
37
+ - **Repository:** [More Information Needed]
38
+ - **Paper [optional]:** [More Information Needed]
39
+ - **Demo [optional]:** [More Information Needed]
40
+
41
+ ## Uses
42
+
43
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
44
+
45
+ ### Direct Use
46
+
47
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
48
+
49
+ [More Information Needed]
50
+
51
+ ### Downstream Use [optional]
52
+
53
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
54
+
55
+ [More Information Needed]
56
+
57
+ ### Out-of-Scope Use
58
+
59
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
60
+
61
+ [More Information Needed]
62
+
63
+ ## Bias, Risks, and Limitations
64
+
65
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
66
+
67
+ [More Information Needed]
68
+
69
+ ### Recommendations
70
+
71
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
72
+
73
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
74
+
75
+ ## How to Get Started with the Model
76
+
77
+ Use the code below to get started with the model.
78
+
79
+ [More Information Needed]
80
+
81
+ ## Training Details
82
+
83
+ ### Training Data
84
+
85
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
86
+
87
+ [More Information Needed]
88
+
89
+ ### Training Procedure
90
+
91
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
92
+
93
+ #### Preprocessing [optional]
94
+
95
+ [More Information Needed]
96
+
97
+
98
+ #### Training Hyperparameters
99
+
100
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
101
+
102
+ #### Speeds, Sizes, Times [optional]
103
+
104
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
105
+
106
+ [More Information Needed]
107
+
108
+ ## Evaluation
109
+
110
+ <!-- This section describes the evaluation protocols and provides the results. -->
111
+
112
+ ### Testing Data, Factors & Metrics
113
+
114
+ #### Testing Data
115
+
116
+ <!-- This should link to a Dataset Card if possible. -->
117
+
118
+ [More Information Needed]
119
+
120
+ #### Factors
121
+
122
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
123
+
124
+ [More Information Needed]
125
+
126
+ #### Metrics
127
+
128
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
129
+
130
+ [More Information Needed]
131
+
132
+ ### Results
133
+
134
+ [More Information Needed]
135
+
136
+ #### Summary
137
+
138
+
139
+
140
+ ## Model Examination [optional]
141
+
142
+ <!-- Relevant interpretability work for the model goes here -->
143
+
144
+ [More Information Needed]
145
+
146
+ ## Environmental Impact
147
+
148
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
149
+
150
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
151
+
152
+ - **Hardware Type:** [More Information Needed]
153
+ - **Hours used:** [More Information Needed]
154
+ - **Cloud Provider:** [More Information Needed]
155
+ - **Compute Region:** [More Information Needed]
156
+ - **Carbon Emitted:** [More Information Needed]
157
+
158
+ ## Technical Specifications [optional]
159
+
160
+ ### Model Architecture and Objective
161
+
162
+ [More Information Needed]
163
+
164
+ ### Compute Infrastructure
165
+
166
+ [More Information Needed]
167
+
168
+ #### Hardware
169
+
170
+ [More Information Needed]
171
+
172
+ #### Software
173
+
174
+ [More Information Needed]
175
+
176
+ ## Citation [optional]
177
+
178
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
179
+
180
+ **BibTeX:**
181
+
182
+ [More Information Needed]
183
+
184
+ **APA:**
185
+
186
+ [More Information Needed]
187
+
188
+ ## Glossary [optional]
189
+
190
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
191
+
192
+ [More Information Needed]
193
+
194
+ ## More Information [optional]
195
+
196
+ [More Information Needed]
197
+
198
+ ## Model Card Authors [optional]
199
+
200
+ [More Information Needed]
201
+
202
+ ## Model Card Contact
203
+
204
+ [More Information Needed]
205
+ ### Framework versions
206
+
207
+ - PEFT 0.18.0
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-Coder-7B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.0",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "up_proj",
33
+ "down_proj",
34
+ "v_proj",
35
+ "k_proj",
36
+ "gate_proj",
37
+ "q_proj",
38
+ "o_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:caeb2c1bc6e87a1522642d0374b01c19c1898ba0e7a7e42d39c305410b9ddffa
3
+ size 161533192
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/README.md ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen2.5-Coder-7B-Instruct
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen2.5-Coder-7B-Instruct
7
+ - lora
8
+ - transformers
9
+ ---
10
+
11
+ # Model Card for Model ID
12
+
13
+ <!-- Provide a quick summary of what the model is/does. -->
14
+
15
+
16
+
17
+ ## Model Details
18
+
19
+ ### Model Description
20
+
21
+ <!-- Provide a longer summary of what this model is. -->
22
+
23
+
24
+
25
+ - **Developed by:** [More Information Needed]
26
+ - **Funded by [optional]:** [More Information Needed]
27
+ - **Shared by [optional]:** [More Information Needed]
28
+ - **Model type:** [More Information Needed]
29
+ - **Language(s) (NLP):** [More Information Needed]
30
+ - **License:** [More Information Needed]
31
+ - **Finetuned from model [optional]:** [More Information Needed]
32
+
33
+ ### Model Sources [optional]
34
+
35
+ <!-- Provide the basic links for the model. -->
36
+
37
+ - **Repository:** [More Information Needed]
38
+ - **Paper [optional]:** [More Information Needed]
39
+ - **Demo [optional]:** [More Information Needed]
40
+
41
+ ## Uses
42
+
43
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
44
+
45
+ ### Direct Use
46
+
47
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
48
+
49
+ [More Information Needed]
50
+
51
+ ### Downstream Use [optional]
52
+
53
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
54
+
55
+ [More Information Needed]
56
+
57
+ ### Out-of-Scope Use
58
+
59
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
60
+
61
+ [More Information Needed]
62
+
63
+ ## Bias, Risks, and Limitations
64
+
65
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
66
+
67
+ [More Information Needed]
68
+
69
+ ### Recommendations
70
+
71
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
72
+
73
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
74
+
75
+ ## How to Get Started with the Model
76
+
77
+ Use the code below to get started with the model.
78
+
79
+ [More Information Needed]
80
+
81
+ ## Training Details
82
+
83
+ ### Training Data
84
+
85
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
86
+
87
+ [More Information Needed]
88
+
89
+ ### Training Procedure
90
+
91
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
92
+
93
+ #### Preprocessing [optional]
94
+
95
+ [More Information Needed]
96
+
97
+
98
+ #### Training Hyperparameters
99
+
100
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
101
+
102
+ #### Speeds, Sizes, Times [optional]
103
+
104
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
105
+
106
+ [More Information Needed]
107
+
108
+ ## Evaluation
109
+
110
+ <!-- This section describes the evaluation protocols and provides the results. -->
111
+
112
+ ### Testing Data, Factors & Metrics
113
+
114
+ #### Testing Data
115
+
116
+ <!-- This should link to a Dataset Card if possible. -->
117
+
118
+ [More Information Needed]
119
+
120
+ #### Factors
121
+
122
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
123
+
124
+ [More Information Needed]
125
+
126
+ #### Metrics
127
+
128
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
129
+
130
+ [More Information Needed]
131
+
132
+ ### Results
133
+
134
+ [More Information Needed]
135
+
136
+ #### Summary
137
+
138
+
139
+
140
+ ## Model Examination [optional]
141
+
142
+ <!-- Relevant interpretability work for the model goes here -->
143
+
144
+ [More Information Needed]
145
+
146
+ ## Environmental Impact
147
+
148
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
149
+
150
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
151
+
152
+ - **Hardware Type:** [More Information Needed]
153
+ - **Hours used:** [More Information Needed]
154
+ - **Cloud Provider:** [More Information Needed]
155
+ - **Compute Region:** [More Information Needed]
156
+ - **Carbon Emitted:** [More Information Needed]
157
+
158
+ ## Technical Specifications [optional]
159
+
160
+ ### Model Architecture and Objective
161
+
162
+ [More Information Needed]
163
+
164
+ ### Compute Infrastructure
165
+
166
+ [More Information Needed]
167
+
168
+ #### Hardware
169
+
170
+ [More Information Needed]
171
+
172
+ #### Software
173
+
174
+ [More Information Needed]
175
+
176
+ ## Citation [optional]
177
+
178
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
179
+
180
+ **BibTeX:**
181
+
182
+ [More Information Needed]
183
+
184
+ **APA:**
185
+
186
+ [More Information Needed]
187
+
188
+ ## Glossary [optional]
189
+
190
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
191
+
192
+ [More Information Needed]
193
+
194
+ ## More Information [optional]
195
+
196
+ [More Information Needed]
197
+
198
+ ## Model Card Authors [optional]
199
+
200
+ [More Information Needed]
201
+
202
+ ## Model Card Contact
203
+
204
+ [More Information Needed]
205
+ ### Framework versions
206
+
207
+ - PEFT 0.18.0
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-Coder-7B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.0",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "down_proj",
33
+ "q_proj",
34
+ "gate_proj",
35
+ "up_proj",
36
+ "o_proj",
37
+ "k_proj",
38
+ "v_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b8cbc4a642579e06be0df1872f407f879f558c59dfbf35d14fcfe47c763999db
3
+ size 161533192
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c0271cf94fdd2b7c495ed886a5fb49d935804f5cc9c999b15a8f0adbecf0c004
3
+ size 323296891
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1164d5fd78d02b07e92601ee09bb9accbaed0dfaad6b9f67c40ada4b8575b947
3
+ size 15429
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/rng_state_1.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7b4f93ac912a8c16de89abdcb6988622f0d891c77e51d367301551a060c94329
3
+ size 15429
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/rng_state_2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bd420ad69ad81aacd3b85622be54ab65c0e6fd03977e0717ff07bf148d66fdd4
3
+ size 15429
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/rng_state_3.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff8e9e0464616bab0890dd931938c9d01accf36751426f3ad1a2feab2a0387c3
3
+ size 15429
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8dc14bea2e5df39750de9d60a5efc115f9c4c966d9399a4184cb0c7814275579
3
+ size 1465
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|im_end|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
+ size 11421896
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/tokenizer_config.json ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ }
181
+ },
182
+ "additional_special_tokens": [
183
+ "<|im_start|>",
184
+ "<|im_end|>",
185
+ "<|object_ref_start|>",
186
+ "<|object_ref_end|>",
187
+ "<|box_start|>",
188
+ "<|box_end|>",
189
+ "<|quad_start|>",
190
+ "<|quad_end|>",
191
+ "<|vision_start|>",
192
+ "<|vision_end|>",
193
+ "<|vision_pad|>",
194
+ "<|image_pad|>",
195
+ "<|video_pad|>"
196
+ ],
197
+ "bos_token": null,
198
+ "clean_up_tokenization_spaces": false,
199
+ "eos_token": "<|im_end|>",
200
+ "errors": "replace",
201
+ "extra_special_tokens": {},
202
+ "model_max_length": 32768,
203
+ "pad_token": "<|endoftext|>",
204
+ "split_special_tokens": false,
205
+ "tokenizer_class": "Qwen2Tokenizer",
206
+ "unk_token": null
207
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/trainer_state.json ADDED
@@ -0,0 +1,749 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.3423485107839781,
6
+ "eval_steps": 1000,
7
+ "global_step": 1000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.00034234851078397807,
14
+ "grad_norm": 0.38582974672317505,
15
+ "learning_rate": 0.0,
16
+ "loss": 1.2938,
17
+ "step": 1
18
+ },
19
+ {
20
+ "epoch": 0.003423485107839781,
21
+ "grad_norm": 0.3895561397075653,
22
+ "learning_rate": 7.692307692307692e-08,
23
+ "loss": 1.301,
24
+ "step": 10
25
+ },
26
+ {
27
+ "epoch": 0.006846970215679562,
28
+ "grad_norm": 0.38466933369636536,
29
+ "learning_rate": 1.6239316239316238e-07,
30
+ "loss": 1.3026,
31
+ "step": 20
32
+ },
33
+ {
34
+ "epoch": 0.010270455323519343,
35
+ "grad_norm": 0.3903484046459198,
36
+ "learning_rate": 2.478632478632479e-07,
37
+ "loss": 1.3113,
38
+ "step": 30
39
+ },
40
+ {
41
+ "epoch": 0.013693940431359124,
42
+ "grad_norm": 0.3902166783809662,
43
+ "learning_rate": 3.333333333333333e-07,
44
+ "loss": 1.3227,
45
+ "step": 40
46
+ },
47
+ {
48
+ "epoch": 0.017117425539198903,
49
+ "grad_norm": 0.39113983511924744,
50
+ "learning_rate": 4.1880341880341877e-07,
51
+ "loss": 1.3091,
52
+ "step": 50
53
+ },
54
+ {
55
+ "epoch": 0.020540910647038686,
56
+ "grad_norm": 0.4007033407688141,
57
+ "learning_rate": 5.042735042735042e-07,
58
+ "loss": 1.3183,
59
+ "step": 60
60
+ },
61
+ {
62
+ "epoch": 0.023964395754878465,
63
+ "grad_norm": 0.417681485414505,
64
+ "learning_rate": 5.897435897435898e-07,
65
+ "loss": 1.3217,
66
+ "step": 70
67
+ },
68
+ {
69
+ "epoch": 0.027387880862718247,
70
+ "grad_norm": 0.46474331617355347,
71
+ "learning_rate": 6.752136752136752e-07,
72
+ "loss": 1.296,
73
+ "step": 80
74
+ },
75
+ {
76
+ "epoch": 0.030811365970558027,
77
+ "grad_norm": 0.418237566947937,
78
+ "learning_rate": 7.606837606837606e-07,
79
+ "loss": 1.3132,
80
+ "step": 90
81
+ },
82
+ {
83
+ "epoch": 0.034234851078397806,
84
+ "grad_norm": 0.4302998185157776,
85
+ "learning_rate": 8.461538461538461e-07,
86
+ "loss": 1.3084,
87
+ "step": 100
88
+ },
89
+ {
90
+ "epoch": 0.03765833618623759,
91
+ "grad_norm": 0.41094598174095154,
92
+ "learning_rate": 9.316239316239316e-07,
93
+ "loss": 1.2953,
94
+ "step": 110
95
+ },
96
+ {
97
+ "epoch": 0.04108182129407737,
98
+ "grad_norm": 0.4528387188911438,
99
+ "learning_rate": 9.999996988736778e-07,
100
+ "loss": 1.2961,
101
+ "step": 120
102
+ },
103
+ {
104
+ "epoch": 0.044505306401917154,
105
+ "grad_norm": 0.4192649722099304,
106
+ "learning_rate": 9.99989159490489e-07,
107
+ "loss": 1.2894,
108
+ "step": 130
109
+ },
110
+ {
111
+ "epoch": 0.04792879150975693,
112
+ "grad_norm": 0.4704149663448334,
113
+ "learning_rate": 9.999635641539022e-07,
114
+ "loss": 1.2844,
115
+ "step": 140
116
+ },
117
+ {
118
+ "epoch": 0.05135227661759671,
119
+ "grad_norm": 0.45329269766807556,
120
+ "learning_rate": 9.99922913634658e-07,
121
+ "loss": 1.293,
122
+ "step": 150
123
+ },
124
+ {
125
+ "epoch": 0.054775761725436495,
126
+ "grad_norm": 0.45733365416526794,
127
+ "learning_rate": 9.998672091568482e-07,
128
+ "loss": 1.2657,
129
+ "step": 160
130
+ },
131
+ {
132
+ "epoch": 0.05819924683327628,
133
+ "grad_norm": 0.4719047248363495,
134
+ "learning_rate": 9.997964523978768e-07,
135
+ "loss": 1.2579,
136
+ "step": 170
137
+ },
138
+ {
139
+ "epoch": 0.06162273194111605,
140
+ "grad_norm": 0.4891057014465332,
141
+ "learning_rate": 9.99710645488411e-07,
142
+ "loss": 1.2544,
143
+ "step": 180
144
+ },
145
+ {
146
+ "epoch": 0.06504621704895584,
147
+ "grad_norm": 0.4674888551235199,
148
+ "learning_rate": 9.996097910123165e-07,
149
+ "loss": 1.2361,
150
+ "step": 190
151
+ },
152
+ {
153
+ "epoch": 0.06846970215679561,
154
+ "grad_norm": 0.4695661962032318,
155
+ "learning_rate": 9.9949389200658e-07,
156
+ "loss": 1.2113,
157
+ "step": 200
158
+ },
159
+ {
160
+ "epoch": 0.0718931872646354,
161
+ "grad_norm": 0.47163236141204834,
162
+ "learning_rate": 9.993629519612165e-07,
163
+ "loss": 1.214,
164
+ "step": 210
165
+ },
166
+ {
167
+ "epoch": 0.07531667237247518,
168
+ "grad_norm": 0.4599274694919586,
169
+ "learning_rate": 9.992169748191667e-07,
170
+ "loss": 1.1921,
171
+ "step": 220
172
+ },
173
+ {
174
+ "epoch": 0.07874015748031496,
175
+ "grad_norm": 0.4423569142818451,
176
+ "learning_rate": 9.990559649761758e-07,
177
+ "loss": 1.1744,
178
+ "step": 230
179
+ },
180
+ {
181
+ "epoch": 0.08216364258815474,
182
+ "grad_norm": 0.4392240047454834,
183
+ "learning_rate": 9.988799272806623e-07,
184
+ "loss": 1.1789,
185
+ "step": 240
186
+ },
187
+ {
188
+ "epoch": 0.08558712769599452,
189
+ "grad_norm": 0.42201587557792664,
190
+ "learning_rate": 9.986888670335715e-07,
191
+ "loss": 1.1489,
192
+ "step": 250
193
+ },
194
+ {
195
+ "epoch": 0.08901061280383431,
196
+ "grad_norm": 0.4007083773612976,
197
+ "learning_rate": 9.984827899882168e-07,
198
+ "loss": 1.128,
199
+ "step": 260
200
+ },
201
+ {
202
+ "epoch": 0.09243409791167409,
203
+ "grad_norm": 0.3840133249759674,
204
+ "learning_rate": 9.982617023501056e-07,
205
+ "loss": 1.1276,
206
+ "step": 270
207
+ },
208
+ {
209
+ "epoch": 0.09585758301951386,
210
+ "grad_norm": 0.36397966742515564,
211
+ "learning_rate": 9.980256107767522e-07,
212
+ "loss": 1.0973,
213
+ "step": 280
214
+ },
215
+ {
216
+ "epoch": 0.09928106812735364,
217
+ "grad_norm": 0.3815755248069763,
218
+ "learning_rate": 9.977745223774784e-07,
219
+ "loss": 1.0934,
220
+ "step": 290
221
+ },
222
+ {
223
+ "epoch": 0.10270455323519342,
224
+ "grad_norm": 0.33746176958084106,
225
+ "learning_rate": 9.975084447131989e-07,
226
+ "loss": 1.0887,
227
+ "step": 300
228
+ },
229
+ {
230
+ "epoch": 0.1061280383430332,
231
+ "grad_norm": 0.3245246112346649,
232
+ "learning_rate": 9.972273857961928e-07,
233
+ "loss": 1.0705,
234
+ "step": 310
235
+ },
236
+ {
237
+ "epoch": 0.10955152345087299,
238
+ "grad_norm": 0.3304518759250641,
239
+ "learning_rate": 9.969313540898638e-07,
240
+ "loss": 1.0617,
241
+ "step": 320
242
+ },
243
+ {
244
+ "epoch": 0.11297500855871277,
245
+ "grad_norm": 0.3013385832309723,
246
+ "learning_rate": 9.966203585084841e-07,
247
+ "loss": 1.0416,
248
+ "step": 330
249
+ },
250
+ {
251
+ "epoch": 0.11639849366655255,
252
+ "grad_norm": 0.30748993158340454,
253
+ "learning_rate": 9.962944084169267e-07,
254
+ "loss": 1.0319,
255
+ "step": 340
256
+ },
257
+ {
258
+ "epoch": 0.11982197877439234,
259
+ "grad_norm": 0.2813571095466614,
260
+ "learning_rate": 9.959535136303834e-07,
261
+ "loss": 1.0233,
262
+ "step": 350
263
+ },
264
+ {
265
+ "epoch": 0.1232454638822321,
266
+ "grad_norm": 0.2859013080596924,
267
+ "learning_rate": 9.955976844140688e-07,
268
+ "loss": 1.0239,
269
+ "step": 360
270
+ },
271
+ {
272
+ "epoch": 0.1266689489900719,
273
+ "grad_norm": 0.27636730670928955,
274
+ "learning_rate": 9.95226931482911e-07,
275
+ "loss": 1.012,
276
+ "step": 370
277
+ },
278
+ {
279
+ "epoch": 0.13009243409791169,
280
+ "grad_norm": 0.27530768513679504,
281
+ "learning_rate": 9.948412660012305e-07,
282
+ "loss": 1.0094,
283
+ "step": 380
284
+ },
285
+ {
286
+ "epoch": 0.13351591920575145,
287
+ "grad_norm": 0.2697387635707855,
288
+ "learning_rate": 9.944406995824013e-07,
289
+ "loss": 1.0136,
290
+ "step": 390
291
+ },
292
+ {
293
+ "epoch": 0.13693940431359122,
294
+ "grad_norm": 0.2743023931980133,
295
+ "learning_rate": 9.94025244288504e-07,
296
+ "loss": 0.9975,
297
+ "step": 400
298
+ },
299
+ {
300
+ "epoch": 0.14036288942143102,
301
+ "grad_norm": 0.27155691385269165,
302
+ "learning_rate": 9.93594912629961e-07,
303
+ "loss": 0.9849,
304
+ "step": 410
305
+ },
306
+ {
307
+ "epoch": 0.1437863745292708,
308
+ "grad_norm": 0.2568703889846802,
309
+ "learning_rate": 9.9314971756516e-07,
310
+ "loss": 0.9918,
311
+ "step": 420
312
+ },
313
+ {
314
+ "epoch": 0.14720985963711058,
315
+ "grad_norm": 0.25010108947753906,
316
+ "learning_rate": 9.926896725000637e-07,
317
+ "loss": 0.9723,
318
+ "step": 430
319
+ },
320
+ {
321
+ "epoch": 0.15063334474495035,
322
+ "grad_norm": 0.25133389234542847,
323
+ "learning_rate": 9.92214791287807e-07,
324
+ "loss": 0.9636,
325
+ "step": 440
326
+ },
327
+ {
328
+ "epoch": 0.15405682985279015,
329
+ "grad_norm": 0.2505231499671936,
330
+ "learning_rate": 9.917250882282785e-07,
331
+ "loss": 0.9578,
332
+ "step": 450
333
+ },
334
+ {
335
+ "epoch": 0.15748031496062992,
336
+ "grad_norm": 0.2534894049167633,
337
+ "learning_rate": 9.912205780676905e-07,
338
+ "loss": 0.9524,
339
+ "step": 460
340
+ },
341
+ {
342
+ "epoch": 0.16090380006846972,
343
+ "grad_norm": 0.23746544122695923,
344
+ "learning_rate": 9.90701275998136e-07,
345
+ "loss": 0.9538,
346
+ "step": 470
347
+ },
348
+ {
349
+ "epoch": 0.16432728517630948,
350
+ "grad_norm": 0.22148397564888,
351
+ "learning_rate": 9.901671976571288e-07,
352
+ "loss": 0.9469,
353
+ "step": 480
354
+ },
355
+ {
356
+ "epoch": 0.16775077028414925,
357
+ "grad_norm": 0.23445384204387665,
358
+ "learning_rate": 9.896183591271354e-07,
359
+ "loss": 0.9428,
360
+ "step": 490
361
+ },
362
+ {
363
+ "epoch": 0.17117425539198905,
364
+ "grad_norm": 0.22655028104782104,
365
+ "learning_rate": 9.890547769350886e-07,
366
+ "loss": 0.9343,
367
+ "step": 500
368
+ },
369
+ {
370
+ "epoch": 0.17459774049982882,
371
+ "grad_norm": 0.2170156091451645,
372
+ "learning_rate": 9.884764680518906e-07,
373
+ "loss": 0.9345,
374
+ "step": 510
375
+ },
376
+ {
377
+ "epoch": 0.17802122560766862,
378
+ "grad_norm": 0.22584597766399384,
379
+ "learning_rate": 9.878834498919025e-07,
380
+ "loss": 0.9309,
381
+ "step": 520
382
+ },
383
+ {
384
+ "epoch": 0.18144471071550838,
385
+ "grad_norm": 0.21805278956890106,
386
+ "learning_rate": 9.872757403124185e-07,
387
+ "loss": 0.9229,
388
+ "step": 530
389
+ },
390
+ {
391
+ "epoch": 0.18486819582334818,
392
+ "grad_norm": 0.21373434364795685,
393
+ "learning_rate": 9.866533576131302e-07,
394
+ "loss": 0.9063,
395
+ "step": 540
396
+ },
397
+ {
398
+ "epoch": 0.18829168093118795,
399
+ "grad_norm": 0.21042130887508392,
400
+ "learning_rate": 9.860163205355733e-07,
401
+ "loss": 0.9146,
402
+ "step": 550
403
+ },
404
+ {
405
+ "epoch": 0.19171516603902772,
406
+ "grad_norm": 0.21380534768104553,
407
+ "learning_rate": 9.853646482625652e-07,
408
+ "loss": 0.9103,
409
+ "step": 560
410
+ },
411
+ {
412
+ "epoch": 0.19513865114686751,
413
+ "grad_norm": 0.20159614086151123,
414
+ "learning_rate": 9.846983604176258e-07,
415
+ "loss": 0.897,
416
+ "step": 570
417
+ },
418
+ {
419
+ "epoch": 0.19856213625470728,
420
+ "grad_norm": 0.2064715176820755,
421
+ "learning_rate": 9.840174770643876e-07,
422
+ "loss": 0.8973,
423
+ "step": 580
424
+ },
425
+ {
426
+ "epoch": 0.20198562136254708,
427
+ "grad_norm": 0.20663177967071533,
428
+ "learning_rate": 9.833220187059912e-07,
429
+ "loss": 0.892,
430
+ "step": 590
431
+ },
432
+ {
433
+ "epoch": 0.20540910647038685,
434
+ "grad_norm": 0.20841750502586365,
435
+ "learning_rate": 9.82612006284468e-07,
436
+ "loss": 0.8834,
437
+ "step": 600
438
+ },
439
+ {
440
+ "epoch": 0.20883259157822665,
441
+ "grad_norm": 0.19864782691001892,
442
+ "learning_rate": 9.818874611801095e-07,
443
+ "loss": 0.8793,
444
+ "step": 610
445
+ },
446
+ {
447
+ "epoch": 0.2122560766860664,
448
+ "grad_norm": 0.18866512179374695,
449
+ "learning_rate": 9.811484052108233e-07,
450
+ "loss": 0.8851,
451
+ "step": 620
452
+ },
453
+ {
454
+ "epoch": 0.21567956179390618,
455
+ "grad_norm": 0.19272519648075104,
456
+ "learning_rate": 9.803948606314761e-07,
457
+ "loss": 0.8783,
458
+ "step": 630
459
+ },
460
+ {
461
+ "epoch": 0.21910304690174598,
462
+ "grad_norm": 0.19224460422992706,
463
+ "learning_rate": 9.796268501332246e-07,
464
+ "loss": 0.8768,
465
+ "step": 640
466
+ },
467
+ {
468
+ "epoch": 0.22252653200958575,
469
+ "grad_norm": 0.18141649663448334,
470
+ "learning_rate": 9.788443968428301e-07,
471
+ "loss": 0.8667,
472
+ "step": 650
473
+ },
474
+ {
475
+ "epoch": 0.22595001711742554,
476
+ "grad_norm": 0.19055131077766418,
477
+ "learning_rate": 9.780475243219647e-07,
478
+ "loss": 0.8702,
479
+ "step": 660
480
+ },
481
+ {
482
+ "epoch": 0.2293735022252653,
483
+ "grad_norm": 0.1838882565498352,
484
+ "learning_rate": 9.77236256566499e-07,
485
+ "loss": 0.8618,
486
+ "step": 670
487
+ },
488
+ {
489
+ "epoch": 0.2327969873331051,
490
+ "grad_norm": 0.18793776631355286,
491
+ "learning_rate": 9.764106180057821e-07,
492
+ "loss": 0.8495,
493
+ "step": 680
494
+ },
495
+ {
496
+ "epoch": 0.23622047244094488,
497
+ "grad_norm": 0.18299244344234467,
498
+ "learning_rate": 9.755706335019044e-07,
499
+ "loss": 0.8459,
500
+ "step": 690
501
+ },
502
+ {
503
+ "epoch": 0.23964395754878468,
504
+ "grad_norm": 0.1785850077867508,
505
+ "learning_rate": 9.747163283489493e-07,
506
+ "loss": 0.8453,
507
+ "step": 700
508
+ },
509
+ {
510
+ "epoch": 0.24306744265662444,
511
+ "grad_norm": 0.16908957064151764,
512
+ "learning_rate": 9.738477282722317e-07,
513
+ "loss": 0.8342,
514
+ "step": 710
515
+ },
516
+ {
517
+ "epoch": 0.2464909277644642,
518
+ "grad_norm": 0.17376649379730225,
519
+ "learning_rate": 9.729648594275233e-07,
520
+ "loss": 0.8369,
521
+ "step": 720
522
+ },
523
+ {
524
+ "epoch": 0.249914412872304,
525
+ "grad_norm": 0.18265017867088318,
526
+ "learning_rate": 9.720677484002649e-07,
527
+ "loss": 0.8385,
528
+ "step": 730
529
+ },
530
+ {
531
+ "epoch": 0.2533378979801438,
532
+ "grad_norm": 0.17172974348068237,
533
+ "learning_rate": 9.711564222047658e-07,
534
+ "loss": 0.8454,
535
+ "step": 740
536
+ },
537
+ {
538
+ "epoch": 0.2567613830879836,
539
+ "grad_norm": 0.1695045828819275,
540
+ "learning_rate": 9.702309082833905e-07,
541
+ "loss": 0.8279,
542
+ "step": 750
543
+ },
544
+ {
545
+ "epoch": 0.26018486819582337,
546
+ "grad_norm": 0.16996362805366516,
547
+ "learning_rate": 9.692912345057318e-07,
548
+ "loss": 0.8362,
549
+ "step": 760
550
+ },
551
+ {
552
+ "epoch": 0.2636083533036631,
553
+ "grad_norm": 0.16263867914676666,
554
+ "learning_rate": 9.68337429167773e-07,
555
+ "loss": 0.8272,
556
+ "step": 770
557
+ },
558
+ {
559
+ "epoch": 0.2670318384115029,
560
+ "grad_norm": 0.17135751247406006,
561
+ "learning_rate": 9.673695209910336e-07,
562
+ "loss": 0.8379,
563
+ "step": 780
564
+ },
565
+ {
566
+ "epoch": 0.2704553235193427,
567
+ "grad_norm": 0.15958866477012634,
568
+ "learning_rate": 9.66387539121707e-07,
569
+ "loss": 0.8195,
570
+ "step": 790
571
+ },
572
+ {
573
+ "epoch": 0.27387880862718245,
574
+ "grad_norm": 0.1589757204055786,
575
+ "learning_rate": 9.653915131297802e-07,
576
+ "loss": 0.8102,
577
+ "step": 800
578
+ },
579
+ {
580
+ "epoch": 0.27730229373502224,
581
+ "grad_norm": 0.16261506080627441,
582
+ "learning_rate": 9.64381473008146e-07,
583
+ "loss": 0.8155,
584
+ "step": 810
585
+ },
586
+ {
587
+ "epoch": 0.28072577884286204,
588
+ "grad_norm": 0.15770071744918823,
589
+ "learning_rate": 9.63357449171697e-07,
590
+ "loss": 0.8176,
591
+ "step": 820
592
+ },
593
+ {
594
+ "epoch": 0.28414926395070184,
595
+ "grad_norm": 0.14913444221019745,
596
+ "learning_rate": 9.623194724564127e-07,
597
+ "loss": 0.8148,
598
+ "step": 830
599
+ },
600
+ {
601
+ "epoch": 0.2875727490585416,
602
+ "grad_norm": 0.15190348029136658,
603
+ "learning_rate": 9.612675741184289e-07,
604
+ "loss": 0.8011,
605
+ "step": 840
606
+ },
607
+ {
608
+ "epoch": 0.2909962341663814,
609
+ "grad_norm": 0.14866051077842712,
610
+ "learning_rate": 9.602017858330964e-07,
611
+ "loss": 0.8132,
612
+ "step": 850
613
+ },
614
+ {
615
+ "epoch": 0.29441971927422117,
616
+ "grad_norm": 0.1489679515361786,
617
+ "learning_rate": 9.591221396940294e-07,
618
+ "loss": 0.8076,
619
+ "step": 860
620
+ },
621
+ {
622
+ "epoch": 0.2978432043820609,
623
+ "grad_norm": 0.1507052481174469,
624
+ "learning_rate": 9.580286682121362e-07,
625
+ "loss": 0.8035,
626
+ "step": 870
627
+ },
628
+ {
629
+ "epoch": 0.3012666894899007,
630
+ "grad_norm": 0.14822350442409515,
631
+ "learning_rate": 9.56921404314642e-07,
632
+ "loss": 0.799,
633
+ "step": 880
634
+ },
635
+ {
636
+ "epoch": 0.3046901745977405,
637
+ "grad_norm": 0.1420404314994812,
638
+ "learning_rate": 9.558003813440972e-07,
639
+ "loss": 0.7885,
640
+ "step": 890
641
+ },
642
+ {
643
+ "epoch": 0.3081136597055803,
644
+ "grad_norm": 0.1396367996931076,
645
+ "learning_rate": 9.546656330573726e-07,
646
+ "loss": 0.7869,
647
+ "step": 900
648
+ },
649
+ {
650
+ "epoch": 0.31153714481342004,
651
+ "grad_norm": 0.14706209301948547,
652
+ "learning_rate": 9.535171936246441e-07,
653
+ "loss": 0.79,
654
+ "step": 910
655
+ },
656
+ {
657
+ "epoch": 0.31496062992125984,
658
+ "grad_norm": 0.1379164308309555,
659
+ "learning_rate": 9.523550976283623e-07,
660
+ "loss": 0.7834,
661
+ "step": 920
662
+ },
663
+ {
664
+ "epoch": 0.31838411502909963,
665
+ "grad_norm": 0.137995183467865,
666
+ "learning_rate": 9.511793800622123e-07,
667
+ "loss": 0.7837,
668
+ "step": 930
669
+ },
670
+ {
671
+ "epoch": 0.32180760013693943,
672
+ "grad_norm": 0.1464492380619049,
673
+ "learning_rate": 9.499900763300596e-07,
674
+ "loss": 0.7861,
675
+ "step": 940
676
+ },
677
+ {
678
+ "epoch": 0.3252310852447792,
679
+ "grad_norm": 0.14009970426559448,
680
+ "learning_rate": 9.487872222448838e-07,
681
+ "loss": 0.7825,
682
+ "step": 950
683
+ },
684
+ {
685
+ "epoch": 0.32865457035261897,
686
+ "grad_norm": 0.1352413296699524,
687
+ "learning_rate": 9.475708540277001e-07,
688
+ "loss": 0.7773,
689
+ "step": 960
690
+ },
691
+ {
692
+ "epoch": 0.33207805546045877,
693
+ "grad_norm": 0.1339689940214157,
694
+ "learning_rate": 9.463410083064692e-07,
695
+ "loss": 0.7737,
696
+ "step": 970
697
+ },
698
+ {
699
+ "epoch": 0.3355015405682985,
700
+ "grad_norm": 0.14391706883907318,
701
+ "learning_rate": 9.450977221149935e-07,
702
+ "loss": 0.7741,
703
+ "step": 980
704
+ },
705
+ {
706
+ "epoch": 0.3389250256761383,
707
+ "grad_norm": 0.1347951740026474,
708
+ "learning_rate": 9.438410328918029e-07,
709
+ "loss": 0.7712,
710
+ "step": 990
711
+ },
712
+ {
713
+ "epoch": 0.3423485107839781,
714
+ "grad_norm": 0.14170120656490326,
715
+ "learning_rate": 9.425709784790266e-07,
716
+ "loss": 0.7726,
717
+ "step": 1000
718
+ },
719
+ {
720
+ "epoch": 0.3423485107839781,
721
+ "eval_loss": 0.503926157951355,
722
+ "eval_runtime": 2.912,
723
+ "eval_samples_per_second": 67.65,
724
+ "eval_steps_per_second": 2.404,
725
+ "step": 1000
726
+ }
727
+ ],
728
+ "logging_steps": 10,
729
+ "max_steps": 5842,
730
+ "num_input_tokens_seen": 0,
731
+ "num_train_epochs": 2,
732
+ "save_steps": 1000,
733
+ "stateful_callbacks": {
734
+ "TrainerControl": {
735
+ "args": {
736
+ "should_epoch_stop": false,
737
+ "should_evaluate": false,
738
+ "should_log": false,
739
+ "should_save": true,
740
+ "should_training_stop": false
741
+ },
742
+ "attributes": {}
743
+ }
744
+ },
745
+ "total_flos": 1.2950252184083104e+19,
746
+ "train_batch_size": 8,
747
+ "trial_name": null,
748
+ "trial_params": null
749
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bad7413c72dd69805137a642a7a4990c7ffcbcf04a2cb4d40021ec4a331fe520
3
+ size 6609
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-1000/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/README.md ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen2.5-Coder-7B-Instruct
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen2.5-Coder-7B-Instruct
7
+ - lora
8
+ - transformers
9
+ ---
10
+
11
+ # Model Card for Model ID
12
+
13
+ <!-- Provide a quick summary of what the model is/does. -->
14
+
15
+
16
+
17
+ ## Model Details
18
+
19
+ ### Model Description
20
+
21
+ <!-- Provide a longer summary of what this model is. -->
22
+
23
+
24
+
25
+ - **Developed by:** [More Information Needed]
26
+ - **Funded by [optional]:** [More Information Needed]
27
+ - **Shared by [optional]:** [More Information Needed]
28
+ - **Model type:** [More Information Needed]
29
+ - **Language(s) (NLP):** [More Information Needed]
30
+ - **License:** [More Information Needed]
31
+ - **Finetuned from model [optional]:** [More Information Needed]
32
+
33
+ ### Model Sources [optional]
34
+
35
+ <!-- Provide the basic links for the model. -->
36
+
37
+ - **Repository:** [More Information Needed]
38
+ - **Paper [optional]:** [More Information Needed]
39
+ - **Demo [optional]:** [More Information Needed]
40
+
41
+ ## Uses
42
+
43
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
44
+
45
+ ### Direct Use
46
+
47
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
48
+
49
+ [More Information Needed]
50
+
51
+ ### Downstream Use [optional]
52
+
53
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
54
+
55
+ [More Information Needed]
56
+
57
+ ### Out-of-Scope Use
58
+
59
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
60
+
61
+ [More Information Needed]
62
+
63
+ ## Bias, Risks, and Limitations
64
+
65
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
66
+
67
+ [More Information Needed]
68
+
69
+ ### Recommendations
70
+
71
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
72
+
73
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
74
+
75
+ ## How to Get Started with the Model
76
+
77
+ Use the code below to get started with the model.
78
+
79
+ [More Information Needed]
80
+
81
+ ## Training Details
82
+
83
+ ### Training Data
84
+
85
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
86
+
87
+ [More Information Needed]
88
+
89
+ ### Training Procedure
90
+
91
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
92
+
93
+ #### Preprocessing [optional]
94
+
95
+ [More Information Needed]
96
+
97
+
98
+ #### Training Hyperparameters
99
+
100
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
101
+
102
+ #### Speeds, Sizes, Times [optional]
103
+
104
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
105
+
106
+ [More Information Needed]
107
+
108
+ ## Evaluation
109
+
110
+ <!-- This section describes the evaluation protocols and provides the results. -->
111
+
112
+ ### Testing Data, Factors & Metrics
113
+
114
+ #### Testing Data
115
+
116
+ <!-- This should link to a Dataset Card if possible. -->
117
+
118
+ [More Information Needed]
119
+
120
+ #### Factors
121
+
122
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
123
+
124
+ [More Information Needed]
125
+
126
+ #### Metrics
127
+
128
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
129
+
130
+ [More Information Needed]
131
+
132
+ ### Results
133
+
134
+ [More Information Needed]
135
+
136
+ #### Summary
137
+
138
+
139
+
140
+ ## Model Examination [optional]
141
+
142
+ <!-- Relevant interpretability work for the model goes here -->
143
+
144
+ [More Information Needed]
145
+
146
+ ## Environmental Impact
147
+
148
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
149
+
150
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
151
+
152
+ - **Hardware Type:** [More Information Needed]
153
+ - **Hours used:** [More Information Needed]
154
+ - **Cloud Provider:** [More Information Needed]
155
+ - **Compute Region:** [More Information Needed]
156
+ - **Carbon Emitted:** [More Information Needed]
157
+
158
+ ## Technical Specifications [optional]
159
+
160
+ ### Model Architecture and Objective
161
+
162
+ [More Information Needed]
163
+
164
+ ### Compute Infrastructure
165
+
166
+ [More Information Needed]
167
+
168
+ #### Hardware
169
+
170
+ [More Information Needed]
171
+
172
+ #### Software
173
+
174
+ [More Information Needed]
175
+
176
+ ## Citation [optional]
177
+
178
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
179
+
180
+ **BibTeX:**
181
+
182
+ [More Information Needed]
183
+
184
+ **APA:**
185
+
186
+ [More Information Needed]
187
+
188
+ ## Glossary [optional]
189
+
190
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
191
+
192
+ [More Information Needed]
193
+
194
+ ## More Information [optional]
195
+
196
+ [More Information Needed]
197
+
198
+ ## Model Card Authors [optional]
199
+
200
+ [More Information Needed]
201
+
202
+ ## Model Card Contact
203
+
204
+ [More Information Needed]
205
+ ### Framework versions
206
+
207
+ - PEFT 0.18.0
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-Coder-7B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.0",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "q_proj",
33
+ "up_proj",
34
+ "down_proj",
35
+ "gate_proj",
36
+ "o_proj",
37
+ "k_proj",
38
+ "v_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2f2cb457d8005d8fbeacc2c5a906ddcf1a7817df4afb39f07145e8f5b1891458
3
+ size 161533192
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b5a50680b61825f7edac67bf005a74bce902ab2e5405c49e55d5732c8744353e
3
+ size 323296891
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fa200b054d0a6d9cad3a733d4bf54975bc90edf75b3a96333380f6b8d61d69db
3
+ size 15429
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/rng_state_1.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9177d5e2b228dde0470135d22f832f46d481c5d35a900940f502bd0052aa85a2
3
+ size 15429
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/rng_state_2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8d3c5d6fc77a29dc385cb763d154dd1a663af0a0aebf8bcb0913b35a2d66208e
3
+ size 15429
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/rng_state_3.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:38ce571ef553dcad189cd06383b995d927643fa108861d7a41780089c2fac382
3
+ size 15429
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:821999720016c63459198dd4460b2b8807ce8c28012372c5c4ccbae9a5f10684
3
+ size 1465
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|im_end|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
+ size 11421896
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/tokenizer_config.json ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ }
181
+ },
182
+ "additional_special_tokens": [
183
+ "<|im_start|>",
184
+ "<|im_end|>",
185
+ "<|object_ref_start|>",
186
+ "<|object_ref_end|>",
187
+ "<|box_start|>",
188
+ "<|box_end|>",
189
+ "<|quad_start|>",
190
+ "<|quad_end|>",
191
+ "<|vision_start|>",
192
+ "<|vision_end|>",
193
+ "<|vision_pad|>",
194
+ "<|image_pad|>",
195
+ "<|video_pad|>"
196
+ ],
197
+ "bos_token": null,
198
+ "clean_up_tokenization_spaces": false,
199
+ "eos_token": "<|im_end|>",
200
+ "errors": "replace",
201
+ "extra_special_tokens": {},
202
+ "model_max_length": 32768,
203
+ "pad_token": "<|endoftext|>",
204
+ "split_special_tokens": false,
205
+ "tokenizer_class": "Qwen2Tokenizer",
206
+ "unk_token": null
207
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/trainer_state.json ADDED
@@ -0,0 +1,1457 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.6846970215679562,
6
+ "eval_steps": 1000,
7
+ "global_step": 2000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.00034234851078397807,
14
+ "grad_norm": 0.38582974672317505,
15
+ "learning_rate": 0.0,
16
+ "loss": 1.2938,
17
+ "step": 1
18
+ },
19
+ {
20
+ "epoch": 0.003423485107839781,
21
+ "grad_norm": 0.3895561397075653,
22
+ "learning_rate": 7.692307692307692e-08,
23
+ "loss": 1.301,
24
+ "step": 10
25
+ },
26
+ {
27
+ "epoch": 0.006846970215679562,
28
+ "grad_norm": 0.38466933369636536,
29
+ "learning_rate": 1.6239316239316238e-07,
30
+ "loss": 1.3026,
31
+ "step": 20
32
+ },
33
+ {
34
+ "epoch": 0.010270455323519343,
35
+ "grad_norm": 0.3903484046459198,
36
+ "learning_rate": 2.478632478632479e-07,
37
+ "loss": 1.3113,
38
+ "step": 30
39
+ },
40
+ {
41
+ "epoch": 0.013693940431359124,
42
+ "grad_norm": 0.3902166783809662,
43
+ "learning_rate": 3.333333333333333e-07,
44
+ "loss": 1.3227,
45
+ "step": 40
46
+ },
47
+ {
48
+ "epoch": 0.017117425539198903,
49
+ "grad_norm": 0.39113983511924744,
50
+ "learning_rate": 4.1880341880341877e-07,
51
+ "loss": 1.3091,
52
+ "step": 50
53
+ },
54
+ {
55
+ "epoch": 0.020540910647038686,
56
+ "grad_norm": 0.4007033407688141,
57
+ "learning_rate": 5.042735042735042e-07,
58
+ "loss": 1.3183,
59
+ "step": 60
60
+ },
61
+ {
62
+ "epoch": 0.023964395754878465,
63
+ "grad_norm": 0.417681485414505,
64
+ "learning_rate": 5.897435897435898e-07,
65
+ "loss": 1.3217,
66
+ "step": 70
67
+ },
68
+ {
69
+ "epoch": 0.027387880862718247,
70
+ "grad_norm": 0.46474331617355347,
71
+ "learning_rate": 6.752136752136752e-07,
72
+ "loss": 1.296,
73
+ "step": 80
74
+ },
75
+ {
76
+ "epoch": 0.030811365970558027,
77
+ "grad_norm": 0.418237566947937,
78
+ "learning_rate": 7.606837606837606e-07,
79
+ "loss": 1.3132,
80
+ "step": 90
81
+ },
82
+ {
83
+ "epoch": 0.034234851078397806,
84
+ "grad_norm": 0.4302998185157776,
85
+ "learning_rate": 8.461538461538461e-07,
86
+ "loss": 1.3084,
87
+ "step": 100
88
+ },
89
+ {
90
+ "epoch": 0.03765833618623759,
91
+ "grad_norm": 0.41094598174095154,
92
+ "learning_rate": 9.316239316239316e-07,
93
+ "loss": 1.2953,
94
+ "step": 110
95
+ },
96
+ {
97
+ "epoch": 0.04108182129407737,
98
+ "grad_norm": 0.4528387188911438,
99
+ "learning_rate": 9.999996988736778e-07,
100
+ "loss": 1.2961,
101
+ "step": 120
102
+ },
103
+ {
104
+ "epoch": 0.044505306401917154,
105
+ "grad_norm": 0.4192649722099304,
106
+ "learning_rate": 9.99989159490489e-07,
107
+ "loss": 1.2894,
108
+ "step": 130
109
+ },
110
+ {
111
+ "epoch": 0.04792879150975693,
112
+ "grad_norm": 0.4704149663448334,
113
+ "learning_rate": 9.999635641539022e-07,
114
+ "loss": 1.2844,
115
+ "step": 140
116
+ },
117
+ {
118
+ "epoch": 0.05135227661759671,
119
+ "grad_norm": 0.45329269766807556,
120
+ "learning_rate": 9.99922913634658e-07,
121
+ "loss": 1.293,
122
+ "step": 150
123
+ },
124
+ {
125
+ "epoch": 0.054775761725436495,
126
+ "grad_norm": 0.45733365416526794,
127
+ "learning_rate": 9.998672091568482e-07,
128
+ "loss": 1.2657,
129
+ "step": 160
130
+ },
131
+ {
132
+ "epoch": 0.05819924683327628,
133
+ "grad_norm": 0.4719047248363495,
134
+ "learning_rate": 9.997964523978768e-07,
135
+ "loss": 1.2579,
136
+ "step": 170
137
+ },
138
+ {
139
+ "epoch": 0.06162273194111605,
140
+ "grad_norm": 0.4891057014465332,
141
+ "learning_rate": 9.99710645488411e-07,
142
+ "loss": 1.2544,
143
+ "step": 180
144
+ },
145
+ {
146
+ "epoch": 0.06504621704895584,
147
+ "grad_norm": 0.4674888551235199,
148
+ "learning_rate": 9.996097910123165e-07,
149
+ "loss": 1.2361,
150
+ "step": 190
151
+ },
152
+ {
153
+ "epoch": 0.06846970215679561,
154
+ "grad_norm": 0.4695661962032318,
155
+ "learning_rate": 9.9949389200658e-07,
156
+ "loss": 1.2113,
157
+ "step": 200
158
+ },
159
+ {
160
+ "epoch": 0.0718931872646354,
161
+ "grad_norm": 0.47163236141204834,
162
+ "learning_rate": 9.993629519612165e-07,
163
+ "loss": 1.214,
164
+ "step": 210
165
+ },
166
+ {
167
+ "epoch": 0.07531667237247518,
168
+ "grad_norm": 0.4599274694919586,
169
+ "learning_rate": 9.992169748191667e-07,
170
+ "loss": 1.1921,
171
+ "step": 220
172
+ },
173
+ {
174
+ "epoch": 0.07874015748031496,
175
+ "grad_norm": 0.4423569142818451,
176
+ "learning_rate": 9.990559649761758e-07,
177
+ "loss": 1.1744,
178
+ "step": 230
179
+ },
180
+ {
181
+ "epoch": 0.08216364258815474,
182
+ "grad_norm": 0.4392240047454834,
183
+ "learning_rate": 9.988799272806623e-07,
184
+ "loss": 1.1789,
185
+ "step": 240
186
+ },
187
+ {
188
+ "epoch": 0.08558712769599452,
189
+ "grad_norm": 0.42201587557792664,
190
+ "learning_rate": 9.986888670335715e-07,
191
+ "loss": 1.1489,
192
+ "step": 250
193
+ },
194
+ {
195
+ "epoch": 0.08901061280383431,
196
+ "grad_norm": 0.4007083773612976,
197
+ "learning_rate": 9.984827899882168e-07,
198
+ "loss": 1.128,
199
+ "step": 260
200
+ },
201
+ {
202
+ "epoch": 0.09243409791167409,
203
+ "grad_norm": 0.3840133249759674,
204
+ "learning_rate": 9.982617023501056e-07,
205
+ "loss": 1.1276,
206
+ "step": 270
207
+ },
208
+ {
209
+ "epoch": 0.09585758301951386,
210
+ "grad_norm": 0.36397966742515564,
211
+ "learning_rate": 9.980256107767522e-07,
212
+ "loss": 1.0973,
213
+ "step": 280
214
+ },
215
+ {
216
+ "epoch": 0.09928106812735364,
217
+ "grad_norm": 0.3815755248069763,
218
+ "learning_rate": 9.977745223774784e-07,
219
+ "loss": 1.0934,
220
+ "step": 290
221
+ },
222
+ {
223
+ "epoch": 0.10270455323519342,
224
+ "grad_norm": 0.33746176958084106,
225
+ "learning_rate": 9.975084447131989e-07,
226
+ "loss": 1.0887,
227
+ "step": 300
228
+ },
229
+ {
230
+ "epoch": 0.1061280383430332,
231
+ "grad_norm": 0.3245246112346649,
232
+ "learning_rate": 9.972273857961928e-07,
233
+ "loss": 1.0705,
234
+ "step": 310
235
+ },
236
+ {
237
+ "epoch": 0.10955152345087299,
238
+ "grad_norm": 0.3304518759250641,
239
+ "learning_rate": 9.969313540898638e-07,
240
+ "loss": 1.0617,
241
+ "step": 320
242
+ },
243
+ {
244
+ "epoch": 0.11297500855871277,
245
+ "grad_norm": 0.3013385832309723,
246
+ "learning_rate": 9.966203585084841e-07,
247
+ "loss": 1.0416,
248
+ "step": 330
249
+ },
250
+ {
251
+ "epoch": 0.11639849366655255,
252
+ "grad_norm": 0.30748993158340454,
253
+ "learning_rate": 9.962944084169267e-07,
254
+ "loss": 1.0319,
255
+ "step": 340
256
+ },
257
+ {
258
+ "epoch": 0.11982197877439234,
259
+ "grad_norm": 0.2813571095466614,
260
+ "learning_rate": 9.959535136303834e-07,
261
+ "loss": 1.0233,
262
+ "step": 350
263
+ },
264
+ {
265
+ "epoch": 0.1232454638822321,
266
+ "grad_norm": 0.2859013080596924,
267
+ "learning_rate": 9.955976844140688e-07,
268
+ "loss": 1.0239,
269
+ "step": 360
270
+ },
271
+ {
272
+ "epoch": 0.1266689489900719,
273
+ "grad_norm": 0.27636730670928955,
274
+ "learning_rate": 9.95226931482911e-07,
275
+ "loss": 1.012,
276
+ "step": 370
277
+ },
278
+ {
279
+ "epoch": 0.13009243409791169,
280
+ "grad_norm": 0.27530768513679504,
281
+ "learning_rate": 9.948412660012305e-07,
282
+ "loss": 1.0094,
283
+ "step": 380
284
+ },
285
+ {
286
+ "epoch": 0.13351591920575145,
287
+ "grad_norm": 0.2697387635707855,
288
+ "learning_rate": 9.944406995824013e-07,
289
+ "loss": 1.0136,
290
+ "step": 390
291
+ },
292
+ {
293
+ "epoch": 0.13693940431359122,
294
+ "grad_norm": 0.2743023931980133,
295
+ "learning_rate": 9.94025244288504e-07,
296
+ "loss": 0.9975,
297
+ "step": 400
298
+ },
299
+ {
300
+ "epoch": 0.14036288942143102,
301
+ "grad_norm": 0.27155691385269165,
302
+ "learning_rate": 9.93594912629961e-07,
303
+ "loss": 0.9849,
304
+ "step": 410
305
+ },
306
+ {
307
+ "epoch": 0.1437863745292708,
308
+ "grad_norm": 0.2568703889846802,
309
+ "learning_rate": 9.9314971756516e-07,
310
+ "loss": 0.9918,
311
+ "step": 420
312
+ },
313
+ {
314
+ "epoch": 0.14720985963711058,
315
+ "grad_norm": 0.25010108947753906,
316
+ "learning_rate": 9.926896725000637e-07,
317
+ "loss": 0.9723,
318
+ "step": 430
319
+ },
320
+ {
321
+ "epoch": 0.15063334474495035,
322
+ "grad_norm": 0.25133389234542847,
323
+ "learning_rate": 9.92214791287807e-07,
324
+ "loss": 0.9636,
325
+ "step": 440
326
+ },
327
+ {
328
+ "epoch": 0.15405682985279015,
329
+ "grad_norm": 0.2505231499671936,
330
+ "learning_rate": 9.917250882282785e-07,
331
+ "loss": 0.9578,
332
+ "step": 450
333
+ },
334
+ {
335
+ "epoch": 0.15748031496062992,
336
+ "grad_norm": 0.2534894049167633,
337
+ "learning_rate": 9.912205780676905e-07,
338
+ "loss": 0.9524,
339
+ "step": 460
340
+ },
341
+ {
342
+ "epoch": 0.16090380006846972,
343
+ "grad_norm": 0.23746544122695923,
344
+ "learning_rate": 9.90701275998136e-07,
345
+ "loss": 0.9538,
346
+ "step": 470
347
+ },
348
+ {
349
+ "epoch": 0.16432728517630948,
350
+ "grad_norm": 0.22148397564888,
351
+ "learning_rate": 9.901671976571288e-07,
352
+ "loss": 0.9469,
353
+ "step": 480
354
+ },
355
+ {
356
+ "epoch": 0.16775077028414925,
357
+ "grad_norm": 0.23445384204387665,
358
+ "learning_rate": 9.896183591271354e-07,
359
+ "loss": 0.9428,
360
+ "step": 490
361
+ },
362
+ {
363
+ "epoch": 0.17117425539198905,
364
+ "grad_norm": 0.22655028104782104,
365
+ "learning_rate": 9.890547769350886e-07,
366
+ "loss": 0.9343,
367
+ "step": 500
368
+ },
369
+ {
370
+ "epoch": 0.17459774049982882,
371
+ "grad_norm": 0.2170156091451645,
372
+ "learning_rate": 9.884764680518906e-07,
373
+ "loss": 0.9345,
374
+ "step": 510
375
+ },
376
+ {
377
+ "epoch": 0.17802122560766862,
378
+ "grad_norm": 0.22584597766399384,
379
+ "learning_rate": 9.878834498919025e-07,
380
+ "loss": 0.9309,
381
+ "step": 520
382
+ },
383
+ {
384
+ "epoch": 0.18144471071550838,
385
+ "grad_norm": 0.21805278956890106,
386
+ "learning_rate": 9.872757403124185e-07,
387
+ "loss": 0.9229,
388
+ "step": 530
389
+ },
390
+ {
391
+ "epoch": 0.18486819582334818,
392
+ "grad_norm": 0.21373434364795685,
393
+ "learning_rate": 9.866533576131302e-07,
394
+ "loss": 0.9063,
395
+ "step": 540
396
+ },
397
+ {
398
+ "epoch": 0.18829168093118795,
399
+ "grad_norm": 0.21042130887508392,
400
+ "learning_rate": 9.860163205355733e-07,
401
+ "loss": 0.9146,
402
+ "step": 550
403
+ },
404
+ {
405
+ "epoch": 0.19171516603902772,
406
+ "grad_norm": 0.21380534768104553,
407
+ "learning_rate": 9.853646482625652e-07,
408
+ "loss": 0.9103,
409
+ "step": 560
410
+ },
411
+ {
412
+ "epoch": 0.19513865114686751,
413
+ "grad_norm": 0.20159614086151123,
414
+ "learning_rate": 9.846983604176258e-07,
415
+ "loss": 0.897,
416
+ "step": 570
417
+ },
418
+ {
419
+ "epoch": 0.19856213625470728,
420
+ "grad_norm": 0.2064715176820755,
421
+ "learning_rate": 9.840174770643876e-07,
422
+ "loss": 0.8973,
423
+ "step": 580
424
+ },
425
+ {
426
+ "epoch": 0.20198562136254708,
427
+ "grad_norm": 0.20663177967071533,
428
+ "learning_rate": 9.833220187059912e-07,
429
+ "loss": 0.892,
430
+ "step": 590
431
+ },
432
+ {
433
+ "epoch": 0.20540910647038685,
434
+ "grad_norm": 0.20841750502586365,
435
+ "learning_rate": 9.82612006284468e-07,
436
+ "loss": 0.8834,
437
+ "step": 600
438
+ },
439
+ {
440
+ "epoch": 0.20883259157822665,
441
+ "grad_norm": 0.19864782691001892,
442
+ "learning_rate": 9.818874611801095e-07,
443
+ "loss": 0.8793,
444
+ "step": 610
445
+ },
446
+ {
447
+ "epoch": 0.2122560766860664,
448
+ "grad_norm": 0.18866512179374695,
449
+ "learning_rate": 9.811484052108233e-07,
450
+ "loss": 0.8851,
451
+ "step": 620
452
+ },
453
+ {
454
+ "epoch": 0.21567956179390618,
455
+ "grad_norm": 0.19272519648075104,
456
+ "learning_rate": 9.803948606314761e-07,
457
+ "loss": 0.8783,
458
+ "step": 630
459
+ },
460
+ {
461
+ "epoch": 0.21910304690174598,
462
+ "grad_norm": 0.19224460422992706,
463
+ "learning_rate": 9.796268501332246e-07,
464
+ "loss": 0.8768,
465
+ "step": 640
466
+ },
467
+ {
468
+ "epoch": 0.22252653200958575,
469
+ "grad_norm": 0.18141649663448334,
470
+ "learning_rate": 9.788443968428301e-07,
471
+ "loss": 0.8667,
472
+ "step": 650
473
+ },
474
+ {
475
+ "epoch": 0.22595001711742554,
476
+ "grad_norm": 0.19055131077766418,
477
+ "learning_rate": 9.780475243219647e-07,
478
+ "loss": 0.8702,
479
+ "step": 660
480
+ },
481
+ {
482
+ "epoch": 0.2293735022252653,
483
+ "grad_norm": 0.1838882565498352,
484
+ "learning_rate": 9.77236256566499e-07,
485
+ "loss": 0.8618,
486
+ "step": 670
487
+ },
488
+ {
489
+ "epoch": 0.2327969873331051,
490
+ "grad_norm": 0.18793776631355286,
491
+ "learning_rate": 9.764106180057821e-07,
492
+ "loss": 0.8495,
493
+ "step": 680
494
+ },
495
+ {
496
+ "epoch": 0.23622047244094488,
497
+ "grad_norm": 0.18299244344234467,
498
+ "learning_rate": 9.755706335019044e-07,
499
+ "loss": 0.8459,
500
+ "step": 690
501
+ },
502
+ {
503
+ "epoch": 0.23964395754878468,
504
+ "grad_norm": 0.1785850077867508,
505
+ "learning_rate": 9.747163283489493e-07,
506
+ "loss": 0.8453,
507
+ "step": 700
508
+ },
509
+ {
510
+ "epoch": 0.24306744265662444,
511
+ "grad_norm": 0.16908957064151764,
512
+ "learning_rate": 9.738477282722317e-07,
513
+ "loss": 0.8342,
514
+ "step": 710
515
+ },
516
+ {
517
+ "epoch": 0.2464909277644642,
518
+ "grad_norm": 0.17376649379730225,
519
+ "learning_rate": 9.729648594275233e-07,
520
+ "loss": 0.8369,
521
+ "step": 720
522
+ },
523
+ {
524
+ "epoch": 0.249914412872304,
525
+ "grad_norm": 0.18265017867088318,
526
+ "learning_rate": 9.720677484002649e-07,
527
+ "loss": 0.8385,
528
+ "step": 730
529
+ },
530
+ {
531
+ "epoch": 0.2533378979801438,
532
+ "grad_norm": 0.17172974348068237,
533
+ "learning_rate": 9.711564222047658e-07,
534
+ "loss": 0.8454,
535
+ "step": 740
536
+ },
537
+ {
538
+ "epoch": 0.2567613830879836,
539
+ "grad_norm": 0.1695045828819275,
540
+ "learning_rate": 9.702309082833905e-07,
541
+ "loss": 0.8279,
542
+ "step": 750
543
+ },
544
+ {
545
+ "epoch": 0.26018486819582337,
546
+ "grad_norm": 0.16996362805366516,
547
+ "learning_rate": 9.692912345057318e-07,
548
+ "loss": 0.8362,
549
+ "step": 760
550
+ },
551
+ {
552
+ "epoch": 0.2636083533036631,
553
+ "grad_norm": 0.16263867914676666,
554
+ "learning_rate": 9.68337429167773e-07,
555
+ "loss": 0.8272,
556
+ "step": 770
557
+ },
558
+ {
559
+ "epoch": 0.2670318384115029,
560
+ "grad_norm": 0.17135751247406006,
561
+ "learning_rate": 9.673695209910336e-07,
562
+ "loss": 0.8379,
563
+ "step": 780
564
+ },
565
+ {
566
+ "epoch": 0.2704553235193427,
567
+ "grad_norm": 0.15958866477012634,
568
+ "learning_rate": 9.66387539121707e-07,
569
+ "loss": 0.8195,
570
+ "step": 790
571
+ },
572
+ {
573
+ "epoch": 0.27387880862718245,
574
+ "grad_norm": 0.1589757204055786,
575
+ "learning_rate": 9.653915131297802e-07,
576
+ "loss": 0.8102,
577
+ "step": 800
578
+ },
579
+ {
580
+ "epoch": 0.27730229373502224,
581
+ "grad_norm": 0.16261506080627441,
582
+ "learning_rate": 9.64381473008146e-07,
583
+ "loss": 0.8155,
584
+ "step": 810
585
+ },
586
+ {
587
+ "epoch": 0.28072577884286204,
588
+ "grad_norm": 0.15770071744918823,
589
+ "learning_rate": 9.63357449171697e-07,
590
+ "loss": 0.8176,
591
+ "step": 820
592
+ },
593
+ {
594
+ "epoch": 0.28414926395070184,
595
+ "grad_norm": 0.14913444221019745,
596
+ "learning_rate": 9.623194724564127e-07,
597
+ "loss": 0.8148,
598
+ "step": 830
599
+ },
600
+ {
601
+ "epoch": 0.2875727490585416,
602
+ "grad_norm": 0.15190348029136658,
603
+ "learning_rate": 9.612675741184289e-07,
604
+ "loss": 0.8011,
605
+ "step": 840
606
+ },
607
+ {
608
+ "epoch": 0.2909962341663814,
609
+ "grad_norm": 0.14866051077842712,
610
+ "learning_rate": 9.602017858330964e-07,
611
+ "loss": 0.8132,
612
+ "step": 850
613
+ },
614
+ {
615
+ "epoch": 0.29441971927422117,
616
+ "grad_norm": 0.1489679515361786,
617
+ "learning_rate": 9.591221396940294e-07,
618
+ "loss": 0.8076,
619
+ "step": 860
620
+ },
621
+ {
622
+ "epoch": 0.2978432043820609,
623
+ "grad_norm": 0.1507052481174469,
624
+ "learning_rate": 9.580286682121362e-07,
625
+ "loss": 0.8035,
626
+ "step": 870
627
+ },
628
+ {
629
+ "epoch": 0.3012666894899007,
630
+ "grad_norm": 0.14822350442409515,
631
+ "learning_rate": 9.56921404314642e-07,
632
+ "loss": 0.799,
633
+ "step": 880
634
+ },
635
+ {
636
+ "epoch": 0.3046901745977405,
637
+ "grad_norm": 0.1420404314994812,
638
+ "learning_rate": 9.558003813440972e-07,
639
+ "loss": 0.7885,
640
+ "step": 890
641
+ },
642
+ {
643
+ "epoch": 0.3081136597055803,
644
+ "grad_norm": 0.1396367996931076,
645
+ "learning_rate": 9.546656330573726e-07,
646
+ "loss": 0.7869,
647
+ "step": 900
648
+ },
649
+ {
650
+ "epoch": 0.31153714481342004,
651
+ "grad_norm": 0.14706209301948547,
652
+ "learning_rate": 9.535171936246441e-07,
653
+ "loss": 0.79,
654
+ "step": 910
655
+ },
656
+ {
657
+ "epoch": 0.31496062992125984,
658
+ "grad_norm": 0.1379164308309555,
659
+ "learning_rate": 9.523550976283623e-07,
660
+ "loss": 0.7834,
661
+ "step": 920
662
+ },
663
+ {
664
+ "epoch": 0.31838411502909963,
665
+ "grad_norm": 0.137995183467865,
666
+ "learning_rate": 9.511793800622123e-07,
667
+ "loss": 0.7837,
668
+ "step": 930
669
+ },
670
+ {
671
+ "epoch": 0.32180760013693943,
672
+ "grad_norm": 0.1464492380619049,
673
+ "learning_rate": 9.499900763300596e-07,
674
+ "loss": 0.7861,
675
+ "step": 940
676
+ },
677
+ {
678
+ "epoch": 0.3252310852447792,
679
+ "grad_norm": 0.14009970426559448,
680
+ "learning_rate": 9.487872222448838e-07,
681
+ "loss": 0.7825,
682
+ "step": 950
683
+ },
684
+ {
685
+ "epoch": 0.32865457035261897,
686
+ "grad_norm": 0.1352413296699524,
687
+ "learning_rate": 9.475708540277001e-07,
688
+ "loss": 0.7773,
689
+ "step": 960
690
+ },
691
+ {
692
+ "epoch": 0.33207805546045877,
693
+ "grad_norm": 0.1339689940214157,
694
+ "learning_rate": 9.463410083064692e-07,
695
+ "loss": 0.7737,
696
+ "step": 970
697
+ },
698
+ {
699
+ "epoch": 0.3355015405682985,
700
+ "grad_norm": 0.14391706883907318,
701
+ "learning_rate": 9.450977221149935e-07,
702
+ "loss": 0.7741,
703
+ "step": 980
704
+ },
705
+ {
706
+ "epoch": 0.3389250256761383,
707
+ "grad_norm": 0.1347951740026474,
708
+ "learning_rate": 9.438410328918029e-07,
709
+ "loss": 0.7712,
710
+ "step": 990
711
+ },
712
+ {
713
+ "epoch": 0.3423485107839781,
714
+ "grad_norm": 0.14170120656490326,
715
+ "learning_rate": 9.425709784790266e-07,
716
+ "loss": 0.7726,
717
+ "step": 1000
718
+ },
719
+ {
720
+ "epoch": 0.3423485107839781,
721
+ "eval_loss": 0.503926157951355,
722
+ "eval_runtime": 2.912,
723
+ "eval_samples_per_second": 67.65,
724
+ "eval_steps_per_second": 2.404,
725
+ "step": 1000
726
+ },
727
+ {
728
+ "epoch": 0.3457719958918179,
729
+ "grad_norm": 0.13902287185192108,
730
+ "learning_rate": 9.412875971212539e-07,
731
+ "loss": 0.7748,
732
+ "step": 1010
733
+ },
734
+ {
735
+ "epoch": 0.34919548099965764,
736
+ "grad_norm": 0.13446791470050812,
737
+ "learning_rate": 9.399909274643823e-07,
738
+ "loss": 0.7616,
739
+ "step": 1020
740
+ },
741
+ {
742
+ "epoch": 0.35261896610749743,
743
+ "grad_norm": 0.13200196623802185,
744
+ "learning_rate": 9.386810085544546e-07,
745
+ "loss": 0.7621,
746
+ "step": 1030
747
+ },
748
+ {
749
+ "epoch": 0.35604245121533723,
750
+ "grad_norm": 0.13774636387825012,
751
+ "learning_rate": 9.373578798364815e-07,
752
+ "loss": 0.7688,
753
+ "step": 1040
754
+ },
755
+ {
756
+ "epoch": 0.35946593632317697,
757
+ "grad_norm": 0.13428834080696106,
758
+ "learning_rate": 9.360215811532561e-07,
759
+ "loss": 0.7596,
760
+ "step": 1050
761
+ },
762
+ {
763
+ "epoch": 0.36288942143101677,
764
+ "grad_norm": 0.13146500289440155,
765
+ "learning_rate": 9.346721527441519e-07,
766
+ "loss": 0.7606,
767
+ "step": 1060
768
+ },
769
+ {
770
+ "epoch": 0.36631290653885656,
771
+ "grad_norm": 0.1290307193994522,
772
+ "learning_rate": 9.333096352439125e-07,
773
+ "loss": 0.7576,
774
+ "step": 1070
775
+ },
776
+ {
777
+ "epoch": 0.36973639164669636,
778
+ "grad_norm": 0.12617669999599457,
779
+ "learning_rate": 9.319340696814273e-07,
780
+ "loss": 0.7616,
781
+ "step": 1080
782
+ },
783
+ {
784
+ "epoch": 0.3731598767545361,
785
+ "grad_norm": 0.13011223077774048,
786
+ "learning_rate": 9.305454974784965e-07,
787
+ "loss": 0.7509,
788
+ "step": 1090
789
+ },
790
+ {
791
+ "epoch": 0.3765833618623759,
792
+ "grad_norm": 0.13311044871807098,
793
+ "learning_rate": 9.291439604485835e-07,
794
+ "loss": 0.7473,
795
+ "step": 1100
796
+ },
797
+ {
798
+ "epoch": 0.3800068469702157,
799
+ "grad_norm": 0.12737637758255005,
800
+ "learning_rate": 9.277295007955555e-07,
801
+ "loss": 0.7411,
802
+ "step": 1110
803
+ },
804
+ {
805
+ "epoch": 0.38343033207805544,
806
+ "grad_norm": 0.12550155818462372,
807
+ "learning_rate": 9.263021611124134e-07,
808
+ "loss": 0.7487,
809
+ "step": 1120
810
+ },
811
+ {
812
+ "epoch": 0.38685381718589523,
813
+ "grad_norm": 0.12280461937189102,
814
+ "learning_rate": 9.248619843800083e-07,
815
+ "loss": 0.75,
816
+ "step": 1130
817
+ },
818
+ {
819
+ "epoch": 0.39027730229373503,
820
+ "grad_norm": 0.1246493011713028,
821
+ "learning_rate": 9.234090139657484e-07,
822
+ "loss": 0.7464,
823
+ "step": 1140
824
+ },
825
+ {
826
+ "epoch": 0.3937007874015748,
827
+ "grad_norm": 0.12128277122974396,
828
+ "learning_rate": 9.219432936222916e-07,
829
+ "loss": 0.7399,
830
+ "step": 1150
831
+ },
832
+ {
833
+ "epoch": 0.39712427250941457,
834
+ "grad_norm": 0.13011115789413452,
835
+ "learning_rate": 9.204648674862295e-07,
836
+ "loss": 0.7441,
837
+ "step": 1160
838
+ },
839
+ {
840
+ "epoch": 0.40054775761725436,
841
+ "grad_norm": 0.12489533424377441,
842
+ "learning_rate": 9.189737800767571e-07,
843
+ "loss": 0.7387,
844
+ "step": 1170
845
+ },
846
+ {
847
+ "epoch": 0.40397124272509416,
848
+ "grad_norm": 0.12287454307079315,
849
+ "learning_rate": 9.174700762943332e-07,
850
+ "loss": 0.7395,
851
+ "step": 1180
852
+ },
853
+ {
854
+ "epoch": 0.4073947278329339,
855
+ "grad_norm": 0.12373016774654388,
856
+ "learning_rate": 9.159538014193276e-07,
857
+ "loss": 0.7397,
858
+ "step": 1190
859
+ },
860
+ {
861
+ "epoch": 0.4108182129407737,
862
+ "grad_norm": 0.12014248222112656,
863
+ "learning_rate": 9.144250011106578e-07,
864
+ "loss": 0.7292,
865
+ "step": 1200
866
+ },
867
+ {
868
+ "epoch": 0.4142416980486135,
869
+ "grad_norm": 0.12502500414848328,
870
+ "learning_rate": 9.128837214044145e-07,
871
+ "loss": 0.7337,
872
+ "step": 1210
873
+ },
874
+ {
875
+ "epoch": 0.4176651831564533,
876
+ "grad_norm": 0.12515854835510254,
877
+ "learning_rate": 9.113300087124746e-07,
878
+ "loss": 0.7326,
879
+ "step": 1220
880
+ },
881
+ {
882
+ "epoch": 0.42108866826429303,
883
+ "grad_norm": 0.12971976399421692,
884
+ "learning_rate": 9.097639098211045e-07,
885
+ "loss": 0.723,
886
+ "step": 1230
887
+ },
888
+ {
889
+ "epoch": 0.4245121533721328,
890
+ "grad_norm": 0.12565718591213226,
891
+ "learning_rate": 9.081854718895503e-07,
892
+ "loss": 0.7284,
893
+ "step": 1240
894
+ },
895
+ {
896
+ "epoch": 0.4279356384799726,
897
+ "grad_norm": 0.12475177645683289,
898
+ "learning_rate": 9.065947424486186e-07,
899
+ "loss": 0.7211,
900
+ "step": 1250
901
+ },
902
+ {
903
+ "epoch": 0.43135912358781237,
904
+ "grad_norm": 0.12246546149253845,
905
+ "learning_rate": 9.049917693992444e-07,
906
+ "loss": 0.729,
907
+ "step": 1260
908
+ },
909
+ {
910
+ "epoch": 0.43478260869565216,
911
+ "grad_norm": 0.12223733216524124,
912
+ "learning_rate": 9.033766010110495e-07,
913
+ "loss": 0.7207,
914
+ "step": 1270
915
+ },
916
+ {
917
+ "epoch": 0.43820609380349196,
918
+ "grad_norm": 0.12425760179758072,
919
+ "learning_rate": 9.017492859208881e-07,
920
+ "loss": 0.7184,
921
+ "step": 1280
922
+ },
923
+ {
924
+ "epoch": 0.44162957891133175,
925
+ "grad_norm": 0.1250336468219757,
926
+ "learning_rate": 9.001098731313833e-07,
927
+ "loss": 0.7144,
928
+ "step": 1290
929
+ },
930
+ {
931
+ "epoch": 0.4450530640191715,
932
+ "grad_norm": 0.12002287805080414,
933
+ "learning_rate": 8.984584120094503e-07,
934
+ "loss": 0.7224,
935
+ "step": 1300
936
+ },
937
+ {
938
+ "epoch": 0.4484765491270113,
939
+ "grad_norm": 0.12844647467136383,
940
+ "learning_rate": 8.967949522848106e-07,
941
+ "loss": 0.7171,
942
+ "step": 1310
943
+ },
944
+ {
945
+ "epoch": 0.4519000342348511,
946
+ "grad_norm": 0.13210491836071014,
947
+ "learning_rate": 8.951195440484946e-07,
948
+ "loss": 0.7132,
949
+ "step": 1320
950
+ },
951
+ {
952
+ "epoch": 0.4553235193426909,
953
+ "grad_norm": 0.11971520632505417,
954
+ "learning_rate": 8.934322377513327e-07,
955
+ "loss": 0.7134,
956
+ "step": 1330
957
+ },
958
+ {
959
+ "epoch": 0.4587470044505306,
960
+ "grad_norm": 0.12985177338123322,
961
+ "learning_rate": 8.917330842024364e-07,
962
+ "loss": 0.7148,
963
+ "step": 1340
964
+ },
965
+ {
966
+ "epoch": 0.4621704895583704,
967
+ "grad_norm": 0.12479525059461594,
968
+ "learning_rate": 8.900221345676685e-07,
969
+ "loss": 0.7092,
970
+ "step": 1350
971
+ },
972
+ {
973
+ "epoch": 0.4655939746662102,
974
+ "grad_norm": 0.11902341991662979,
975
+ "learning_rate": 8.882994403681018e-07,
976
+ "loss": 0.7064,
977
+ "step": 1360
978
+ },
979
+ {
980
+ "epoch": 0.46901745977404996,
981
+ "grad_norm": 0.1295621246099472,
982
+ "learning_rate": 8.865650534784682e-07,
983
+ "loss": 0.7118,
984
+ "step": 1370
985
+ },
986
+ {
987
+ "epoch": 0.47244094488188976,
988
+ "grad_norm": 0.12812475860118866,
989
+ "learning_rate": 8.848190261255964e-07,
990
+ "loss": 0.7061,
991
+ "step": 1380
992
+ },
993
+ {
994
+ "epoch": 0.47586442998972955,
995
+ "grad_norm": 0.12716683745384216,
996
+ "learning_rate": 8.830614108868395e-07,
997
+ "loss": 0.7068,
998
+ "step": 1390
999
+ },
1000
+ {
1001
+ "epoch": 0.47928791509756935,
1002
+ "grad_norm": 0.12054262310266495,
1003
+ "learning_rate": 8.812922606884907e-07,
1004
+ "loss": 0.7025,
1005
+ "step": 1400
1006
+ },
1007
+ {
1008
+ "epoch": 0.4827114002054091,
1009
+ "grad_norm": 0.12071160227060318,
1010
+ "learning_rate": 8.795116288041914e-07,
1011
+ "loss": 0.6948,
1012
+ "step": 1410
1013
+ },
1014
+ {
1015
+ "epoch": 0.4861348853132489,
1016
+ "grad_norm": 0.12057632952928543,
1017
+ "learning_rate": 8.777195688533254e-07,
1018
+ "loss": 0.703,
1019
+ "step": 1420
1020
+ },
1021
+ {
1022
+ "epoch": 0.4895583704210887,
1023
+ "grad_norm": 0.13040630519390106,
1024
+ "learning_rate": 8.759161347994045e-07,
1025
+ "loss": 0.7056,
1026
+ "step": 1430
1027
+ },
1028
+ {
1029
+ "epoch": 0.4929818555289284,
1030
+ "grad_norm": 0.11976247280836105,
1031
+ "learning_rate": 8.741013809484448e-07,
1032
+ "loss": 0.7019,
1033
+ "step": 1440
1034
+ },
1035
+ {
1036
+ "epoch": 0.4964053406367682,
1037
+ "grad_norm": 0.12721805274486542,
1038
+ "learning_rate": 8.722753619473296e-07,
1039
+ "loss": 0.7043,
1040
+ "step": 1450
1041
+ },
1042
+ {
1043
+ "epoch": 0.499828825744608,
1044
+ "grad_norm": 0.12659522891044617,
1045
+ "learning_rate": 8.704381327821649e-07,
1046
+ "loss": 0.6999,
1047
+ "step": 1460
1048
+ },
1049
+ {
1050
+ "epoch": 0.5032523108524478,
1051
+ "grad_norm": 0.11685533076524734,
1052
+ "learning_rate": 8.685897487766239e-07,
1053
+ "loss": 0.693,
1054
+ "step": 1470
1055
+ },
1056
+ {
1057
+ "epoch": 0.5066757959602876,
1058
+ "grad_norm": 0.12018760293722153,
1059
+ "learning_rate": 8.667302655902802e-07,
1060
+ "loss": 0.691,
1061
+ "step": 1480
1062
+ },
1063
+ {
1064
+ "epoch": 0.5100992810681274,
1065
+ "grad_norm": 0.12754450738430023,
1066
+ "learning_rate": 8.648597392169318e-07,
1067
+ "loss": 0.6934,
1068
+ "step": 1490
1069
+ },
1070
+ {
1071
+ "epoch": 0.5135227661759671,
1072
+ "grad_norm": 0.12374494224786758,
1073
+ "learning_rate": 8.629782259829162e-07,
1074
+ "loss": 0.6916,
1075
+ "step": 1500
1076
+ },
1077
+ {
1078
+ "epoch": 0.5169462512838069,
1079
+ "grad_norm": 0.11604287475347519,
1080
+ "learning_rate": 8.610857825454128e-07,
1081
+ "loss": 0.6965,
1082
+ "step": 1510
1083
+ },
1084
+ {
1085
+ "epoch": 0.5203697363916467,
1086
+ "grad_norm": 0.12265919893980026,
1087
+ "learning_rate": 8.591824658907374e-07,
1088
+ "loss": 0.6879,
1089
+ "step": 1520
1090
+ },
1091
+ {
1092
+ "epoch": 0.5237932214994865,
1093
+ "grad_norm": 0.1258995532989502,
1094
+ "learning_rate": 8.572683333326265e-07,
1095
+ "loss": 0.6925,
1096
+ "step": 1530
1097
+ },
1098
+ {
1099
+ "epoch": 0.5272167066073262,
1100
+ "grad_norm": 0.118409663438797,
1101
+ "learning_rate": 8.553434425105108e-07,
1102
+ "loss": 0.6918,
1103
+ "step": 1540
1104
+ },
1105
+ {
1106
+ "epoch": 0.5306401917151661,
1107
+ "grad_norm": 0.12140092998743057,
1108
+ "learning_rate": 8.5340785138778e-07,
1109
+ "loss": 0.6891,
1110
+ "step": 1550
1111
+ },
1112
+ {
1113
+ "epoch": 0.5340636768230058,
1114
+ "grad_norm": 0.12385173887014389,
1115
+ "learning_rate": 8.514616182500376e-07,
1116
+ "loss": 0.6868,
1117
+ "step": 1560
1118
+ },
1119
+ {
1120
+ "epoch": 0.5374871619308456,
1121
+ "grad_norm": 0.11857569962739944,
1122
+ "learning_rate": 8.495048017033448e-07,
1123
+ "loss": 0.6844,
1124
+ "step": 1570
1125
+ },
1126
+ {
1127
+ "epoch": 0.5409106470386854,
1128
+ "grad_norm": 0.119228295981884,
1129
+ "learning_rate": 8.475374606724567e-07,
1130
+ "loss": 0.6803,
1131
+ "step": 1580
1132
+ },
1133
+ {
1134
+ "epoch": 0.5443341321465252,
1135
+ "grad_norm": 0.11927839368581772,
1136
+ "learning_rate": 8.455596543990474e-07,
1137
+ "loss": 0.6877,
1138
+ "step": 1590
1139
+ },
1140
+ {
1141
+ "epoch": 0.5477576172543649,
1142
+ "grad_norm": 0.11546283960342407,
1143
+ "learning_rate": 8.435714424399265e-07,
1144
+ "loss": 0.6828,
1145
+ "step": 1600
1146
+ },
1147
+ {
1148
+ "epoch": 0.5511811023622047,
1149
+ "grad_norm": 0.11681130528450012,
1150
+ "learning_rate": 8.41572884665245e-07,
1151
+ "loss": 0.6772,
1152
+ "step": 1610
1153
+ },
1154
+ {
1155
+ "epoch": 0.5546045874700445,
1156
+ "grad_norm": 0.11748861521482468,
1157
+ "learning_rate": 8.395640412566933e-07,
1158
+ "loss": 0.6796,
1159
+ "step": 1620
1160
+ },
1161
+ {
1162
+ "epoch": 0.5580280725778843,
1163
+ "grad_norm": 0.11813998967409134,
1164
+ "learning_rate": 8.375449727056885e-07,
1165
+ "loss": 0.6798,
1166
+ "step": 1630
1167
+ },
1168
+ {
1169
+ "epoch": 0.5614515576857241,
1170
+ "grad_norm": 0.11564704775810242,
1171
+ "learning_rate": 8.355157398115524e-07,
1172
+ "loss": 0.6808,
1173
+ "step": 1640
1174
+ },
1175
+ {
1176
+ "epoch": 0.5648750427935638,
1177
+ "grad_norm": 0.12719090282917023,
1178
+ "learning_rate": 8.334764036796822e-07,
1179
+ "loss": 0.6799,
1180
+ "step": 1650
1181
+ },
1182
+ {
1183
+ "epoch": 0.5682985279014037,
1184
+ "grad_norm": 0.12066428363323212,
1185
+ "learning_rate": 8.314270257197084e-07,
1186
+ "loss": 0.68,
1187
+ "step": 1660
1188
+ },
1189
+ {
1190
+ "epoch": 0.5717220130092434,
1191
+ "grad_norm": 0.11323338001966476,
1192
+ "learning_rate": 8.293676676436474e-07,
1193
+ "loss": 0.6714,
1194
+ "step": 1670
1195
+ },
1196
+ {
1197
+ "epoch": 0.5751454981170832,
1198
+ "grad_norm": 0.11135194450616837,
1199
+ "learning_rate": 8.272983914640419e-07,
1200
+ "loss": 0.6717,
1201
+ "step": 1680
1202
+ },
1203
+ {
1204
+ "epoch": 0.578568983224923,
1205
+ "grad_norm": 0.11780041456222534,
1206
+ "learning_rate": 8.252192594920944e-07,
1207
+ "loss": 0.6787,
1208
+ "step": 1690
1209
+ },
1210
+ {
1211
+ "epoch": 0.5819924683327627,
1212
+ "grad_norm": 0.11305012553930283,
1213
+ "learning_rate": 8.231303343357905e-07,
1214
+ "loss": 0.6755,
1215
+ "step": 1700
1216
+ },
1217
+ {
1218
+ "epoch": 0.5854159534406025,
1219
+ "grad_norm": 0.11846613138914108,
1220
+ "learning_rate": 8.210316788980136e-07,
1221
+ "loss": 0.6727,
1222
+ "step": 1710
1223
+ },
1224
+ {
1225
+ "epoch": 0.5888394385484423,
1226
+ "grad_norm": 0.11588136106729507,
1227
+ "learning_rate": 8.189233563746507e-07,
1228
+ "loss": 0.6607,
1229
+ "step": 1720
1230
+ },
1231
+ {
1232
+ "epoch": 0.5922629236562821,
1233
+ "grad_norm": 0.11886405199766159,
1234
+ "learning_rate": 8.168054302526898e-07,
1235
+ "loss": 0.6711,
1236
+ "step": 1730
1237
+ },
1238
+ {
1239
+ "epoch": 0.5956864087641218,
1240
+ "grad_norm": 0.12265244871377945,
1241
+ "learning_rate": 8.146779643083074e-07,
1242
+ "loss": 0.6714,
1243
+ "step": 1740
1244
+ },
1245
+ {
1246
+ "epoch": 0.5991098938719617,
1247
+ "grad_norm": 0.1145981103181839,
1248
+ "learning_rate": 8.125410226049487e-07,
1249
+ "loss": 0.6702,
1250
+ "step": 1750
1251
+ },
1252
+ {
1253
+ "epoch": 0.6025333789798014,
1254
+ "grad_norm": 0.11456596106290817,
1255
+ "learning_rate": 8.103946694913985e-07,
1256
+ "loss": 0.6711,
1257
+ "step": 1760
1258
+ },
1259
+ {
1260
+ "epoch": 0.6059568640876413,
1261
+ "grad_norm": 0.1151827871799469,
1262
+ "learning_rate": 8.082389695998427e-07,
1263
+ "loss": 0.6574,
1264
+ "step": 1770
1265
+ },
1266
+ {
1267
+ "epoch": 0.609380349195481,
1268
+ "grad_norm": 0.10862886905670166,
1269
+ "learning_rate": 8.060739878439231e-07,
1270
+ "loss": 0.6574,
1271
+ "step": 1780
1272
+ },
1273
+ {
1274
+ "epoch": 0.6128038343033207,
1275
+ "grad_norm": 0.11756917089223862,
1276
+ "learning_rate": 8.038997894167818e-07,
1277
+ "loss": 0.6586,
1278
+ "step": 1790
1279
+ },
1280
+ {
1281
+ "epoch": 0.6162273194111606,
1282
+ "grad_norm": 0.12379521876573563,
1283
+ "learning_rate": 8.01716439789099e-07,
1284
+ "loss": 0.6609,
1285
+ "step": 1800
1286
+ },
1287
+ {
1288
+ "epoch": 0.6196508045190003,
1289
+ "grad_norm": 0.11942828446626663,
1290
+ "learning_rate": 7.995240047071202e-07,
1291
+ "loss": 0.6623,
1292
+ "step": 1810
1293
+ },
1294
+ {
1295
+ "epoch": 0.6230742896268401,
1296
+ "grad_norm": 0.1148899644613266,
1297
+ "learning_rate": 7.97322550190678e-07,
1298
+ "loss": 0.6625,
1299
+ "step": 1820
1300
+ },
1301
+ {
1302
+ "epoch": 0.6264977747346799,
1303
+ "grad_norm": 0.11534523963928223,
1304
+ "learning_rate": 7.951121425312028e-07,
1305
+ "loss": 0.652,
1306
+ "step": 1830
1307
+ },
1308
+ {
1309
+ "epoch": 0.6299212598425197,
1310
+ "grad_norm": 0.12606197595596313,
1311
+ "learning_rate": 7.92892848289727e-07,
1312
+ "loss": 0.6694,
1313
+ "step": 1840
1314
+ },
1315
+ {
1316
+ "epoch": 0.6333447449503594,
1317
+ "grad_norm": 0.11369583755731583,
1318
+ "learning_rate": 7.90664734294881e-07,
1319
+ "loss": 0.6646,
1320
+ "step": 1850
1321
+ },
1322
+ {
1323
+ "epoch": 0.6367682300581993,
1324
+ "grad_norm": 0.12474564462900162,
1325
+ "learning_rate": 7.884278676408802e-07,
1326
+ "loss": 0.6577,
1327
+ "step": 1860
1328
+ },
1329
+ {
1330
+ "epoch": 0.640191715166039,
1331
+ "grad_norm": 0.11723605543375015,
1332
+ "learning_rate": 7.861823156855055e-07,
1333
+ "loss": 0.6548,
1334
+ "step": 1870
1335
+ },
1336
+ {
1337
+ "epoch": 0.6436152002738789,
1338
+ "grad_norm": 0.10989502817392349,
1339
+ "learning_rate": 7.839281460480739e-07,
1340
+ "loss": 0.6614,
1341
+ "step": 1880
1342
+ },
1343
+ {
1344
+ "epoch": 0.6470386853817186,
1345
+ "grad_norm": 0.11751297116279602,
1346
+ "learning_rate": 7.816654266074032e-07,
1347
+ "loss": 0.6537,
1348
+ "step": 1890
1349
+ },
1350
+ {
1351
+ "epoch": 0.6504621704895583,
1352
+ "grad_norm": 0.11671362817287445,
1353
+ "learning_rate": 7.793942254997677e-07,
1354
+ "loss": 0.6478,
1355
+ "step": 1900
1356
+ },
1357
+ {
1358
+ "epoch": 0.6538856555973982,
1359
+ "grad_norm": 0.13112154603004456,
1360
+ "learning_rate": 7.771146111168459e-07,
1361
+ "loss": 0.6568,
1362
+ "step": 1910
1363
+ },
1364
+ {
1365
+ "epoch": 0.6573091407052379,
1366
+ "grad_norm": 0.11394202709197998,
1367
+ "learning_rate": 7.748266521036624e-07,
1368
+ "loss": 0.6561,
1369
+ "step": 1920
1370
+ },
1371
+ {
1372
+ "epoch": 0.6607326258130777,
1373
+ "grad_norm": 0.12043800204992294,
1374
+ "learning_rate": 7.725304173565191e-07,
1375
+ "loss": 0.6562,
1376
+ "step": 1930
1377
+ },
1378
+ {
1379
+ "epoch": 0.6641561109209175,
1380
+ "grad_norm": 0.11428793519735336,
1381
+ "learning_rate": 7.70225976020922e-07,
1382
+ "loss": 0.6518,
1383
+ "step": 1940
1384
+ },
1385
+ {
1386
+ "epoch": 0.6675795960287573,
1387
+ "grad_norm": 0.11207692325115204,
1388
+ "learning_rate": 7.679133974894982e-07,
1389
+ "loss": 0.6546,
1390
+ "step": 1950
1391
+ },
1392
+ {
1393
+ "epoch": 0.671003081136597,
1394
+ "grad_norm": 0.11501138657331467,
1395
+ "learning_rate": 7.65592751399907e-07,
1396
+ "loss": 0.6423,
1397
+ "step": 1960
1398
+ },
1399
+ {
1400
+ "epoch": 0.6744265662444369,
1401
+ "grad_norm": 0.11726605147123337,
1402
+ "learning_rate": 7.632641076327421e-07,
1403
+ "loss": 0.6549,
1404
+ "step": 1970
1405
+ },
1406
+ {
1407
+ "epoch": 0.6778500513522766,
1408
+ "grad_norm": 0.11668796092271805,
1409
+ "learning_rate": 7.609275363094278e-07,
1410
+ "loss": 0.6502,
1411
+ "step": 1980
1412
+ },
1413
+ {
1414
+ "epoch": 0.6812735364601163,
1415
+ "grad_norm": 0.11436960846185684,
1416
+ "learning_rate": 7.585831077901075e-07,
1417
+ "loss": 0.6447,
1418
+ "step": 1990
1419
+ },
1420
+ {
1421
+ "epoch": 0.6846970215679562,
1422
+ "grad_norm": 0.11399060487747192,
1423
+ "learning_rate": 7.562308926715248e-07,
1424
+ "loss": 0.6478,
1425
+ "step": 2000
1426
+ },
1427
+ {
1428
+ "epoch": 0.6846970215679562,
1429
+ "eval_loss": 0.42816340923309326,
1430
+ "eval_runtime": 2.8925,
1431
+ "eval_samples_per_second": 68.108,
1432
+ "eval_steps_per_second": 2.42,
1433
+ "step": 2000
1434
+ }
1435
+ ],
1436
+ "logging_steps": 10,
1437
+ "max_steps": 5842,
1438
+ "num_input_tokens_seen": 0,
1439
+ "num_train_epochs": 2,
1440
+ "save_steps": 1000,
1441
+ "stateful_callbacks": {
1442
+ "TrainerControl": {
1443
+ "args": {
1444
+ "should_epoch_stop": false,
1445
+ "should_evaluate": false,
1446
+ "should_log": false,
1447
+ "should_save": true,
1448
+ "should_training_stop": false
1449
+ },
1450
+ "attributes": {}
1451
+ }
1452
+ },
1453
+ "total_flos": 2.5887396757191524e+19,
1454
+ "train_batch_size": 8,
1455
+ "trial_name": null,
1456
+ "trial_params": null
1457
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e9544dbadc5ca7d5928d2db0df0d845ead9931fb5f17f62705c5a8e2935eb63
3
+ size 6609
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-2000/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/README.md ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen2.5-Coder-7B-Instruct
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:Qwen/Qwen2.5-Coder-7B-Instruct
7
+ - lora
8
+ - transformers
9
+ ---
10
+
11
+ # Model Card for Model ID
12
+
13
+ <!-- Provide a quick summary of what the model is/does. -->
14
+
15
+
16
+
17
+ ## Model Details
18
+
19
+ ### Model Description
20
+
21
+ <!-- Provide a longer summary of what this model is. -->
22
+
23
+
24
+
25
+ - **Developed by:** [More Information Needed]
26
+ - **Funded by [optional]:** [More Information Needed]
27
+ - **Shared by [optional]:** [More Information Needed]
28
+ - **Model type:** [More Information Needed]
29
+ - **Language(s) (NLP):** [More Information Needed]
30
+ - **License:** [More Information Needed]
31
+ - **Finetuned from model [optional]:** [More Information Needed]
32
+
33
+ ### Model Sources [optional]
34
+
35
+ <!-- Provide the basic links for the model. -->
36
+
37
+ - **Repository:** [More Information Needed]
38
+ - **Paper [optional]:** [More Information Needed]
39
+ - **Demo [optional]:** [More Information Needed]
40
+
41
+ ## Uses
42
+
43
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
44
+
45
+ ### Direct Use
46
+
47
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
48
+
49
+ [More Information Needed]
50
+
51
+ ### Downstream Use [optional]
52
+
53
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
54
+
55
+ [More Information Needed]
56
+
57
+ ### Out-of-Scope Use
58
+
59
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
60
+
61
+ [More Information Needed]
62
+
63
+ ## Bias, Risks, and Limitations
64
+
65
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
66
+
67
+ [More Information Needed]
68
+
69
+ ### Recommendations
70
+
71
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
72
+
73
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
74
+
75
+ ## How to Get Started with the Model
76
+
77
+ Use the code below to get started with the model.
78
+
79
+ [More Information Needed]
80
+
81
+ ## Training Details
82
+
83
+ ### Training Data
84
+
85
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
86
+
87
+ [More Information Needed]
88
+
89
+ ### Training Procedure
90
+
91
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
92
+
93
+ #### Preprocessing [optional]
94
+
95
+ [More Information Needed]
96
+
97
+
98
+ #### Training Hyperparameters
99
+
100
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
101
+
102
+ #### Speeds, Sizes, Times [optional]
103
+
104
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
105
+
106
+ [More Information Needed]
107
+
108
+ ## Evaluation
109
+
110
+ <!-- This section describes the evaluation protocols and provides the results. -->
111
+
112
+ ### Testing Data, Factors & Metrics
113
+
114
+ #### Testing Data
115
+
116
+ <!-- This should link to a Dataset Card if possible. -->
117
+
118
+ [More Information Needed]
119
+
120
+ #### Factors
121
+
122
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
123
+
124
+ [More Information Needed]
125
+
126
+ #### Metrics
127
+
128
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
129
+
130
+ [More Information Needed]
131
+
132
+ ### Results
133
+
134
+ [More Information Needed]
135
+
136
+ #### Summary
137
+
138
+
139
+
140
+ ## Model Examination [optional]
141
+
142
+ <!-- Relevant interpretability work for the model goes here -->
143
+
144
+ [More Information Needed]
145
+
146
+ ## Environmental Impact
147
+
148
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
149
+
150
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
151
+
152
+ - **Hardware Type:** [More Information Needed]
153
+ - **Hours used:** [More Information Needed]
154
+ - **Cloud Provider:** [More Information Needed]
155
+ - **Compute Region:** [More Information Needed]
156
+ - **Carbon Emitted:** [More Information Needed]
157
+
158
+ ## Technical Specifications [optional]
159
+
160
+ ### Model Architecture and Objective
161
+
162
+ [More Information Needed]
163
+
164
+ ### Compute Infrastructure
165
+
166
+ [More Information Needed]
167
+
168
+ #### Hardware
169
+
170
+ [More Information Needed]
171
+
172
+ #### Software
173
+
174
+ [More Information Needed]
175
+
176
+ ## Citation [optional]
177
+
178
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
179
+
180
+ **BibTeX:**
181
+
182
+ [More Information Needed]
183
+
184
+ **APA:**
185
+
186
+ [More Information Needed]
187
+
188
+ ## Glossary [optional]
189
+
190
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
191
+
192
+ [More Information Needed]
193
+
194
+ ## More Information [optional]
195
+
196
+ [More Information Needed]
197
+
198
+ ## Model Card Authors [optional]
199
+
200
+ [More Information Needed]
201
+
202
+ ## Model Card Contact
203
+
204
+ [More Information Needed]
205
+ ### Framework versions
206
+
207
+ - PEFT 0.18.0
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-Coder-7B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.0",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "q_proj",
33
+ "up_proj",
34
+ "down_proj",
35
+ "gate_proj",
36
+ "o_proj",
37
+ "k_proj",
38
+ "v_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d63e5866009b0d88b7085e44bf0ee374e1698a7afc81dd23b067ae938b106255
3
+ size 161533192
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:73806a67d898cd58c9e89f8f164560373ead8776edfc312748d5e2c94494431f
3
+ size 323296891
qwen2.5-coder-7B-tool-if-partial-code_cont-legacy-system-2ep-muts-fnr-reflection-True-patch-instruction-tool-classic-variant-legacy-iim-system_lr_1e-06_bs_8_ga_8_ep_2_pt_llm_muts,synth_muts,partial_synth_muts,git_understanding_as_muts_iim_system/checkpoint-3000/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9201f71267a2e9cc547b58e744aa602817c11758923eca2bb5bab68dc5fd274d
3
+ size 15429