Text Generation
Safetensors
PyTorch
Persian
English
gpt2
persian
instruct
prog-love commited on
Commit
fd78a2a
·
verified ·
1 Parent(s): 2a323bb

Training in progress, step 12000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c8c712d5d70c820407b6c94c687368c68becb6d6d84dc3a2bcdc4617b373e019
3
  size 470900144
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eff0b817ad62fcc65fc9dee0438d878ab8e71018c1b539677298cd06c0087a52
3
  size 470900144
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cc21ebc10c308798dd2c8f0f7afbcff41dd6ebca458838faddce47a52cf87151
3
  size 942082751
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1c3994c87e1c68f3619f1f97be5d888a519060a20b7771be79e62ace5774a670
3
  size 942082751
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:67ccb55173fadea7c0282f3777765b577d8947fd4a9edb2a09d0d2467a98377e
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d244da687768ca9b4a58837838181e8b205496715e3f8cb878b13a489eff3806
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:022915fa8ac65fae0f13853519b8aa7ee79efd632a399c1f64c6b87bab0b057a
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ae029a8e6af9f014c9ef8b713fcd28cef1f184648ddde977ce75efc89ca9a242
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bd0ab4f2ddae6f03ec40afd304de8313743a9194ba7cdd247fd992e5b4fb480a
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9d19e7633d39b5670a0b0a7c3ebfe8f61674a68c2c0192b1e58c8ed35834a040
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 1.0862215282256695,
6
  "eval_steps": 500,
7
- "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -288,6 +288,146 @@
288
  "learning_rate": 2.8380885453267747e-05,
289
  "loss": 2.185189056396484,
290
  "step": 8000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
291
  }
292
  ],
293
  "logging_steps": 200,
@@ -307,7 +447,7 @@
307
  "attributes": {}
308
  }
309
  },
310
- "total_flos": 4.462240400198861e+16,
311
  "train_batch_size": 8,
312
  "trial_name": null,
313
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.6293492650802812,
6
  "eval_steps": 500,
7
+ "global_step": 12000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
288
  "learning_rate": 2.8380885453267747e-05,
289
  "loss": 2.185189056396484,
290
  "step": 8000
291
+ },
292
+ {
293
+ "epoch": 1.1133779150684002,
294
+ "grad_norm": 0.8224326968193054,
295
+ "learning_rate": 2.7537596626844694e-05,
296
+ "loss": 2.187441101074219,
297
+ "step": 8200
298
+ },
299
+ {
300
+ "epoch": 1.1405343019111307,
301
+ "grad_norm": 0.8320621848106384,
302
+ "learning_rate": 2.6694307800421644e-05,
303
+ "loss": 2.1721092224121095,
304
+ "step": 8400
305
+ },
306
+ {
307
+ "epoch": 1.1676906887538614,
308
+ "grad_norm": 0.8610696196556091,
309
+ "learning_rate": 2.5851018973998595e-05,
310
+ "loss": 2.178549957275391,
311
+ "step": 8600
312
+ },
313
+ {
314
+ "epoch": 1.1948470755965919,
315
+ "grad_norm": 0.8490989208221436,
316
+ "learning_rate": 2.5007730147575545e-05,
317
+ "loss": 2.1787890625,
318
+ "step": 8800
319
+ },
320
+ {
321
+ "epoch": 1.2220034624393223,
322
+ "grad_norm": 0.843935489654541,
323
+ "learning_rate": 2.4164441321152495e-05,
324
+ "loss": 2.1710661315917967,
325
+ "step": 9000
326
+ },
327
+ {
328
+ "epoch": 1.249159849282053,
329
+ "grad_norm": 0.831099808216095,
330
+ "learning_rate": 2.3321152494729446e-05,
331
+ "loss": 2.1735246276855467,
332
+ "step": 9200
333
+ },
334
+ {
335
+ "epoch": 1.2763162361247837,
336
+ "grad_norm": 0.8413128852844238,
337
+ "learning_rate": 2.2477863668306396e-05,
338
+ "loss": 2.1660111999511718,
339
+ "step": 9400
340
+ },
341
+ {
342
+ "epoch": 1.3034726229675142,
343
+ "grad_norm": 0.8495444655418396,
344
+ "learning_rate": 2.1634574841883346e-05,
345
+ "loss": 2.169669647216797,
346
+ "step": 9600
347
+ },
348
+ {
349
+ "epoch": 1.3306290098102447,
350
+ "grad_norm": 0.8403661847114563,
351
+ "learning_rate": 2.0791286015460293e-05,
352
+ "loss": 2.1661985778808592,
353
+ "step": 9800
354
+ },
355
+ {
356
+ "epoch": 1.3577853966529754,
357
+ "grad_norm": 0.8322446346282959,
358
+ "learning_rate": 1.9947997189037247e-05,
359
+ "loss": 2.1720236206054686,
360
+ "step": 10000
361
+ },
362
+ {
363
+ "epoch": 1.3849417834957058,
364
+ "grad_norm": 0.8749573826789856,
365
+ "learning_rate": 1.9104708362614197e-05,
366
+ "loss": 2.1759602355957033,
367
+ "step": 10200
368
+ },
369
+ {
370
+ "epoch": 1.4120981703384365,
371
+ "grad_norm": 0.8350988030433655,
372
+ "learning_rate": 1.8261419536191144e-05,
373
+ "loss": 2.1585594177246095,
374
+ "step": 10400
375
+ },
376
+ {
377
+ "epoch": 1.439254557181167,
378
+ "grad_norm": 0.8635809421539307,
379
+ "learning_rate": 1.7418130709768098e-05,
380
+ "loss": 2.160636138916016,
381
+ "step": 10600
382
+ },
383
+ {
384
+ "epoch": 1.4664109440238975,
385
+ "grad_norm": 0.8428000211715698,
386
+ "learning_rate": 1.6574841883345048e-05,
387
+ "loss": 2.155465545654297,
388
+ "step": 10800
389
+ },
390
+ {
391
+ "epoch": 1.4935673308666282,
392
+ "grad_norm": 0.8638054728507996,
393
+ "learning_rate": 1.5731553056921995e-05,
394
+ "loss": 2.168834991455078,
395
+ "step": 11000
396
+ },
397
+ {
398
+ "epoch": 1.5207237177093589,
399
+ "grad_norm": 0.8284506797790527,
400
+ "learning_rate": 1.4888264230498947e-05,
401
+ "loss": 2.1634239196777343,
402
+ "step": 11200
403
+ },
404
+ {
405
+ "epoch": 1.5478801045520894,
406
+ "grad_norm": 0.8089849352836609,
407
+ "learning_rate": 1.4044975404075897e-05,
408
+ "loss": 2.153994903564453,
409
+ "step": 11400
410
+ },
411
+ {
412
+ "epoch": 1.5750364913948198,
413
+ "grad_norm": 0.849999725818634,
414
+ "learning_rate": 1.3201686577652846e-05,
415
+ "loss": 2.160932159423828,
416
+ "step": 11600
417
+ },
418
+ {
419
+ "epoch": 1.6021928782375505,
420
+ "grad_norm": 0.8596501350402832,
421
+ "learning_rate": 1.2358397751229796e-05,
422
+ "loss": 2.1521231079101564,
423
+ "step": 11800
424
+ },
425
+ {
426
+ "epoch": 1.6293492650802812,
427
+ "grad_norm": 0.8179290890693665,
428
+ "learning_rate": 1.1515108924806746e-05,
429
+ "loss": 2.1560073852539063,
430
+ "step": 12000
431
  }
432
  ],
433
  "logging_steps": 200,
 
447
  "attributes": {}
448
  }
449
  },
450
+ "total_flos": 6.693482621357261e+16,
451
  "train_batch_size": 8,
452
  "trial_name": null,
453
  "trial_params": null