rovdetection commited on
Commit
afa6876
·
verified ·
1 Parent(s): d22c1ec

Training in progress, step 1000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c357e9c2944702a41bcf9e0c2ac136e48a71e118595736666655bcc876eb4677
3
  size 4523108832
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6a1a977dbd01ad2f8a9eed9e7f79d95b196931e09186bccc4d8f5f04cbed2d7
3
  size 4523108832
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a2e1912645084bd8899cf84ece2cfd87234cfbcc7c5298a2f7b71be0c49ca68b
3
  size 2911851147
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5aaf5712450f016bf21c43fb383dbe666b12a078f8860c6a53a008f9aa13666b
3
  size 2911851147
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a8e2011629d8bed3ef560fa11175cac55684c4e12a72634bb24abf767b6c7399
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f77569c2e850b04af982cc8c1389f1430851448915c593b69e5da36ce05b71d7
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:14ae2a2128444abab378aa06c09a61a84665f758fcc19fc46f5789b0bc1b5665
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f61cd911648f3cff02d47bc6b9e58cb3242dd8a153313085c99fe548e711bd9c
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1fb5abe0c6c486932d56ee6ec9792e12e43df2b7c9d233a22935f823c7902b1d
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 1.0188,
6
  "eval_steps": 500,
7
- "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -358,6 +358,356 @@
358
  "learning_rate": 0.00019002,
359
  "loss": 2.7336082458496094,
360
  "step": 500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
361
  }
362
  ],
363
  "logging_steps": 10,
@@ -377,7 +727,7 @@
377
  "attributes": {}
378
  }
379
  },
380
- "total_flos": 4.23467821564969e+16,
381
  "train_batch_size": 1,
382
  "trial_name": null,
383
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 3.0064,
6
  "eval_steps": 500,
7
+ "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
358
  "learning_rate": 0.00019002,
359
  "loss": 2.7336082458496094,
360
  "step": 500
361
+ },
362
+ {
363
+ "epoch": 1.0198,
364
+ "grad_norm": 0.6944838762283325,
365
+ "learning_rate": 0.00018982000000000002,
366
+ "loss": 2.67153377532959,
367
+ "step": 510
368
+ },
369
+ {
370
+ "epoch": 1.0208,
371
+ "grad_norm": 0.7149194478988647,
372
+ "learning_rate": 0.00018962000000000002,
373
+ "loss": 2.589804267883301,
374
+ "step": 520
375
+ },
376
+ {
377
+ "epoch": 1.0218,
378
+ "grad_norm": 0.7364429235458374,
379
+ "learning_rate": 0.00018942,
380
+ "loss": 2.5354875564575194,
381
+ "step": 530
382
+ },
383
+ {
384
+ "epoch": 1.0228,
385
+ "grad_norm": 0.5923320651054382,
386
+ "learning_rate": 0.00018922,
387
+ "loss": 2.6230649948120117,
388
+ "step": 540
389
+ },
390
+ {
391
+ "epoch": 1.0238,
392
+ "grad_norm": 0.7289549112319946,
393
+ "learning_rate": 0.00018902000000000003,
394
+ "loss": 2.5119558334350587,
395
+ "step": 550
396
+ },
397
+ {
398
+ "epoch": 1.0248,
399
+ "grad_norm": 0.658437967300415,
400
+ "learning_rate": 0.00018882000000000003,
401
+ "loss": 2.46578369140625,
402
+ "step": 560
403
+ },
404
+ {
405
+ "epoch": 1.0258,
406
+ "grad_norm": 0.6258341073989868,
407
+ "learning_rate": 0.00018862000000000002,
408
+ "loss": 2.585083770751953,
409
+ "step": 570
410
+ },
411
+ {
412
+ "epoch": 1.0268,
413
+ "grad_norm": 0.6197068691253662,
414
+ "learning_rate": 0.00018842000000000002,
415
+ "loss": 2.590087127685547,
416
+ "step": 580
417
+ },
418
+ {
419
+ "epoch": 1.0278,
420
+ "grad_norm": 0.6865118145942688,
421
+ "learning_rate": 0.00018822,
422
+ "loss": 2.4772613525390623,
423
+ "step": 590
424
+ },
425
+ {
426
+ "epoch": 1.0288,
427
+ "grad_norm": 0.6310247778892517,
428
+ "learning_rate": 0.00018802,
429
+ "loss": 2.505804252624512,
430
+ "step": 600
431
+ },
432
+ {
433
+ "epoch": 1.0298,
434
+ "grad_norm": 0.6243994832038879,
435
+ "learning_rate": 0.00018782000000000003,
436
+ "loss": 2.4622268676757812,
437
+ "step": 610
438
+ },
439
+ {
440
+ "epoch": 1.0308,
441
+ "grad_norm": 0.6914836168289185,
442
+ "learning_rate": 0.00018762000000000002,
443
+ "loss": 2.4515810012817383,
444
+ "step": 620
445
+ },
446
+ {
447
+ "epoch": 2.0006,
448
+ "grad_norm": 0.7113664746284485,
449
+ "learning_rate": 0.00018742000000000002,
450
+ "loss": 2.673104667663574,
451
+ "step": 630
452
+ },
453
+ {
454
+ "epoch": 2.0016,
455
+ "grad_norm": 0.660602331161499,
456
+ "learning_rate": 0.00018722,
457
+ "loss": 2.4536123275756836,
458
+ "step": 640
459
+ },
460
+ {
461
+ "epoch": 2.0026,
462
+ "grad_norm": 0.6207917928695679,
463
+ "learning_rate": 0.00018702,
464
+ "loss": 2.409420204162598,
465
+ "step": 650
466
+ },
467
+ {
468
+ "epoch": 2.0036,
469
+ "grad_norm": 0.7043523192405701,
470
+ "learning_rate": 0.00018682000000000003,
471
+ "loss": 2.391322898864746,
472
+ "step": 660
473
+ },
474
+ {
475
+ "epoch": 2.0046,
476
+ "grad_norm": 0.6161942481994629,
477
+ "learning_rate": 0.00018662000000000003,
478
+ "loss": 2.3142091751098635,
479
+ "step": 670
480
+ },
481
+ {
482
+ "epoch": 2.0056,
483
+ "grad_norm": 0.7580428123474121,
484
+ "learning_rate": 0.00018642000000000002,
485
+ "loss": 2.345711326599121,
486
+ "step": 680
487
+ },
488
+ {
489
+ "epoch": 2.0066,
490
+ "grad_norm": 0.7074770927429199,
491
+ "learning_rate": 0.00018622000000000002,
492
+ "loss": 2.318608283996582,
493
+ "step": 690
494
+ },
495
+ {
496
+ "epoch": 2.0076,
497
+ "grad_norm": 0.6267780661582947,
498
+ "learning_rate": 0.00018602,
499
+ "loss": 2.3219219207763673,
500
+ "step": 700
501
+ },
502
+ {
503
+ "epoch": 2.0086,
504
+ "grad_norm": 0.6792202591896057,
505
+ "learning_rate": 0.00018582,
506
+ "loss": 2.2827287673950196,
507
+ "step": 710
508
+ },
509
+ {
510
+ "epoch": 2.0096,
511
+ "grad_norm": 0.671473503112793,
512
+ "learning_rate": 0.00018562000000000003,
513
+ "loss": 2.300338935852051,
514
+ "step": 720
515
+ },
516
+ {
517
+ "epoch": 2.0106,
518
+ "grad_norm": 0.6469615697860718,
519
+ "learning_rate": 0.00018542000000000002,
520
+ "loss": 2.3048261642456054,
521
+ "step": 730
522
+ },
523
+ {
524
+ "epoch": 2.0116,
525
+ "grad_norm": 0.6969014406204224,
526
+ "learning_rate": 0.00018522000000000002,
527
+ "loss": 2.217598533630371,
528
+ "step": 740
529
+ },
530
+ {
531
+ "epoch": 2.0126,
532
+ "grad_norm": 0.641813337802887,
533
+ "learning_rate": 0.00018502000000000001,
534
+ "loss": 2.1945074081420897,
535
+ "step": 750
536
+ },
537
+ {
538
+ "epoch": 2.0136,
539
+ "grad_norm": 0.6202750205993652,
540
+ "learning_rate": 0.00018482,
541
+ "loss": 2.2036405563354493,
542
+ "step": 760
543
+ },
544
+ {
545
+ "epoch": 2.0146,
546
+ "grad_norm": 0.6696969866752625,
547
+ "learning_rate": 0.00018462,
548
+ "loss": 2.199502944946289,
549
+ "step": 770
550
+ },
551
+ {
552
+ "epoch": 2.0156,
553
+ "grad_norm": 0.7106838226318359,
554
+ "learning_rate": 0.00018442000000000003,
555
+ "loss": 2.2403232574462892,
556
+ "step": 780
557
+ },
558
+ {
559
+ "epoch": 2.0166,
560
+ "grad_norm": 0.6160182356834412,
561
+ "learning_rate": 0.00018422000000000002,
562
+ "loss": 2.1474706649780275,
563
+ "step": 790
564
+ },
565
+ {
566
+ "epoch": 2.0176,
567
+ "grad_norm": 0.6780025362968445,
568
+ "learning_rate": 0.00018402000000000002,
569
+ "loss": 2.233156204223633,
570
+ "step": 800
571
+ },
572
+ {
573
+ "epoch": 2.0186,
574
+ "grad_norm": 0.6111968755722046,
575
+ "learning_rate": 0.00018382,
576
+ "loss": 2.2461593627929686,
577
+ "step": 810
578
+ },
579
+ {
580
+ "epoch": 2.0196,
581
+ "grad_norm": 0.7200072407722473,
582
+ "learning_rate": 0.00018362,
583
+ "loss": 2.1219438552856444,
584
+ "step": 820
585
+ },
586
+ {
587
+ "epoch": 2.0206,
588
+ "grad_norm": 0.6476977467536926,
589
+ "learning_rate": 0.00018342,
590
+ "loss": 2.0821203231811523,
591
+ "step": 830
592
+ },
593
+ {
594
+ "epoch": 2.0216,
595
+ "grad_norm": 0.6472647786140442,
596
+ "learning_rate": 0.00018322000000000002,
597
+ "loss": 2.0447275161743166,
598
+ "step": 840
599
+ },
600
+ {
601
+ "epoch": 2.0226,
602
+ "grad_norm": 0.7900026440620422,
603
+ "learning_rate": 0.00018302000000000002,
604
+ "loss": 2.087274932861328,
605
+ "step": 850
606
+ },
607
+ {
608
+ "epoch": 2.0236,
609
+ "grad_norm": 0.6523037552833557,
610
+ "learning_rate": 0.00018282000000000001,
611
+ "loss": 2.0110971450805666,
612
+ "step": 860
613
+ },
614
+ {
615
+ "epoch": 2.0246,
616
+ "grad_norm": 0.5921297073364258,
617
+ "learning_rate": 0.00018262,
618
+ "loss": 1.981992530822754,
619
+ "step": 870
620
+ },
621
+ {
622
+ "epoch": 2.0256,
623
+ "grad_norm": 0.7393470406532288,
624
+ "learning_rate": 0.00018242,
625
+ "loss": 2.044318962097168,
626
+ "step": 880
627
+ },
628
+ {
629
+ "epoch": 2.0266,
630
+ "grad_norm": 0.697785496711731,
631
+ "learning_rate": 0.00018222,
632
+ "loss": 2.05950870513916,
633
+ "step": 890
634
+ },
635
+ {
636
+ "epoch": 2.0276,
637
+ "grad_norm": 0.6266019940376282,
638
+ "learning_rate": 0.00018202000000000002,
639
+ "loss": 1.9614368438720704,
640
+ "step": 900
641
+ },
642
+ {
643
+ "epoch": 2.0286,
644
+ "grad_norm": 0.6764276623725891,
645
+ "learning_rate": 0.00018182000000000002,
646
+ "loss": 1.9438285827636719,
647
+ "step": 910
648
+ },
649
+ {
650
+ "epoch": 2.0296,
651
+ "grad_norm": 0.7023544907569885,
652
+ "learning_rate": 0.00018162,
653
+ "loss": 2.02318115234375,
654
+ "step": 920
655
+ },
656
+ {
657
+ "epoch": 2.0306,
658
+ "grad_norm": 0.8344048857688904,
659
+ "learning_rate": 0.00018142,
660
+ "loss": 1.822168731689453,
661
+ "step": 930
662
+ },
663
+ {
664
+ "epoch": 3.0004,
665
+ "grad_norm": 0.7352348566055298,
666
+ "learning_rate": 0.00018122,
667
+ "loss": 2.1745040893554686,
668
+ "step": 940
669
+ },
670
+ {
671
+ "epoch": 3.0014,
672
+ "grad_norm": 0.7016635537147522,
673
+ "learning_rate": 0.00018102000000000003,
674
+ "loss": 1.9799747467041016,
675
+ "step": 950
676
+ },
677
+ {
678
+ "epoch": 3.0024,
679
+ "grad_norm": 0.7483706474304199,
680
+ "learning_rate": 0.00018082000000000002,
681
+ "loss": 1.9175453186035156,
682
+ "step": 960
683
+ },
684
+ {
685
+ "epoch": 3.0034,
686
+ "grad_norm": 0.776687502861023,
687
+ "learning_rate": 0.00018062000000000002,
688
+ "loss": 1.8681112289428712,
689
+ "step": 970
690
+ },
691
+ {
692
+ "epoch": 3.0044,
693
+ "grad_norm": 0.7005246877670288,
694
+ "learning_rate": 0.00018042,
695
+ "loss": 1.8719995498657227,
696
+ "step": 980
697
+ },
698
+ {
699
+ "epoch": 3.0054,
700
+ "grad_norm": 0.729756772518158,
701
+ "learning_rate": 0.00018022,
702
+ "loss": 1.8054256439208984,
703
+ "step": 990
704
+ },
705
+ {
706
+ "epoch": 3.0064,
707
+ "grad_norm": 0.7293346524238586,
708
+ "learning_rate": 0.00018002,
709
+ "loss": 1.862677001953125,
710
+ "step": 1000
711
  }
712
  ],
713
  "logging_steps": 10,
 
727
  "attributes": {}
728
  }
729
  },
730
+ "total_flos": 8.465719089364992e+16,
731
  "train_batch_size": 1,
732
  "trial_name": null,
733
  "trial_params": null