nttx commited on
Commit
3617519
·
verified ·
1 Parent(s): af20630

Training in progress, step 5000, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2b8b2e9b70bd50a6107cf935c4ee7553fe4d165acbb631aebb19fdd2952bd8c6
3
  size 335604696
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7ffcd8b545710d2deb1452b4a64b7bc1dad2673f31e03dcb3b1797de89cc7950
3
  size 335604696
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:49089f5da9e62f3703191358d317a32f5e301aae4b5d381f3dd7b7771cd0e4e5
3
  size 671472978
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:331d061a629d7e43ed6f87056cf8b7a5c9d87ad133577087062ab4d83f230875
3
  size 671472978
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e8802196682461502be670175c89363a21ec3127850a663775a099368fb8766e
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e7648dde45e1b2b06d70c9f35ad9adedc1a2d1a3b142ab4448f8c383a8e1910e
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7f6d93467eee9f7eb66e857b295a064a9a7b557fc916f9b5131020a127e4f67a
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:016e25fda59c3776a60450c0d95b5f609ee100d53eaff927c1f903b335520ed3
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": 1.4104703664779663,
3
  "best_model_checkpoint": "miner_id_24/checkpoint-4500",
4
- "epoch": 1.6285171446666062,
5
  "eval_steps": 500,
6
- "global_step": 4500,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -717,6 +717,84 @@
717
  "eval_samples_per_second": 21.162,
718
  "eval_steps_per_second": 10.581,
719
  "step": 4500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
720
  }
721
  ],
722
  "logging_steps": 50,
@@ -731,7 +809,7 @@
731
  "early_stopping_threshold": 0.0
732
  },
733
  "attributes": {
734
- "early_stopping_patience_counter": 0
735
  }
736
  },
737
  "TrainerControl": {
@@ -740,12 +818,12 @@
740
  "should_evaluate": false,
741
  "should_log": false,
742
  "should_save": true,
743
- "should_training_stop": false
744
  },
745
  "attributes": {}
746
  }
747
  },
748
- "total_flos": 8.248564640812892e+17,
749
  "train_batch_size": 2,
750
  "trial_name": null,
751
  "trial_params": null
 
1
  {
2
  "best_metric": 1.4104703664779663,
3
  "best_model_checkpoint": "miner_id_24/checkpoint-4500",
4
+ "epoch": 1.809463494074007,
5
  "eval_steps": 500,
6
+ "global_step": 5000,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
717
  "eval_samples_per_second": 21.162,
718
  "eval_steps_per_second": 10.581,
719
  "step": 4500
720
+ },
721
+ {
722
+ "epoch": 1.6466117796073463,
723
+ "grad_norm": 11.539615631103516,
724
+ "learning_rate": 4.050702638550275e-06,
725
+ "loss": 3.8056,
726
+ "step": 4550
727
+ },
728
+ {
729
+ "epoch": 1.6647064145480865,
730
+ "grad_norm": 14.798728942871094,
731
+ "learning_rate": 3.2051298603643753e-06,
732
+ "loss": 3.6837,
733
+ "step": 4600
734
+ },
735
+ {
736
+ "epoch": 1.6828010494888266,
737
+ "grad_norm": 14.411064147949219,
738
+ "learning_rate": 2.4570213114592954e-06,
739
+ "loss": 3.7863,
740
+ "step": 4650
741
+ },
742
+ {
743
+ "epoch": 1.7008956844295666,
744
+ "grad_norm": 13.782867431640625,
745
+ "learning_rate": 1.8071302737293295e-06,
746
+ "loss": 3.6069,
747
+ "step": 4700
748
+ },
749
+ {
750
+ "epoch": 1.7189903193703067,
751
+ "grad_norm": 24.582422256469727,
752
+ "learning_rate": 1.2561111323605712e-06,
753
+ "loss": 3.6882,
754
+ "step": 4750
755
+ },
756
+ {
757
+ "epoch": 1.737084954311047,
758
+ "grad_norm": 20.012916564941406,
759
+ "learning_rate": 8.04518716920466e-07,
760
+ "loss": 3.8095,
761
+ "step": 4800
762
+ },
763
+ {
764
+ "epoch": 1.755179589251787,
765
+ "grad_norm": 18.923974990844727,
766
+ "learning_rate": 4.5280774269154115e-07,
767
+ "loss": 3.7323,
768
+ "step": 4850
769
+ },
770
+ {
771
+ "epoch": 1.773274224192527,
772
+ "grad_norm": 17.673463821411133,
773
+ "learning_rate": 2.0133235281156736e-07,
774
+ "loss": 3.7895,
775
+ "step": 4900
776
+ },
777
+ {
778
+ "epoch": 1.791368859133267,
779
+ "grad_norm": 12.837079048156738,
780
+ "learning_rate": 5.0345761681491746e-08,
781
+ "loss": 3.683,
782
+ "step": 4950
783
+ },
784
+ {
785
+ "epoch": 1.809463494074007,
786
+ "grad_norm": 15.028837203979492,
787
+ "learning_rate": 0.0,
788
+ "loss": 3.6422,
789
+ "step": 5000
790
+ },
791
+ {
792
+ "epoch": 1.809463494074007,
793
+ "eval_loss": 1.4128326177597046,
794
+ "eval_runtime": 55.1436,
795
+ "eval_samples_per_second": 21.109,
796
+ "eval_steps_per_second": 10.554,
797
+ "step": 5000
798
  }
799
  ],
800
  "logging_steps": 50,
 
809
  "early_stopping_threshold": 0.0
810
  },
811
  "attributes": {
812
+ "early_stopping_patience_counter": 1
813
  }
814
  },
815
  "TrainerControl": {
 
818
  "should_evaluate": false,
819
  "should_log": false,
820
  "should_save": true,
821
+ "should_training_stop": true
822
  },
823
  "attributes": {}
824
  }
825
  },
826
+ "total_flos": 9.167552121761956e+17,
827
  "train_batch_size": 2,
828
  "trial_name": null,
829
  "trial_params": null