Instructions to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.4 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.4 with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.4", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Unsloth Desktop
Training in progress, step 500, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 194563400
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c9ef2e5ad911591225c49c0e3d5bed60b25b307b1854c4d369cf00e663301f3a
|
| 3 |
size 194563400
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 100256339
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9832ebaf36cdf6d40d6deb7aae74ee6b568c1f2520824054bd7e025e86df3a78
|
| 3 |
size 100256339
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14645
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8dfdd239cda5e70b7f9cfbafcf82b4a87687ec4adfc28e9a013f1d080f24df00
|
| 3 |
size 14645
|
last-checkpoint/scaler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1383
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:407d570969b2eef421afd35f3175846d76cc3ff924cec8e15dfd49f2a4410212
|
| 3 |
size 1383
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f8a901bbf8d17ac8e841a67b5eb485f25ceae4fe07151d782bb24b1b7d3a7e5a
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 7.
|
| 6 |
"eval_steps": 500,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -10788,11 +10788,231 @@
|
|
| 10788 |
"rewards/compute_weighted_reward/mean": 0.9329167008399963,
|
| 10789 |
"rewards/compute_weighted_reward/std": 1.4120367765426636,
|
| 10790 |
"step": 490
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10791 |
}
|
| 10792 |
],
|
| 10793 |
"logging_steps": 1,
|
| 10794 |
"max_steps": 500,
|
| 10795 |
-
"num_input_tokens_seen":
|
| 10796 |
"num_train_epochs": 8,
|
| 10797 |
"save_steps": 10,
|
| 10798 |
"stateful_callbacks": {
|
|
@@ -10802,7 +11022,7 @@
|
|
| 10802 |
"should_evaluate": false,
|
| 10803 |
"should_log": false,
|
| 10804 |
"should_save": true,
|
| 10805 |
-
"should_training_stop":
|
| 10806 |
},
|
| 10807 |
"attributes": {}
|
| 10808 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 7.246376811594203,
|
| 6 |
"eval_steps": 500,
|
| 7 |
+
"global_step": 500,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 10788 |
"rewards/compute_weighted_reward/mean": 0.9329167008399963,
|
| 10789 |
"rewards/compute_weighted_reward/std": 1.4120367765426636,
|
| 10790 |
"step": 490
|
| 10791 |
+
},
|
| 10792 |
+
{
|
| 10793 |
+
"completion_length": 750.7916870117188,
|
| 10794 |
+
"completions/clipped_ratio": 0.20833333333333337,
|
| 10795 |
+
"completions/max_length": 1400.0,
|
| 10796 |
+
"completions/max_terminated_length": 1372.0,
|
| 10797 |
+
"completions/mean_length": 750.7916870117188,
|
| 10798 |
+
"completions/mean_terminated_length": 579.9473876953125,
|
| 10799 |
+
"completions/min_length": 181.0,
|
| 10800 |
+
"completions/min_terminated_length": 181.0,
|
| 10801 |
+
"epoch": 7.115942028985507,
|
| 10802 |
+
"frac_reward_zero_std": 0.0,
|
| 10803 |
+
"grad_norm": 1.0721700191497803,
|
| 10804 |
+
"kl": 0.0014365280221682042,
|
| 10805 |
+
"learning_rate": 3.50714075049563e-10,
|
| 10806 |
+
"loss": 0.2133,
|
| 10807 |
+
"num_tokens": 12727787.0,
|
| 10808 |
+
"reward": 0.862083375453949,
|
| 10809 |
+
"reward_std": 1.5412702560424805,
|
| 10810 |
+
"rewards/compute_weighted_reward/mean": 0.8620834350585938,
|
| 10811 |
+
"rewards/compute_weighted_reward/std": 1.502502202987671,
|
| 10812 |
+
"step": 491
|
| 10813 |
+
},
|
| 10814 |
+
{
|
| 10815 |
+
"completion_length": 891.0416717529297,
|
| 10816 |
+
"completions/clipped_ratio": 0.375,
|
| 10817 |
+
"completions/max_length": 1400.0,
|
| 10818 |
+
"completions/max_terminated_length": 1347.0,
|
| 10819 |
+
"completions/mean_length": 891.0416870117188,
|
| 10820 |
+
"completions/mean_terminated_length": 585.6666870117188,
|
| 10821 |
+
"completions/min_length": 247.0,
|
| 10822 |
+
"completions/min_terminated_length": 247.0,
|
| 10823 |
+
"epoch": 7.130434782608695,
|
| 10824 |
+
"frac_reward_zero_std": 0.25,
|
| 10825 |
+
"grad_norm": 0.9120261073112488,
|
| 10826 |
+
"kl": 0.0014478433004114777,
|
| 10827 |
+
"learning_rate": 2.9472477730797527e-10,
|
| 10828 |
+
"loss": 0.0696,
|
| 10829 |
+
"num_tokens": 12756516.0,
|
| 10830 |
+
"reward": 0.41958338022232056,
|
| 10831 |
+
"reward_std": 1.0971274375915527,
|
| 10832 |
+
"rewards/compute_weighted_reward/mean": 0.41958341002464294,
|
| 10833 |
+
"rewards/compute_weighted_reward/std": 1.6684163808822632,
|
| 10834 |
+
"step": 492
|
| 10835 |
+
},
|
| 10836 |
+
{
|
| 10837 |
+
"completion_length": 740.6666870117188,
|
| 10838 |
+
"completions/clipped_ratio": 0.25,
|
| 10839 |
+
"completions/max_length": 1400.0,
|
| 10840 |
+
"completions/max_terminated_length": 1211.0,
|
| 10841 |
+
"completions/mean_length": 741.5,
|
| 10842 |
+
"completions/mean_terminated_length": 522.0,
|
| 10843 |
+
"completions/min_length": 271.0,
|
| 10844 |
+
"completions/min_terminated_length": 271.0,
|
| 10845 |
+
"epoch": 7.144927536231884,
|
| 10846 |
+
"frac_reward_zero_std": 0.25,
|
| 10847 |
+
"grad_norm": 0.7683882117271423,
|
| 10848 |
+
"kl": 0.0011400138610042632,
|
| 10849 |
+
"learning_rate": 2.4359497401758025e-10,
|
| 10850 |
+
"loss": 0.1188,
|
| 10851 |
+
"num_tokens": 12782406.0,
|
| 10852 |
+
"reward": 0.7000000476837158,
|
| 10853 |
+
"reward_std": 1.0930380821228027,
|
| 10854 |
+
"rewards/compute_weighted_reward/mean": 0.7000000476837158,
|
| 10855 |
+
"rewards/compute_weighted_reward/std": 1.4413580894470215,
|
| 10856 |
+
"step": 493
|
| 10857 |
+
},
|
| 10858 |
+
{
|
| 10859 |
+
"completion_length": 850.1250152587891,
|
| 10860 |
+
"completions/clipped_ratio": 0.29166666666666663,
|
| 10861 |
+
"completions/max_length": 1400.0,
|
| 10862 |
+
"completions/max_terminated_length": 1153.0,
|
| 10863 |
+
"completions/mean_length": 850.125,
|
| 10864 |
+
"completions/mean_terminated_length": 623.7058715820312,
|
| 10865 |
+
"completions/min_length": 225.0,
|
| 10866 |
+
"completions/min_terminated_length": 225.0,
|
| 10867 |
+
"epoch": 7.159420289855072,
|
| 10868 |
+
"frac_reward_zero_std": 0.0,
|
| 10869 |
+
"grad_norm": 1.1599197387695312,
|
| 10870 |
+
"kl": 0.001941966766025871,
|
| 10871 |
+
"learning_rate": 1.973271571728441e-10,
|
| 10872 |
+
"loss": 0.2045,
|
| 10873 |
+
"num_tokens": 12810153.0,
|
| 10874 |
+
"reward": 0.7408332824707031,
|
| 10875 |
+
"reward_std": 1.43536376953125,
|
| 10876 |
+
"rewards/compute_weighted_reward/mean": 0.7408332824707031,
|
| 10877 |
+
"rewards/compute_weighted_reward/std": 1.4577913284301758,
|
| 10878 |
+
"step": 494
|
| 10879 |
+
},
|
| 10880 |
+
{
|
| 10881 |
+
"completion_length": 1021.2917175292969,
|
| 10882 |
+
"completions/clipped_ratio": 0.33333333333333337,
|
| 10883 |
+
"completions/max_length": 1400.0,
|
| 10884 |
+
"completions/max_terminated_length": 1260.0,
|
| 10885 |
+
"completions/mean_length": 1026.125,
|
| 10886 |
+
"completions/mean_terminated_length": 839.1875,
|
| 10887 |
+
"completions/min_length": 356.0,
|
| 10888 |
+
"completions/min_terminated_length": 356.0,
|
| 10889 |
+
"epoch": 7.173913043478261,
|
| 10890 |
+
"frac_reward_zero_std": 0.25,
|
| 10891 |
+
"grad_norm": 0.7053532600402832,
|
| 10892 |
+
"kl": 0.0008829567377688363,
|
| 10893 |
+
"learning_rate": 1.559235818018978e-10,
|
| 10894 |
+
"loss": 0.1238,
|
| 10895 |
+
"num_tokens": 12843456.0,
|
| 10896 |
+
"reward": 0.5083333849906921,
|
| 10897 |
+
"reward_std": 1.2783448696136475,
|
| 10898 |
+
"rewards/compute_weighted_reward/mean": 0.5083333849906921,
|
| 10899 |
+
"rewards/compute_weighted_reward/std": 1.5852407217025757,
|
| 10900 |
+
"step": 495
|
| 10901 |
+
},
|
| 10902 |
+
{
|
| 10903 |
+
"completion_length": 727.4583587646484,
|
| 10904 |
+
"completions/clipped_ratio": 0.04166666666666663,
|
| 10905 |
+
"completions/max_length": 1400.0,
|
| 10906 |
+
"completions/max_terminated_length": 1203.0,
|
| 10907 |
+
"completions/mean_length": 727.4583740234375,
|
| 10908 |
+
"completions/mean_terminated_length": 698.2174072265625,
|
| 10909 |
+
"completions/min_length": 388.0,
|
| 10910 |
+
"completions/min_terminated_length": 388.0,
|
| 10911 |
+
"epoch": 7.188405797101449,
|
| 10912 |
+
"frac_reward_zero_std": 0.25,
|
| 10913 |
+
"grad_norm": 0.5732879638671875,
|
| 10914 |
+
"kl": 0.0010317230480723083,
|
| 10915 |
+
"learning_rate": 1.193862658566025e-10,
|
| 10916 |
+
"loss": 0.0601,
|
| 10917 |
+
"num_tokens": 12868703.0,
|
| 10918 |
+
"reward": 1.3066667318344116,
|
| 10919 |
+
"reward_std": 0.6335288882255554,
|
| 10920 |
+
"rewards/compute_weighted_reward/mean": 1.3066667318344116,
|
| 10921 |
+
"rewards/compute_weighted_reward/std": 0.8279028534889221,
|
| 10922 |
+
"step": 496
|
| 10923 |
+
},
|
| 10924 |
+
{
|
| 10925 |
+
"completion_length": 589.8333511352539,
|
| 10926 |
+
"completions/clipped_ratio": 0.125,
|
| 10927 |
+
"completions/max_length": 1400.0,
|
| 10928 |
+
"completions/max_terminated_length": 970.0,
|
| 10929 |
+
"completions/mean_length": 589.8333740234375,
|
| 10930 |
+
"completions/mean_terminated_length": 474.0952453613281,
|
| 10931 |
+
"completions/min_length": 225.0,
|
| 10932 |
+
"completions/min_terminated_length": 225.0,
|
| 10933 |
+
"epoch": 7.202898550724638,
|
| 10934 |
+
"frac_reward_zero_std": 0.25,
|
| 10935 |
+
"grad_norm": 0.5461244583129883,
|
| 10936 |
+
"kl": 0.0010659504623617977,
|
| 10937 |
+
"learning_rate": 8.771699011416167e-11,
|
| 10938 |
+
"loss": 0.0441,
|
| 10939 |
+
"num_tokens": 12890785.0,
|
| 10940 |
+
"reward": 1.7158334255218506,
|
| 10941 |
+
"reward_std": 0.7497227191925049,
|
| 10942 |
+
"rewards/compute_weighted_reward/mean": 1.715833306312561,
|
| 10943 |
+
"rewards/compute_weighted_reward/std": 1.0876857042312622,
|
| 10944 |
+
"step": 497
|
| 10945 |
+
},
|
| 10946 |
+
{
|
| 10947 |
+
"completion_length": 707.1250305175781,
|
| 10948 |
+
"completions/clipped_ratio": 0.16666666666666663,
|
| 10949 |
+
"completions/max_length": 1400.0,
|
| 10950 |
+
"completions/max_terminated_length": 1097.0,
|
| 10951 |
+
"completions/mean_length": 707.125,
|
| 10952 |
+
"completions/mean_terminated_length": 568.5499877929688,
|
| 10953 |
+
"completions/min_length": 247.0,
|
| 10954 |
+
"completions/min_terminated_length": 247.0,
|
| 10955 |
+
"epoch": 7.217391304347826,
|
| 10956 |
+
"frac_reward_zero_std": 0.25,
|
| 10957 |
+
"grad_norm": 0.5935053825378418,
|
| 10958 |
+
"kl": 0.0011391782754799351,
|
| 10959 |
+
"learning_rate": 6.091729809042379e-11,
|
| 10960 |
+
"loss": 0.1227,
|
| 10961 |
+
"num_tokens": 12915424.0,
|
| 10962 |
+
"reward": 1.2154167890548706,
|
| 10963 |
+
"reward_std": 0.981461226940155,
|
| 10964 |
+
"rewards/compute_weighted_reward/mean": 1.2154165506362915,
|
| 10965 |
+
"rewards/compute_weighted_reward/std": 1.4390183687210083,
|
| 10966 |
+
"step": 498
|
| 10967 |
+
},
|
| 10968 |
+
{
|
| 10969 |
+
"completion_length": 871.5833587646484,
|
| 10970 |
+
"completions/clipped_ratio": 0.41666666666666663,
|
| 10971 |
+
"completions/max_length": 1400.0,
|
| 10972 |
+
"completions/max_terminated_length": 1264.0,
|
| 10973 |
+
"completions/mean_length": 871.5833740234375,
|
| 10974 |
+
"completions/mean_terminated_length": 494.14288330078125,
|
| 10975 |
+
"completions/min_length": 258.0,
|
| 10976 |
+
"completions/min_terminated_length": 258.0,
|
| 10977 |
+
"epoch": 7.231884057971015,
|
| 10978 |
+
"frac_reward_zero_std": 0.0,
|
| 10979 |
+
"grad_norm": 0.7544428706169128,
|
| 10980 |
+
"kl": 0.0014083839778322726,
|
| 10981 |
+
"learning_rate": 3.898849596456477e-11,
|
| 10982 |
+
"loss": 0.118,
|
| 10983 |
+
"num_tokens": 12944154.0,
|
| 10984 |
+
"reward": 0.16750004887580872,
|
| 10985 |
+
"reward_std": 1.1228570938110352,
|
| 10986 |
+
"rewards/compute_weighted_reward/mean": 0.16750003397464752,
|
| 10987 |
+
"rewards/compute_weighted_reward/std": 1.6420857906341553,
|
| 10988 |
+
"step": 499
|
| 10989 |
+
},
|
| 10990 |
+
{
|
| 10991 |
+
"completion_length": 507.12501525878906,
|
| 10992 |
+
"completions/clipped_ratio": 0.04166666666666663,
|
| 10993 |
+
"completions/max_length": 1400.0,
|
| 10994 |
+
"completions/max_terminated_length": 820.0,
|
| 10995 |
+
"completions/mean_length": 507.125,
|
| 10996 |
+
"completions/mean_terminated_length": 468.3043518066406,
|
| 10997 |
+
"completions/min_length": 216.0,
|
| 10998 |
+
"completions/min_terminated_length": 216.0,
|
| 10999 |
+
"epoch": 7.246376811594203,
|
| 11000 |
+
"frac_reward_zero_std": 0.75,
|
| 11001 |
+
"grad_norm": 0.35263198614120483,
|
| 11002 |
+
"kl": 0.0007562682876596227,
|
| 11003 |
+
"learning_rate": 2.193165251545004e-11,
|
| 11004 |
+
"loss": 0.0424,
|
| 11005 |
+
"num_tokens": 12964083.0,
|
| 11006 |
+
"reward": 1.8279168605804443,
|
| 11007 |
+
"reward_std": 0.33918437361717224,
|
| 11008 |
+
"rewards/compute_weighted_reward/mean": 1.8279167413711548,
|
| 11009 |
+
"rewards/compute_weighted_reward/std": 1.045437216758728,
|
| 11010 |
+
"step": 500
|
| 11011 |
}
|
| 11012 |
],
|
| 11013 |
"logging_steps": 1,
|
| 11014 |
"max_steps": 500,
|
| 11015 |
+
"num_input_tokens_seen": 12964083,
|
| 11016 |
"num_train_epochs": 8,
|
| 11017 |
"save_steps": 10,
|
| 11018 |
"stateful_callbacks": {
|
|
|
|
| 11022 |
"should_evaluate": false,
|
| 11023 |
"should_log": false,
|
| 11024 |
"should_save": true,
|
| 11025 |
+
"should_training_stop": true
|
| 11026 |
},
|
| 11027 |
"attributes": {}
|
| 11028 |
}
|