Instructions to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.5 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.5 with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.5", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Unsloth Desktop
Training in progress, step 500, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 194563400
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:40e781c129f82797c0a6821bcdc7e7c877cdc491a2e369fbbdc5f4d8f8555217
|
| 3 |
size 194563400
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 100256339
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6804192c71eec3b4a512a29b5211006e9dc557c1b052e87f96ec0a401d77f542
|
| 3 |
size 100256339
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14645
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2d61dbc3f2c2b23c059c6b8721c9652c292e77f8e6bca5c2f6d12181ba94bfe4
|
| 3 |
size 14645
|
last-checkpoint/scaler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1383
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:43f429e2475c017e0929ea1a41fab783edaaa7552eb6003f1b53b0de645a5e96
|
| 3 |
size 1383
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9d0ddb482cfa0bc26b42f43c3031f7f23c3b5c7fc0c791438490253e4dc5338f
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 15.
|
| 6 |
"eval_steps": 500,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -10788,11 +10788,231 @@
|
|
| 10788 |
"rewards/compute_weighted_reward/mean": 1.4883333444595337,
|
| 10789 |
"rewards/compute_weighted_reward/std": 1.16587233543396,
|
| 10790 |
"step": 490
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10791 |
}
|
| 10792 |
],
|
| 10793 |
"logging_steps": 1,
|
| 10794 |
"max_steps": 500,
|
| 10795 |
-
"num_input_tokens_seen":
|
| 10796 |
"num_train_epochs": 16,
|
| 10797 |
"save_steps": 10,
|
| 10798 |
"stateful_callbacks": {
|
|
@@ -10802,7 +11022,7 @@
|
|
| 10802 |
"should_evaluate": false,
|
| 10803 |
"should_log": false,
|
| 10804 |
"should_save": true,
|
| 10805 |
-
"should_training_stop":
|
| 10806 |
},
|
| 10807 |
"attributes": {}
|
| 10808 |
}
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 15.625,
|
| 6 |
"eval_steps": 500,
|
| 7 |
+
"global_step": 500,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 10788 |
"rewards/compute_weighted_reward/mean": 1.4883333444595337,
|
| 10789 |
"rewards/compute_weighted_reward/std": 1.16587233543396,
|
| 10790 |
"step": 490
|
| 10791 |
+
},
|
| 10792 |
+
{
|
| 10793 |
+
"completion_length": 583.5833587646484,
|
| 10794 |
+
"completions/clipped_ratio": 0.04166666666666663,
|
| 10795 |
+
"completions/max_length": 1400.0,
|
| 10796 |
+
"completions/max_terminated_length": 1331.0,
|
| 10797 |
+
"completions/mean_length": 583.5833740234375,
|
| 10798 |
+
"completions/mean_terminated_length": 548.0869750976562,
|
| 10799 |
+
"completions/min_length": 207.0,
|
| 10800 |
+
"completions/min_terminated_length": 207.0,
|
| 10801 |
+
"epoch": 15.34375,
|
| 10802 |
+
"frac_reward_zero_std": 0.25,
|
| 10803 |
+
"grad_norm": 0.647857666015625,
|
| 10804 |
+
"kl": 0.02400309219956398,
|
| 10805 |
+
"learning_rate": 1.0289003460074165e-09,
|
| 10806 |
+
"loss": 0.0489,
|
| 10807 |
+
"num_tokens": 12713527.0,
|
| 10808 |
+
"reward": 1.3562501668930054,
|
| 10809 |
+
"reward_std": 0.5569199323654175,
|
| 10810 |
+
"rewards/compute_weighted_reward/mean": 1.3562501668930054,
|
| 10811 |
+
"rewards/compute_weighted_reward/std": 0.889302670955658,
|
| 10812 |
+
"step": 491
|
| 10813 |
+
},
|
| 10814 |
+
{
|
| 10815 |
+
"completion_length": 728.3750076293945,
|
| 10816 |
+
"completions/clipped_ratio": 0.20833333333333337,
|
| 10817 |
+
"completions/max_length": 1400.0,
|
| 10818 |
+
"completions/max_terminated_length": 1028.0,
|
| 10819 |
+
"completions/mean_length": 731.0416870117188,
|
| 10820 |
+
"completions/mean_terminated_length": 555.0,
|
| 10821 |
+
"completions/min_length": 282.0,
|
| 10822 |
+
"completions/min_terminated_length": 282.0,
|
| 10823 |
+
"epoch": 15.375,
|
| 10824 |
+
"frac_reward_zero_std": 0.25,
|
| 10825 |
+
"grad_norm": 0.8000059723854065,
|
| 10826 |
+
"kl": 0.023584565613418818,
|
| 10827 |
+
"learning_rate": 8.767851876239074e-10,
|
| 10828 |
+
"loss": 0.0869,
|
| 10829 |
+
"num_tokens": 12738974.0,
|
| 10830 |
+
"reward": 1.4620835781097412,
|
| 10831 |
+
"reward_std": 0.9179548025131226,
|
| 10832 |
+
"rewards/compute_weighted_reward/mean": 1.4620834589004517,
|
| 10833 |
+
"rewards/compute_weighted_reward/std": 1.6754546165466309,
|
| 10834 |
+
"step": 492
|
| 10835 |
+
},
|
| 10836 |
+
{
|
| 10837 |
+
"completion_length": 594.0000305175781,
|
| 10838 |
+
"completions/clipped_ratio": 0.20833333333333337,
|
| 10839 |
+
"completions/max_length": 1400.0,
|
| 10840 |
+
"completions/max_terminated_length": 733.0,
|
| 10841 |
+
"completions/mean_length": 594.0,
|
| 10842 |
+
"completions/mean_terminated_length": 381.8947448730469,
|
| 10843 |
+
"completions/min_length": 158.0,
|
| 10844 |
+
"completions/min_terminated_length": 158.0,
|
| 10845 |
+
"epoch": 15.40625,
|
| 10846 |
+
"frac_reward_zero_std": 0.0,
|
| 10847 |
+
"grad_norm": 0.656679630279541,
|
| 10848 |
+
"kl": 0.019565630238503218,
|
| 10849 |
+
"learning_rate": 7.368119432699382e-10,
|
| 10850 |
+
"loss": 0.1084,
|
| 10851 |
+
"num_tokens": 12760364.0,
|
| 10852 |
+
"reward": 0.9337500333786011,
|
| 10853 |
+
"reward_std": 1.1254143714904785,
|
| 10854 |
+
"rewards/compute_weighted_reward/mean": 0.9337499737739563,
|
| 10855 |
+
"rewards/compute_weighted_reward/std": 1.215303897857666,
|
| 10856 |
+
"step": 493
|
| 10857 |
+
},
|
| 10858 |
+
{
|
| 10859 |
+
"completion_length": 919.7500305175781,
|
| 10860 |
+
"completions/clipped_ratio": 0.45833333333333337,
|
| 10861 |
+
"completions/max_length": 1400.0,
|
| 10862 |
+
"completions/max_terminated_length": 1107.0,
|
| 10863 |
+
"completions/mean_length": 919.75,
|
| 10864 |
+
"completions/mean_terminated_length": 513.3846435546875,
|
| 10865 |
+
"completions/min_length": 259.0,
|
| 10866 |
+
"completions/min_terminated_length": 259.0,
|
| 10867 |
+
"epoch": 15.4375,
|
| 10868 |
+
"frac_reward_zero_std": 0.25,
|
| 10869 |
+
"grad_norm": 0.7352123260498047,
|
| 10870 |
+
"kl": 0.022763641085475683,
|
| 10871 |
+
"learning_rate": 6.089874350439505e-10,
|
| 10872 |
+
"loss": 0.1486,
|
| 10873 |
+
"num_tokens": 12789854.0,
|
| 10874 |
+
"reward": 0.534583330154419,
|
| 10875 |
+
"reward_std": 1.1599116325378418,
|
| 10876 |
+
"rewards/compute_weighted_reward/mean": 0.534583330154419,
|
| 10877 |
+
"rewards/compute_weighted_reward/std": 1.7653746604919434,
|
| 10878 |
+
"step": 494
|
| 10879 |
+
},
|
| 10880 |
+
{
|
| 10881 |
+
"completion_length": 816.6666870117188,
|
| 10882 |
+
"completions/clipped_ratio": 0.33333333333333337,
|
| 10883 |
+
"completions/max_length": 1400.0,
|
| 10884 |
+
"completions/max_terminated_length": 1353.0,
|
| 10885 |
+
"completions/mean_length": 821.7916870117188,
|
| 10886 |
+
"completions/mean_terminated_length": 532.6875,
|
| 10887 |
+
"completions/min_length": 285.0,
|
| 10888 |
+
"completions/min_terminated_length": 285.0,
|
| 10889 |
+
"epoch": 15.46875,
|
| 10890 |
+
"frac_reward_zero_std": 0.0,
|
| 10891 |
+
"grad_norm": 1.3775248527526855,
|
| 10892 |
+
"kl": 0.03886409197002649,
|
| 10893 |
+
"learning_rate": 4.933178929321102e-10,
|
| 10894 |
+
"loss": 0.1886,
|
| 10895 |
+
"num_tokens": 12817605.0,
|
| 10896 |
+
"reward": 0.5058333873748779,
|
| 10897 |
+
"reward_std": 1.149335265159607,
|
| 10898 |
+
"rewards/compute_weighted_reward/mean": 0.5058333277702332,
|
| 10899 |
+
"rewards/compute_weighted_reward/std": 1.560180902481079,
|
| 10900 |
+
"step": 495
|
| 10901 |
+
},
|
| 10902 |
+
{
|
| 10903 |
+
"completion_length": 809.1250305175781,
|
| 10904 |
+
"completions/clipped_ratio": 0.16666666666666663,
|
| 10905 |
+
"completions/max_length": 1400.0,
|
| 10906 |
+
"completions/max_terminated_length": 1317.0,
|
| 10907 |
+
"completions/mean_length": 809.125,
|
| 10908 |
+
"completions/mean_terminated_length": 690.9500122070312,
|
| 10909 |
+
"completions/min_length": 253.0,
|
| 10910 |
+
"completions/min_terminated_length": 253.0,
|
| 10911 |
+
"epoch": 15.5,
|
| 10912 |
+
"frac_reward_zero_std": 0.0,
|
| 10913 |
+
"grad_norm": 1.130301594734192,
|
| 10914 |
+
"kl": 0.024388552643358707,
|
| 10915 |
+
"learning_rate": 3.898089545047445e-10,
|
| 10916 |
+
"loss": 0.0514,
|
| 10917 |
+
"num_tokens": 12844950.0,
|
| 10918 |
+
"reward": 1.0716667175292969,
|
| 10919 |
+
"reward_std": 1.2712818384170532,
|
| 10920 |
+
"rewards/compute_weighted_reward/mean": 1.0716667175292969,
|
| 10921 |
+
"rewards/compute_weighted_reward/std": 1.4085689783096313,
|
| 10922 |
+
"step": 496
|
| 10923 |
+
},
|
| 10924 |
+
{
|
| 10925 |
+
"completion_length": 1121.9583435058594,
|
| 10926 |
+
"completions/clipped_ratio": 0.625,
|
| 10927 |
+
"completions/max_length": 1400.0,
|
| 10928 |
+
"completions/max_terminated_length": 930.0,
|
| 10929 |
+
"completions/mean_length": 1121.9583740234375,
|
| 10930 |
+
"completions/mean_terminated_length": 658.5555419921875,
|
| 10931 |
+
"completions/min_length": 392.0,
|
| 10932 |
+
"completions/min_terminated_length": 392.0,
|
| 10933 |
+
"epoch": 15.53125,
|
| 10934 |
+
"frac_reward_zero_std": 0.0,
|
| 10935 |
+
"grad_norm": 0.9501820206642151,
|
| 10936 |
+
"kl": 0.040883428417146206,
|
| 10937 |
+
"learning_rate": 2.9846566464150626e-10,
|
| 10938 |
+
"loss": 0.1841,
|
| 10939 |
+
"num_tokens": 12879269.0,
|
| 10940 |
+
"reward": -0.44583332538604736,
|
| 10941 |
+
"reward_std": 1.1888765096664429,
|
| 10942 |
+
"rewards/compute_weighted_reward/mean": -0.445833295583725,
|
| 10943 |
+
"rewards/compute_weighted_reward/std": 1.4850410223007202,
|
| 10944 |
+
"step": 497
|
| 10945 |
+
},
|
| 10946 |
+
{
|
| 10947 |
+
"completion_length": 645.6250076293945,
|
| 10948 |
+
"completions/clipped_ratio": 0.125,
|
| 10949 |
+
"completions/max_length": 1400.0,
|
| 10950 |
+
"completions/max_terminated_length": 1119.0,
|
| 10951 |
+
"completions/mean_length": 645.625,
|
| 10952 |
+
"completions/mean_terminated_length": 537.857177734375,
|
| 10953 |
+
"completions/min_length": 185.0,
|
| 10954 |
+
"completions/min_terminated_length": 185.0,
|
| 10955 |
+
"epoch": 15.5625,
|
| 10956 |
+
"frac_reward_zero_std": 0.5,
|
| 10957 |
+
"grad_norm": 0.46877413988113403,
|
| 10958 |
+
"kl": 0.013578799553215504,
|
| 10959 |
+
"learning_rate": 2.1929247528540418e-10,
|
| 10960 |
+
"loss": 0.0527,
|
| 10961 |
+
"num_tokens": 12901676.0,
|
| 10962 |
+
"reward": 2.0500001907348633,
|
| 10963 |
+
"reward_std": 0.5961085557937622,
|
| 10964 |
+
"rewards/compute_weighted_reward/mean": 2.0500001907348633,
|
| 10965 |
+
"rewards/compute_weighted_reward/std": 1.0217292308807373,
|
| 10966 |
+
"step": 498
|
| 10967 |
+
},
|
| 10968 |
+
{
|
| 10969 |
+
"completion_length": 947.8333740234375,
|
| 10970 |
+
"completions/clipped_ratio": 0.45833333333333337,
|
| 10971 |
+
"completions/max_length": 1400.0,
|
| 10972 |
+
"completions/max_terminated_length": 915.0,
|
| 10973 |
+
"completions/mean_length": 947.8333740234375,
|
| 10974 |
+
"completions/mean_terminated_length": 565.2307739257812,
|
| 10975 |
+
"completions/min_length": 307.0,
|
| 10976 |
+
"completions/min_terminated_length": 307.0,
|
| 10977 |
+
"epoch": 15.59375,
|
| 10978 |
+
"frac_reward_zero_std": 0.0,
|
| 10979 |
+
"grad_norm": 0.9576979279518127,
|
| 10980 |
+
"kl": 0.03686598874628544,
|
| 10981 |
+
"learning_rate": 1.5229324522605947e-10,
|
| 10982 |
+
"loss": 0.2095,
|
| 10983 |
+
"num_tokens": 12932380.0,
|
| 10984 |
+
"reward": 0.09458336234092712,
|
| 10985 |
+
"reward_std": 1.4334077835083008,
|
| 10986 |
+
"rewards/compute_weighted_reward/mean": 0.09458336979150772,
|
| 10987 |
+
"rewards/compute_weighted_reward/std": 1.6301053762435913,
|
| 10988 |
+
"step": 499
|
| 10989 |
+
},
|
| 10990 |
+
{
|
| 10991 |
+
"completion_length": 1035.250015258789,
|
| 10992 |
+
"completions/clipped_ratio": 0.5416666666666667,
|
| 10993 |
+
"completions/max_length": 1400.0,
|
| 10994 |
+
"completions/max_terminated_length": 1097.0,
|
| 10995 |
+
"completions/mean_length": 1035.25,
|
| 10996 |
+
"completions/mean_terminated_length": 604.1818237304688,
|
| 10997 |
+
"completions/min_length": 294.0,
|
| 10998 |
+
"completions/min_terminated_length": 294.0,
|
| 10999 |
+
"epoch": 15.625,
|
| 11000 |
+
"frac_reward_zero_std": 0.0,
|
| 11001 |
+
"grad_norm": 2.101461172103882,
|
| 11002 |
+
"kl": 0.04244990833103657,
|
| 11003 |
+
"learning_rate": 9.747123991141193e-11,
|
| 11004 |
+
"loss": 0.1304,
|
| 11005 |
+
"num_tokens": 12965248.0,
|
| 11006 |
+
"reward": 0.10583341121673584,
|
| 11007 |
+
"reward_std": 1.346000075340271,
|
| 11008 |
+
"rewards/compute_weighted_reward/mean": 0.10583335161209106,
|
| 11009 |
+
"rewards/compute_weighted_reward/std": 1.7509969472885132,
|
| 11010 |
+
"step": 500
|
| 11011 |
}
|
| 11012 |
],
|
| 11013 |
"logging_steps": 1,
|
| 11014 |
"max_steps": 500,
|
| 11015 |
+
"num_input_tokens_seen": 12965248,
|
| 11016 |
"num_train_epochs": 16,
|
| 11017 |
"save_steps": 10,
|
| 11018 |
"stateful_callbacks": {
|
|
|
|
| 11022 |
"should_evaluate": false,
|
| 11023 |
"should_log": false,
|
| 11024 |
"should_save": true,
|
| 11025 |
+
"should_training_stop": true
|
| 11026 |
},
|
| 11027 |
"attributes": {}
|
| 11028 |
}
|