Instructions to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.2 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.2 with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.2", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Unsloth Desktop
Training in progress, step 470, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 194563400
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e97f0845aa6a67e07d4147bdbc645881cd725e485863c57bcc61520e8aa526f0
|
| 3 |
size 194563400
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 100256339
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e076fb439be773d56148898b13cdfb393391bdee8311e86bd9314c0519333d5a
|
| 3 |
size 100256339
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14645
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f759ddb7949658691c66ca99f40d508b0c01db148b293eb4b7d316ed5233a57e
|
| 3 |
size 14645
|
last-checkpoint/scaler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1383
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8471f0babb163e30a8ef2747dbd9c4e7f32c7d3318b6dd67c171a8d396f73ab6
|
| 3 |
size 1383
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b15150ddaa3faf35858f43938a74f78c04812ac2062419c41ac0afaa4a8ae794
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 24.
|
| 6 |
"eval_steps": 500,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -10128,11 +10128,231 @@
|
|
| 10128 |
"rewards/compute_weighted_reward/mean": 0.4879167079925537,
|
| 10129 |
"rewards/compute_weighted_reward/std": 1.3894351720809937,
|
| 10130 |
"step": 460
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10131 |
}
|
| 10132 |
],
|
| 10133 |
"logging_steps": 1,
|
| 10134 |
"max_steps": 500,
|
| 10135 |
-
"num_input_tokens_seen":
|
| 10136 |
"num_train_epochs": 27,
|
| 10137 |
"save_steps": 10,
|
| 10138 |
"stateful_callbacks": {
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 24.736842105263158,
|
| 6 |
"eval_steps": 500,
|
| 7 |
+
"global_step": 470,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 10128 |
"rewards/compute_weighted_reward/mean": 0.4879167079925537,
|
| 10129 |
"rewards/compute_weighted_reward/std": 1.3894351720809937,
|
| 10130 |
"step": 460
|
| 10131 |
+
},
|
| 10132 |
+
{
|
| 10133 |
+
"completion_length": 1400.0,
|
| 10134 |
+
"completions/clipped_ratio": 0.25,
|
| 10135 |
+
"completions/max_length": 1400.0,
|
| 10136 |
+
"completions/max_terminated_length": 975.0,
|
| 10137 |
+
"completions/mean_length": 795.3333740234375,
|
| 10138 |
+
"completions/mean_terminated_length": 593.7777709960938,
|
| 10139 |
+
"completions/min_length": 327.0,
|
| 10140 |
+
"completions/min_terminated_length": 327.0,
|
| 10141 |
+
"epoch": 24.263157894736842,
|
| 10142 |
+
"frac_reward_zero_std": 0.25,
|
| 10143 |
+
"grad_norm": 2.256335496902466,
|
| 10144 |
+
"kl": 0.006880732020363212,
|
| 10145 |
+
"learning_rate": 4.472163787765637e-09,
|
| 10146 |
+
"loss": 0.0069,
|
| 10147 |
+
"num_tokens": 12572861.0,
|
| 10148 |
+
"reward": 1.6516666412353516,
|
| 10149 |
+
"reward_std": 1.1727336645126343,
|
| 10150 |
+
"rewards/compute_weighted_reward/mean": 1.6516666412353516,
|
| 10151 |
+
"rewards/compute_weighted_reward/std": 1.6009553670883179,
|
| 10152 |
+
"step": 461
|
| 10153 |
+
},
|
| 10154 |
+
{
|
| 10155 |
+
"completion_length": 1400.0,
|
| 10156 |
+
"completions/clipped_ratio": 0.25,
|
| 10157 |
+
"completions/max_length": 1400.0,
|
| 10158 |
+
"completions/max_terminated_length": 1265.0,
|
| 10159 |
+
"completions/mean_length": 871.6666870117188,
|
| 10160 |
+
"completions/mean_terminated_length": 695.5555419921875,
|
| 10161 |
+
"completions/min_length": 264.0,
|
| 10162 |
+
"completions/min_terminated_length": 264.0,
|
| 10163 |
+
"epoch": 24.31578947368421,
|
| 10164 |
+
"frac_reward_zero_std": 0.25,
|
| 10165 |
+
"grad_norm": 2.1168675422668457,
|
| 10166 |
+
"kl": 0.005777199752628803,
|
| 10167 |
+
"learning_rate": 4.2680502467932756e-09,
|
| 10168 |
+
"loss": 0.0058,
|
| 10169 |
+
"num_tokens": 12601245.0,
|
| 10170 |
+
"reward": 0.7454167604446411,
|
| 10171 |
+
"reward_std": 0.9587024450302124,
|
| 10172 |
+
"rewards/compute_weighted_reward/mean": 0.7454167008399963,
|
| 10173 |
+
"rewards/compute_weighted_reward/std": 1.4346518516540527,
|
| 10174 |
+
"step": 462
|
| 10175 |
+
},
|
| 10176 |
+
{
|
| 10177 |
+
"completion_length": 1400.0,
|
| 10178 |
+
"completions/clipped_ratio": 0.29166666666666663,
|
| 10179 |
+
"completions/max_length": 1400.0,
|
| 10180 |
+
"completions/max_terminated_length": 1360.0,
|
| 10181 |
+
"completions/mean_length": 902.2083740234375,
|
| 10182 |
+
"completions/mean_terminated_length": 697.2352905273438,
|
| 10183 |
+
"completions/min_length": 335.0,
|
| 10184 |
+
"completions/min_terminated_length": 335.0,
|
| 10185 |
+
"epoch": 24.36842105263158,
|
| 10186 |
+
"frac_reward_zero_std": 0.0,
|
| 10187 |
+
"grad_norm": 17.3585147857666,
|
| 10188 |
+
"kl": 0.0072209787322208285,
|
| 10189 |
+
"learning_rate": 4.068602545994249e-09,
|
| 10190 |
+
"loss": 0.0072,
|
| 10191 |
+
"num_tokens": 12630596.0,
|
| 10192 |
+
"reward": 0.4395833909511566,
|
| 10193 |
+
"reward_std": 1.1506423950195312,
|
| 10194 |
+
"rewards/compute_weighted_reward/mean": 0.4395833909511566,
|
| 10195 |
+
"rewards/compute_weighted_reward/std": 1.45299232006073,
|
| 10196 |
+
"step": 463
|
| 10197 |
+
},
|
| 10198 |
+
{
|
| 10199 |
+
"completion_length": 1400.0,
|
| 10200 |
+
"completions/clipped_ratio": 0.08333333333333337,
|
| 10201 |
+
"completions/max_length": 1400.0,
|
| 10202 |
+
"completions/max_terminated_length": 1076.0,
|
| 10203 |
+
"completions/mean_length": 646.2916870117188,
|
| 10204 |
+
"completions/mean_terminated_length": 577.7727661132812,
|
| 10205 |
+
"completions/min_length": 328.0,
|
| 10206 |
+
"completions/min_terminated_length": 328.0,
|
| 10207 |
+
"epoch": 24.42105263157895,
|
| 10208 |
+
"frac_reward_zero_std": 0.25,
|
| 10209 |
+
"grad_norm": 4.130858898162842,
|
| 10210 |
+
"kl": 0.011352596920914948,
|
| 10211 |
+
"learning_rate": 3.873830406168111e-09,
|
| 10212 |
+
"loss": 0.0114,
|
| 10213 |
+
"num_tokens": 12653349.0,
|
| 10214 |
+
"reward": 1.3262500762939453,
|
| 10215 |
+
"reward_std": 0.8212506175041199,
|
| 10216 |
+
"rewards/compute_weighted_reward/mean": 1.3262500762939453,
|
| 10217 |
+
"rewards/compute_weighted_reward/std": 1.3137494325637817,
|
| 10218 |
+
"step": 464
|
| 10219 |
+
},
|
| 10220 |
+
{
|
| 10221 |
+
"completion_length": 1400.0,
|
| 10222 |
+
"completions/clipped_ratio": 0.125,
|
| 10223 |
+
"completions/max_length": 1400.0,
|
| 10224 |
+
"completions/max_terminated_length": 1138.0,
|
| 10225 |
+
"completions/mean_length": 720.2083740234375,
|
| 10226 |
+
"completions/mean_terminated_length": 623.0952758789062,
|
| 10227 |
+
"completions/min_length": 260.0,
|
| 10228 |
+
"completions/min_terminated_length": 260.0,
|
| 10229 |
+
"epoch": 24.473684210526315,
|
| 10230 |
+
"frac_reward_zero_std": 0.25,
|
| 10231 |
+
"grad_norm": 5.011742115020752,
|
| 10232 |
+
"kl": 0.007324963924475014,
|
| 10233 |
+
"learning_rate": 3.6837433202341896e-09,
|
| 10234 |
+
"loss": 0.0073,
|
| 10235 |
+
"num_tokens": 12677888.0,
|
| 10236 |
+
"reward": 1.6141667366027832,
|
| 10237 |
+
"reward_std": 1.100345492362976,
|
| 10238 |
+
"rewards/compute_weighted_reward/mean": 1.6141667366027832,
|
| 10239 |
+
"rewards/compute_weighted_reward/std": 1.6239241361618042,
|
| 10240 |
+
"step": 465
|
| 10241 |
+
},
|
| 10242 |
+
{
|
| 10243 |
+
"completion_length": 1400.0,
|
| 10244 |
+
"completions/clipped_ratio": 0.375,
|
| 10245 |
+
"completions/max_length": 1400.0,
|
| 10246 |
+
"completions/max_terminated_length": 1268.0,
|
| 10247 |
+
"completions/mean_length": 916.6666870117188,
|
| 10248 |
+
"completions/mean_terminated_length": 626.6666870117188,
|
| 10249 |
+
"completions/min_length": 227.0,
|
| 10250 |
+
"completions/min_terminated_length": 227.0,
|
| 10251 |
+
"epoch": 24.526315789473685,
|
| 10252 |
+
"frac_reward_zero_std": 0.0,
|
| 10253 |
+
"grad_norm": 4.586663722991943,
|
| 10254 |
+
"kl": 0.006580363435205072,
|
| 10255 |
+
"learning_rate": 3.4983505527688583e-09,
|
| 10256 |
+
"loss": 0.0066,
|
| 10257 |
+
"num_tokens": 12707922.0,
|
| 10258 |
+
"reward": 0.42875003814697266,
|
| 10259 |
+
"reward_std": 1.1746701002120972,
|
| 10260 |
+
"rewards/compute_weighted_reward/mean": 0.42875003814697266,
|
| 10261 |
+
"rewards/compute_weighted_reward/std": 1.6151788234710693,
|
| 10262 |
+
"step": 466
|
| 10263 |
+
},
|
| 10264 |
+
{
|
| 10265 |
+
"completion_length": 1400.0,
|
| 10266 |
+
"completions/clipped_ratio": 0.45833333333333337,
|
| 10267 |
+
"completions/max_length": 1400.0,
|
| 10268 |
+
"completions/max_terminated_length": 1269.0,
|
| 10269 |
+
"completions/mean_length": 1025.25,
|
| 10270 |
+
"completions/mean_terminated_length": 708.1538696289062,
|
| 10271 |
+
"completions/min_length": 122.0,
|
| 10272 |
+
"completions/min_terminated_length": 122.0,
|
| 10273 |
+
"epoch": 24.57894736842105,
|
| 10274 |
+
"frac_reward_zero_std": 0.0,
|
| 10275 |
+
"grad_norm": 7.444450855255127,
|
| 10276 |
+
"kl": 0.008480201940983534,
|
| 10277 |
+
"learning_rate": 3.317661139554062e-09,
|
| 10278 |
+
"loss": 0.0085,
|
| 10279 |
+
"num_tokens": 12740580.0,
|
| 10280 |
+
"reward": -0.2133333384990692,
|
| 10281 |
+
"reward_std": 1.4371541738510132,
|
| 10282 |
+
"rewards/compute_weighted_reward/mean": -0.21333330869674683,
|
| 10283 |
+
"rewards/compute_weighted_reward/std": 1.4814025163650513,
|
| 10284 |
+
"step": 467
|
| 10285 |
+
},
|
| 10286 |
+
{
|
| 10287 |
+
"completion_length": 1400.0,
|
| 10288 |
+
"completions/clipped_ratio": 0.5,
|
| 10289 |
+
"completions/max_length": 1400.0,
|
| 10290 |
+
"completions/max_terminated_length": 1121.0,
|
| 10291 |
+
"completions/mean_length": 957.75,
|
| 10292 |
+
"completions/mean_terminated_length": 515.5,
|
| 10293 |
+
"completions/min_length": 291.0,
|
| 10294 |
+
"completions/min_terminated_length": 291.0,
|
| 10295 |
+
"epoch": 24.63157894736842,
|
| 10296 |
+
"frac_reward_zero_std": 0.0,
|
| 10297 |
+
"grad_norm": 12.957950592041016,
|
| 10298 |
+
"kl": 0.0020423264359124005,
|
| 10299 |
+
"learning_rate": 3.141683887136892e-09,
|
| 10300 |
+
"loss": 0.002,
|
| 10301 |
+
"num_tokens": 12770844.0,
|
| 10302 |
+
"reward": 0.46041667461395264,
|
| 10303 |
+
"reward_std": 0.7552928924560547,
|
| 10304 |
+
"rewards/compute_weighted_reward/mean": 0.460416704416275,
|
| 10305 |
+
"rewards/compute_weighted_reward/std": 1.6309572458267212,
|
| 10306 |
+
"step": 468
|
| 10307 |
+
},
|
| 10308 |
+
{
|
| 10309 |
+
"completion_length": 1400.0,
|
| 10310 |
+
"completions/clipped_ratio": 0.33333333333333337,
|
| 10311 |
+
"completions/max_length": 1400.0,
|
| 10312 |
+
"completions/max_terminated_length": 1305.0,
|
| 10313 |
+
"completions/mean_length": 895.875,
|
| 10314 |
+
"completions/mean_terminated_length": 643.8125,
|
| 10315 |
+
"completions/min_length": 227.0,
|
| 10316 |
+
"completions/min_terminated_length": 227.0,
|
| 10317 |
+
"epoch": 24.68421052631579,
|
| 10318 |
+
"frac_reward_zero_std": 0.25,
|
| 10319 |
+
"grad_norm": 2.4682981967926025,
|
| 10320 |
+
"kl": 0.004297248087823391,
|
| 10321 |
+
"learning_rate": 2.9704273724003525e-09,
|
| 10322 |
+
"loss": 0.0043,
|
| 10323 |
+
"num_tokens": 12799539.0,
|
| 10324 |
+
"reward": 0.7029167413711548,
|
| 10325 |
+
"reward_std": 0.8361326456069946,
|
| 10326 |
+
"rewards/compute_weighted_reward/mean": 0.70291668176651,
|
| 10327 |
+
"rewards/compute_weighted_reward/std": 1.548310399055481,
|
| 10328 |
+
"step": 469
|
| 10329 |
+
},
|
| 10330 |
+
{
|
| 10331 |
+
"completion_length": 1400.0,
|
| 10332 |
+
"completions/clipped_ratio": 0.125,
|
| 10333 |
+
"completions/max_length": 1400.0,
|
| 10334 |
+
"completions/max_terminated_length": 1051.0,
|
| 10335 |
+
"completions/mean_length": 685.0416870117188,
|
| 10336 |
+
"completions/mean_terminated_length": 582.90478515625,
|
| 10337 |
+
"completions/min_length": 152.0,
|
| 10338 |
+
"completions/min_terminated_length": 152.0,
|
| 10339 |
+
"epoch": 24.736842105263158,
|
| 10340 |
+
"frac_reward_zero_std": 0.0,
|
| 10341 |
+
"grad_norm": 8.62110710144043,
|
| 10342 |
+
"kl": 0.0092578538460657,
|
| 10343 |
+
"learning_rate": 2.803899942145382e-09,
|
| 10344 |
+
"loss": 0.0093,
|
| 10345 |
+
"num_tokens": 12823528.0,
|
| 10346 |
+
"reward": 0.8570834994316101,
|
| 10347 |
+
"reward_std": 1.151677131652832,
|
| 10348 |
+
"rewards/compute_weighted_reward/mean": 0.8570833802223206,
|
| 10349 |
+
"rewards/compute_weighted_reward/std": 1.384502649307251,
|
| 10350 |
+
"step": 470
|
| 10351 |
}
|
| 10352 |
],
|
| 10353 |
"logging_steps": 1,
|
| 10354 |
"max_steps": 500,
|
| 10355 |
+
"num_input_tokens_seen": 12823528,
|
| 10356 |
"num_train_epochs": 27,
|
| 10357 |
"save_steps": 10,
|
| 10358 |
"stateful_callbacks": {
|