Instructions to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.7 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.7 with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.7", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Unsloth Desktop
Training in progress, step 350, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 194563400
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4486f35279323679530cf73f57d30330d896d78adba3db63862bf00fd96a3426
|
| 3 |
size 194563400
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 100256339
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bb7b055ac92ba714cd690e0a196dd8b598d7d11d17591f4d7df3379cba6e1df8
|
| 3 |
size 100256339
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14645
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e43cfc9b78bf4fcc2c474982c0b41c03250d3b56f391c4d232a2e2b1a37ff8d3
|
| 3 |
size 14645
|
last-checkpoint/scaler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1383
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e7a3b7c7f6ae3d4b9e2fcb0774d91cea73c152177e3f6818dfc11db3a9a74275
|
| 3 |
size 1383
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3f0606b2be3dd375d9eeb91bd17f7f75fb345244906b8d79f255436b74591a97
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 2.
|
| 6 |
"eval_steps": 500,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -7488,11 +7488,231 @@
|
|
| 7488 |
"rewards/compute_weighted_reward/mean": 1.880416750907898,
|
| 7489 |
"rewards/compute_weighted_reward/std": 1.6687992811203003,
|
| 7490 |
"step": 340
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 7491 |
}
|
| 7492 |
],
|
| 7493 |
"logging_steps": 1,
|
| 7494 |
"max_steps": 1500,
|
| 7495 |
-
"num_input_tokens_seen":
|
| 7496 |
"num_train_epochs": 12,
|
| 7497 |
"save_steps": 10,
|
| 7498 |
"stateful_callbacks": {
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 2.8,
|
| 6 |
"eval_steps": 500,
|
| 7 |
+
"global_step": 350,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 7488 |
"rewards/compute_weighted_reward/mean": 1.880416750907898,
|
| 7489 |
"rewards/compute_weighted_reward/std": 1.6687992811203003,
|
| 7490 |
"step": 340
|
| 7491 |
+
},
|
| 7492 |
+
{
|
| 7493 |
+
"completion_length": 675.5833333333334,
|
| 7494 |
+
"completions/clipped_ratio": 0.125,
|
| 7495 |
+
"completions/max_length": 2000.0,
|
| 7496 |
+
"completions/max_terminated_length": 1690.0,
|
| 7497 |
+
"completions/mean_length": 675.5833740234375,
|
| 7498 |
+
"completions/mean_terminated_length": 486.3809509277344,
|
| 7499 |
+
"completions/min_length": 232.0,
|
| 7500 |
+
"completions/min_terminated_length": 232.0,
|
| 7501 |
+
"epoch": 2.7279999999999998,
|
| 7502 |
+
"frac_reward_zero_std": 0.25,
|
| 7503 |
+
"grad_norm": 0.4411628544330597,
|
| 7504 |
+
"kl": 0.0001766148191488052,
|
| 7505 |
+
"learning_rate": 4.9003041134365705e-08,
|
| 7506 |
+
"loss": 0.1452,
|
| 7507 |
+
"num_tokens": 10454794.0,
|
| 7508 |
+
"reward": 1.412083387374878,
|
| 7509 |
+
"reward_std": 1.055234432220459,
|
| 7510 |
+
"rewards/compute_weighted_reward/mean": 1.4120832681655884,
|
| 7511 |
+
"rewards/compute_weighted_reward/std": 1.6767903566360474,
|
| 7512 |
+
"step": 341
|
| 7513 |
+
},
|
| 7514 |
+
{
|
| 7515 |
+
"completion_length": 960.5833333333334,
|
| 7516 |
+
"completions/clipped_ratio": 0.125,
|
| 7517 |
+
"completions/max_length": 1737.0,
|
| 7518 |
+
"completions/max_terminated_length": 1578.0,
|
| 7519 |
+
"completions/mean_length": 960.5833740234375,
|
| 7520 |
+
"completions/mean_terminated_length": 849.6666870117188,
|
| 7521 |
+
"completions/min_length": 349.0,
|
| 7522 |
+
"completions/min_terminated_length": 349.0,
|
| 7523 |
+
"epoch": 2.7359999999999998,
|
| 7524 |
+
"frac_reward_zero_std": 0.25,
|
| 7525 |
+
"grad_norm": 0.3251778483390808,
|
| 7526 |
+
"kl": 0.00015508719335836454,
|
| 7527 |
+
"learning_rate": 4.898574603995548e-08,
|
| 7528 |
+
"loss": 0.0453,
|
| 7529 |
+
"num_tokens": 10486392.0,
|
| 7530 |
+
"reward": 0.7616666555404663,
|
| 7531 |
+
"reward_std": 0.807964026927948,
|
| 7532 |
+
"rewards/compute_weighted_reward/mean": 0.7616665959358215,
|
| 7533 |
+
"rewards/compute_weighted_reward/std": 1.4406177997589111,
|
| 7534 |
+
"step": 342
|
| 7535 |
+
},
|
| 7536 |
+
{
|
| 7537 |
+
"completion_length": 1160.0,
|
| 7538 |
+
"completions/clipped_ratio": 0.41666666666666663,
|
| 7539 |
+
"completions/max_length": 2000.0,
|
| 7540 |
+
"completions/max_terminated_length": 1064.0,
|
| 7541 |
+
"completions/mean_length": 1165.3333740234375,
|
| 7542 |
+
"completions/mean_terminated_length": 569.1428833007812,
|
| 7543 |
+
"completions/min_length": 188.0,
|
| 7544 |
+
"completions/min_terminated_length": 188.0,
|
| 7545 |
+
"epoch": 2.7439999999999998,
|
| 7546 |
+
"frac_reward_zero_std": 0.0,
|
| 7547 |
+
"grad_norm": 0.5680258870124817,
|
| 7548 |
+
"kl": 0.00021060577925406201,
|
| 7549 |
+
"learning_rate": 4.896830532173602e-08,
|
| 7550 |
+
"loss": 0.1619,
|
| 7551 |
+
"num_tokens": 10522532.0,
|
| 7552 |
+
"reward": 0.5170833468437195,
|
| 7553 |
+
"reward_std": 1.2685867547988892,
|
| 7554 |
+
"rewards/compute_weighted_reward/mean": 0.5170833468437195,
|
| 7555 |
+
"rewards/compute_weighted_reward/std": 1.8057841062545776,
|
| 7556 |
+
"step": 343
|
| 7557 |
+
},
|
| 7558 |
+
{
|
| 7559 |
+
"completion_length": 743.375,
|
| 7560 |
+
"completions/clipped_ratio": 0.16666666666666663,
|
| 7561 |
+
"completions/max_length": 1949.0,
|
| 7562 |
+
"completions/max_terminated_length": 1716.0,
|
| 7563 |
+
"completions/mean_length": 743.375,
|
| 7564 |
+
"completions/mean_terminated_length": 502.25,
|
| 7565 |
+
"completions/min_length": 286.0,
|
| 7566 |
+
"completions/min_terminated_length": 286.0,
|
| 7567 |
+
"epoch": 2.752,
|
| 7568 |
+
"frac_reward_zero_std": 0.25,
|
| 7569 |
+
"grad_norm": 0.2251403033733368,
|
| 7570 |
+
"kl": 0.00010871579085384535,
|
| 7571 |
+
"learning_rate": 4.895071908559452e-08,
|
| 7572 |
+
"loss": 0.0623,
|
| 7573 |
+
"num_tokens": 10548191.0,
|
| 7574 |
+
"reward": 2.2512502670288086,
|
| 7575 |
+
"reward_std": 0.5843120813369751,
|
| 7576 |
+
"rewards/compute_weighted_reward/mean": 2.2512500286102295,
|
| 7577 |
+
"rewards/compute_weighted_reward/std": 1.9334185123443604,
|
| 7578 |
+
"step": 344
|
| 7579 |
+
},
|
| 7580 |
+
{
|
| 7581 |
+
"completion_length": 646.6666666666666,
|
| 7582 |
+
"completions/clipped_ratio": 0.08333333333333337,
|
| 7583 |
+
"completions/max_length": 2000.0,
|
| 7584 |
+
"completions/max_terminated_length": 1780.0,
|
| 7585 |
+
"completions/mean_length": 646.6666870117188,
|
| 7586 |
+
"completions/mean_terminated_length": 523.6363525390625,
|
| 7587 |
+
"completions/min_length": 250.0,
|
| 7588 |
+
"completions/min_terminated_length": 250.0,
|
| 7589 |
+
"epoch": 2.76,
|
| 7590 |
+
"frac_reward_zero_std": 0.5,
|
| 7591 |
+
"grad_norm": 0.2939091920852661,
|
| 7592 |
+
"kl": 0.00013193290965318738,
|
| 7593 |
+
"learning_rate": 4.8932987438301677e-08,
|
| 7594 |
+
"loss": 0.0348,
|
| 7595 |
+
"num_tokens": 10571151.0,
|
| 7596 |
+
"reward": 1.0816667079925537,
|
| 7597 |
+
"reward_std": 0.6287143230438232,
|
| 7598 |
+
"rewards/compute_weighted_reward/mean": 1.0816667079925537,
|
| 7599 |
+
"rewards/compute_weighted_reward/std": 1.3874456882476807,
|
| 7600 |
+
"step": 345
|
| 7601 |
+
},
|
| 7602 |
+
{
|
| 7603 |
+
"completion_length": 862.1666666666666,
|
| 7604 |
+
"completions/clipped_ratio": 0.20833333333333337,
|
| 7605 |
+
"completions/max_length": 2000.0,
|
| 7606 |
+
"completions/max_terminated_length": 1420.0,
|
| 7607 |
+
"completions/mean_length": 862.1666870117188,
|
| 7608 |
+
"completions/mean_terminated_length": 562.73681640625,
|
| 7609 |
+
"completions/min_length": 223.0,
|
| 7610 |
+
"completions/min_terminated_length": 223.0,
|
| 7611 |
+
"epoch": 2.768,
|
| 7612 |
+
"frac_reward_zero_std": 0.0,
|
| 7613 |
+
"grad_norm": 0.375482976436615,
|
| 7614 |
+
"kl": 0.00013598250078909283,
|
| 7615 |
+
"learning_rate": 4.891511048751102e-08,
|
| 7616 |
+
"loss": 0.1055,
|
| 7617 |
+
"num_tokens": 10599601.0,
|
| 7618 |
+
"reward": 1.0750000476837158,
|
| 7619 |
+
"reward_std": 0.5297895669937134,
|
| 7620 |
+
"rewards/compute_weighted_reward/mean": 1.0749999284744263,
|
| 7621 |
+
"rewards/compute_weighted_reward/std": 1.3649685382843018,
|
| 7622 |
+
"step": 346
|
| 7623 |
+
},
|
| 7624 |
+
{
|
| 7625 |
+
"completion_length": 1025.125,
|
| 7626 |
+
"completions/clipped_ratio": 0.20833333333333337,
|
| 7627 |
+
"completions/max_length": 2000.0,
|
| 7628 |
+
"completions/max_terminated_length": 1771.0,
|
| 7629 |
+
"completions/mean_length": 1025.125,
|
| 7630 |
+
"completions/mean_terminated_length": 768.5789794921875,
|
| 7631 |
+
"completions/min_length": 408.0,
|
| 7632 |
+
"completions/min_terminated_length": 408.0,
|
| 7633 |
+
"epoch": 2.776,
|
| 7634 |
+
"frac_reward_zero_std": 0.25,
|
| 7635 |
+
"grad_norm": 0.35587796568870544,
|
| 7636 |
+
"kl": 0.0001902539578016634,
|
| 7637 |
+
"learning_rate": 4.889708834175823e-08,
|
| 7638 |
+
"loss": 0.1037,
|
| 7639 |
+
"num_tokens": 10631908.0,
|
| 7640 |
+
"reward": 0.41458332538604736,
|
| 7641 |
+
"reward_std": 0.8512474894523621,
|
| 7642 |
+
"rewards/compute_weighted_reward/mean": 0.41458332538604736,
|
| 7643 |
+
"rewards/compute_weighted_reward/std": 1.1446017026901245,
|
| 7644 |
+
"step": 347
|
| 7645 |
+
},
|
| 7646 |
+
{
|
| 7647 |
+
"completion_length": 760.375,
|
| 7648 |
+
"completions/clipped_ratio": 0.16666666666666663,
|
| 7649 |
+
"completions/max_length": 2000.0,
|
| 7650 |
+
"completions/max_terminated_length": 1163.0,
|
| 7651 |
+
"completions/mean_length": 760.375,
|
| 7652 |
+
"completions/mean_terminated_length": 512.4500122070312,
|
| 7653 |
+
"completions/min_length": 279.0,
|
| 7654 |
+
"completions/min_terminated_length": 279.0,
|
| 7655 |
+
"epoch": 2.784,
|
| 7656 |
+
"frac_reward_zero_std": 0.25,
|
| 7657 |
+
"grad_norm": 0.3131572902202606,
|
| 7658 |
+
"kl": 0.00019689797333436823,
|
| 7659 |
+
"learning_rate": 4.8878921110460536e-08,
|
| 7660 |
+
"loss": 0.0717,
|
| 7661 |
+
"num_tokens": 10658035.0,
|
| 7662 |
+
"reward": 0.4520833194255829,
|
| 7663 |
+
"reward_std": 0.8793530464172363,
|
| 7664 |
+
"rewards/compute_weighted_reward/mean": 0.4520833492279053,
|
| 7665 |
+
"rewards/compute_weighted_reward/std": 1.1837999820709229,
|
| 7666 |
+
"step": 348
|
| 7667 |
+
},
|
| 7668 |
+
{
|
| 7669 |
+
"completion_length": 1075.4166666666667,
|
| 7670 |
+
"completions/clipped_ratio": 0.33333333333333337,
|
| 7671 |
+
"completions/max_length": 2000.0,
|
| 7672 |
+
"completions/max_terminated_length": 1009.0,
|
| 7673 |
+
"completions/mean_length": 1118.0,
|
| 7674 |
+
"completions/mean_terminated_length": 677.0,
|
| 7675 |
+
"completions/min_length": 352.0,
|
| 7676 |
+
"completions/min_terminated_length": 352.0,
|
| 7677 |
+
"epoch": 2.792,
|
| 7678 |
+
"frac_reward_zero_std": 0.0,
|
| 7679 |
+
"grad_norm": 0.5552175045013428,
|
| 7680 |
+
"kl": 0.0001898959982706098,
|
| 7681 |
+
"learning_rate": 4.8860608903916e-08,
|
| 7682 |
+
"loss": 0.2009,
|
| 7683 |
+
"num_tokens": 10693225.0,
|
| 7684 |
+
"reward": 0.3566667437553406,
|
| 7685 |
+
"reward_std": 1.2520555257797241,
|
| 7686 |
+
"rewards/compute_weighted_reward/mean": 0.3566666841506958,
|
| 7687 |
+
"rewards/compute_weighted_reward/std": 1.4636603593826294,
|
| 7688 |
+
"step": 349
|
| 7689 |
+
},
|
| 7690 |
+
{
|
| 7691 |
+
"completion_length": 1150.9166666666667,
|
| 7692 |
+
"completions/clipped_ratio": 0.375,
|
| 7693 |
+
"completions/max_length": 2000.0,
|
| 7694 |
+
"completions/max_terminated_length": 1194.0,
|
| 7695 |
+
"completions/mean_length": 1150.916748046875,
|
| 7696 |
+
"completions/mean_terminated_length": 641.4666748046875,
|
| 7697 |
+
"completions/min_length": 273.0,
|
| 7698 |
+
"completions/min_terminated_length": 273.0,
|
| 7699 |
+
"epoch": 2.8,
|
| 7700 |
+
"frac_reward_zero_std": 0.0,
|
| 7701 |
+
"grad_norm": 0.36329564452171326,
|
| 7702 |
+
"kl": 0.00010918113108952336,
|
| 7703 |
+
"learning_rate": 4.8842151833302874e-08,
|
| 7704 |
+
"loss": 0.1516,
|
| 7705 |
+
"num_tokens": 10728071.0,
|
| 7706 |
+
"reward": -0.19333335757255554,
|
| 7707 |
+
"reward_std": 0.5602983236312866,
|
| 7708 |
+
"rewards/compute_weighted_reward/mean": -0.19333334267139435,
|
| 7709 |
+
"rewards/compute_weighted_reward/std": 0.6461839079856873,
|
| 7710 |
+
"step": 350
|
| 7711 |
}
|
| 7712 |
],
|
| 7713 |
"logging_steps": 1,
|
| 7714 |
"max_steps": 1500,
|
| 7715 |
+
"num_input_tokens_seen": 10728071,
|
| 7716 |
"num_train_epochs": 12,
|
| 7717 |
"save_steps": 10,
|
| 7718 |
"stateful_callbacks": {
|