Instructions to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.7 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.7 with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.7", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Unsloth Desktop
Training in progress, step 490, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 194563400
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1402844c18aeb9c024b30f9f23a9c32c507ea60a64defb20c169a2539f9385d1
|
| 3 |
size 194563400
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 100256339
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8d70bace88c1389d4b58e3fb6f248400c0142e1a1ca3e51d35870aa8eb36ebce
|
| 3 |
size 100256339
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14645
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:055ac3c369a5e930fe04e9758176075cca45da315a2b1d5fc770be3582b8afee
|
| 3 |
size 14645
|
last-checkpoint/scaler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1383
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f1992ea3bf098e43650cdec4c9702d09710b1b566a09c123b480dba8f6266dd0
|
| 3 |
size 1383
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9e559e024e64978c57350fefa54232ddfa7d2568f6cf9d5df8a027da19545eb1
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 2.
|
| 6 |
"eval_steps": 500,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -10568,11 +10568,231 @@
|
|
| 10568 |
"rewards/compute_weighted_reward/mean": 0.07333331555128098,
|
| 10569 |
"rewards/compute_weighted_reward/std": 0.9759306311607361,
|
| 10570 |
"step": 480
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10571 |
}
|
| 10572 |
],
|
| 10573 |
"logging_steps": 1,
|
| 10574 |
"max_steps": 1500,
|
| 10575 |
-
"num_input_tokens_seen":
|
| 10576 |
"num_train_epochs": 7,
|
| 10577 |
"save_steps": 10,
|
| 10578 |
"stateful_callbacks": {
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 2.2477064220183487,
|
| 6 |
"eval_steps": 500,
|
| 7 |
+
"global_step": 490,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 10568 |
"rewards/compute_weighted_reward/mean": 0.07333331555128098,
|
| 10569 |
"rewards/compute_weighted_reward/std": 0.9759306311607361,
|
| 10570 |
"step": 480
|
| 10571 |
+
},
|
| 10572 |
+
{
|
| 10573 |
+
"completion_length": 1422.7083333333333,
|
| 10574 |
+
"completions/clipped_ratio": 0.625,
|
| 10575 |
+
"completions/max_length": 2000.0,
|
| 10576 |
+
"completions/max_terminated_length": 1969.0,
|
| 10577 |
+
"completions/mean_length": 1507.875,
|
| 10578 |
+
"completions/mean_terminated_length": 687.6666870117188,
|
| 10579 |
+
"completions/min_length": 392.0,
|
| 10580 |
+
"completions/min_terminated_length": 392.0,
|
| 10581 |
+
"epoch": 2.206422018348624,
|
| 10582 |
+
"frac_reward_zero_std": 0.0,
|
| 10583 |
+
"grad_norm": 0.36272257566452026,
|
| 10584 |
+
"kl": 0.0001912083589559188,
|
| 10585 |
+
"learning_rate": 4.298349500846628e-08,
|
| 10586 |
+
"loss": 0.1471,
|
| 10587 |
+
"num_tokens": 15153385.0,
|
| 10588 |
+
"reward": -0.25708338618278503,
|
| 10589 |
+
"reward_std": 0.6896874308586121,
|
| 10590 |
+
"rewards/compute_weighted_reward/mean": -0.25708332657814026,
|
| 10591 |
+
"rewards/compute_weighted_reward/std": 1.1804289817810059,
|
| 10592 |
+
"step": 481
|
| 10593 |
+
},
|
| 10594 |
+
{
|
| 10595 |
+
"completion_length": 1335.875,
|
| 10596 |
+
"completions/clipped_ratio": 0.5416666666666667,
|
| 10597 |
+
"completions/max_length": 2000.0,
|
| 10598 |
+
"completions/max_terminated_length": 1679.0,
|
| 10599 |
+
"completions/mean_length": 1403.375,
|
| 10600 |
+
"completions/mean_terminated_length": 698.2727661132812,
|
| 10601 |
+
"completions/min_length": 270.0,
|
| 10602 |
+
"completions/min_terminated_length": 270.0,
|
| 10603 |
+
"epoch": 2.2110091743119265,
|
| 10604 |
+
"frac_reward_zero_std": 0.0,
|
| 10605 |
+
"grad_norm": 0.5260336399078369,
|
| 10606 |
+
"kl": 0.0002945731584986788,
|
| 10607 |
+
"learning_rate": 4.294303276506441e-08,
|
| 10608 |
+
"loss": 0.2248,
|
| 10609 |
+
"num_tokens": 15195334.0,
|
| 10610 |
+
"reward": -0.29208335280418396,
|
| 10611 |
+
"reward_std": 0.6641569137573242,
|
| 10612 |
+
"rewards/compute_weighted_reward/mean": -0.29208335280418396,
|
| 10613 |
+
"rewards/compute_weighted_reward/std": 0.8769609928131104,
|
| 10614 |
+
"step": 482
|
| 10615 |
+
},
|
| 10616 |
+
{
|
| 10617 |
+
"completion_length": 1571.375,
|
| 10618 |
+
"completions/clipped_ratio": 0.625,
|
| 10619 |
+
"completions/max_length": 2000.0,
|
| 10620 |
+
"completions/max_terminated_length": 1997.0,
|
| 10621 |
+
"completions/mean_length": 1571.375,
|
| 10622 |
+
"completions/mean_terminated_length": 857.0,
|
| 10623 |
+
"completions/min_length": 270.0,
|
| 10624 |
+
"completions/min_terminated_length": 270.0,
|
| 10625 |
+
"epoch": 2.2155963302752295,
|
| 10626 |
+
"frac_reward_zero_std": 0.25,
|
| 10627 |
+
"grad_norm": 0.38027381896972656,
|
| 10628 |
+
"kl": 0.0002809951434604348,
|
| 10629 |
+
"learning_rate": 4.290247335263362e-08,
|
| 10630 |
+
"loss": 0.0275,
|
| 10631 |
+
"num_tokens": 15237241.0,
|
| 10632 |
+
"reward": 0.1479167640209198,
|
| 10633 |
+
"reward_std": 0.5390461683273315,
|
| 10634 |
+
"rewards/compute_weighted_reward/mean": 0.14791667461395264,
|
| 10635 |
+
"rewards/compute_weighted_reward/std": 1.8478398323059082,
|
| 10636 |
+
"step": 483
|
| 10637 |
+
},
|
| 10638 |
+
{
|
| 10639 |
+
"completion_length": 1442.0833333333333,
|
| 10640 |
+
"completions/clipped_ratio": 0.5416666666666667,
|
| 10641 |
+
"completions/max_length": 2000.0,
|
| 10642 |
+
"completions/max_terminated_length": 1958.0,
|
| 10643 |
+
"completions/mean_length": 1442.0833740234375,
|
| 10644 |
+
"completions/mean_terminated_length": 782.727294921875,
|
| 10645 |
+
"completions/min_length": 454.0,
|
| 10646 |
+
"completions/min_terminated_length": 454.0,
|
| 10647 |
+
"epoch": 2.220183486238532,
|
| 10648 |
+
"frac_reward_zero_std": 0.0,
|
| 10649 |
+
"grad_norm": 0.42852839827537537,
|
| 10650 |
+
"kl": 0.0002766387536515443,
|
| 10651 |
+
"learning_rate": 4.2861816990820083e-08,
|
| 10652 |
+
"loss": 0.1764,
|
| 10653 |
+
"num_tokens": 15276579.0,
|
| 10654 |
+
"reward": -0.39499998092651367,
|
| 10655 |
+
"reward_std": 0.7189545631408691,
|
| 10656 |
+
"rewards/compute_weighted_reward/mean": -0.39499998092651367,
|
| 10657 |
+
"rewards/compute_weighted_reward/std": 0.8256170153617859,
|
| 10658 |
+
"step": 484
|
| 10659 |
+
},
|
| 10660 |
+
{
|
| 10661 |
+
"completion_length": 1310.7916666666667,
|
| 10662 |
+
"completions/clipped_ratio": 0.5,
|
| 10663 |
+
"completions/max_length": 2000.0,
|
| 10664 |
+
"completions/max_terminated_length": 966.0,
|
| 10665 |
+
"completions/mean_length": 1313.4583740234375,
|
| 10666 |
+
"completions/mean_terminated_length": 626.9166870117188,
|
| 10667 |
+
"completions/min_length": 345.0,
|
| 10668 |
+
"completions/min_terminated_length": 345.0,
|
| 10669 |
+
"epoch": 2.2247706422018347,
|
| 10670 |
+
"frac_reward_zero_std": 0.25,
|
| 10671 |
+
"grad_norm": 0.3990933299064636,
|
| 10672 |
+
"kl": 0.0002945650800635728,
|
| 10673 |
+
"learning_rate": 4.282106389979501e-08,
|
| 10674 |
+
"loss": 0.1089,
|
| 10675 |
+
"num_tokens": 15316136.0,
|
| 10676 |
+
"reward": -0.3725000321865082,
|
| 10677 |
+
"reward_std": 0.6126512289047241,
|
| 10678 |
+
"rewards/compute_weighted_reward/mean": -0.3724999725818634,
|
| 10679 |
+
"rewards/compute_weighted_reward/std": 0.906643271446228,
|
| 10680 |
+
"step": 485
|
| 10681 |
+
},
|
| 10682 |
+
{
|
| 10683 |
+
"completion_length": 1326.75,
|
| 10684 |
+
"completions/clipped_ratio": 0.41666666666666663,
|
| 10685 |
+
"completions/max_length": 2000.0,
|
| 10686 |
+
"completions/max_terminated_length": 1965.0,
|
| 10687 |
+
"completions/mean_length": 1331.875,
|
| 10688 |
+
"completions/mean_terminated_length": 854.6428833007812,
|
| 10689 |
+
"completions/min_length": 401.0,
|
| 10690 |
+
"completions/min_terminated_length": 401.0,
|
| 10691 |
+
"epoch": 2.229357798165138,
|
| 10692 |
+
"frac_reward_zero_std": 0.0,
|
| 10693 |
+
"grad_norm": 0.5519124269485474,
|
| 10694 |
+
"kl": 0.0003618740717380812,
|
| 10695 |
+
"learning_rate": 4.2780214300253425e-08,
|
| 10696 |
+
"loss": 0.137,
|
| 10697 |
+
"num_tokens": 15354791.0,
|
| 10698 |
+
"reward": 0.06541667878627777,
|
| 10699 |
+
"reward_std": 1.19478178024292,
|
| 10700 |
+
"rewards/compute_weighted_reward/mean": 0.06541666388511658,
|
| 10701 |
+
"rewards/compute_weighted_reward/std": 1.288430094718933,
|
| 10702 |
+
"step": 486
|
| 10703 |
+
},
|
| 10704 |
+
{
|
| 10705 |
+
"completion_length": 1770.0416666666667,
|
| 10706 |
+
"completions/clipped_ratio": 0.7916666666666666,
|
| 10707 |
+
"completions/max_length": 2000.0,
|
| 10708 |
+
"completions/max_terminated_length": 1416.0,
|
| 10709 |
+
"completions/mean_length": 1770.041748046875,
|
| 10710 |
+
"completions/mean_terminated_length": 896.2000122070312,
|
| 10711 |
+
"completions/min_length": 429.0,
|
| 10712 |
+
"completions/min_terminated_length": 429.0,
|
| 10713 |
+
"epoch": 2.2339449541284404,
|
| 10714 |
+
"frac_reward_zero_std": 0.0,
|
| 10715 |
+
"grad_norm": 0.5295096635818481,
|
| 10716 |
+
"kl": 0.0003984466287268636,
|
| 10717 |
+
"learning_rate": 4.273926841341302e-08,
|
| 10718 |
+
"loss": 0.147,
|
| 10719 |
+
"num_tokens": 15402444.0,
|
| 10720 |
+
"reward": -0.47291669249534607,
|
| 10721 |
+
"reward_std": 0.8791345357894897,
|
| 10722 |
+
"rewards/compute_weighted_reward/mean": -0.47291669249534607,
|
| 10723 |
+
"rewards/compute_weighted_reward/std": 1.13621985912323,
|
| 10724 |
+
"step": 487
|
| 10725 |
+
},
|
| 10726 |
+
{
|
| 10727 |
+
"completion_length": 1174.9166666666667,
|
| 10728 |
+
"completions/clipped_ratio": 0.375,
|
| 10729 |
+
"completions/max_length": 2000.0,
|
| 10730 |
+
"completions/max_terminated_length": 1251.0,
|
| 10731 |
+
"completions/mean_length": 1174.916748046875,
|
| 10732 |
+
"completions/mean_terminated_length": 679.86669921875,
|
| 10733 |
+
"completions/min_length": 389.0,
|
| 10734 |
+
"completions/min_terminated_length": 389.0,
|
| 10735 |
+
"epoch": 2.238532110091743,
|
| 10736 |
+
"frac_reward_zero_std": 0.25,
|
| 10737 |
+
"grad_norm": 0.31133127212524414,
|
| 10738 |
+
"kl": 0.00023924819227734892,
|
| 10739 |
+
"learning_rate": 4.269822646101289e-08,
|
| 10740 |
+
"loss": 0.1609,
|
| 10741 |
+
"num_tokens": 15437032.0,
|
| 10742 |
+
"reward": 0.6670833826065063,
|
| 10743 |
+
"reward_std": 0.39568936824798584,
|
| 10744 |
+
"rewards/compute_weighted_reward/mean": 0.6670833230018616,
|
| 10745 |
+
"rewards/compute_weighted_reward/std": 1.9870023727416992,
|
| 10746 |
+
"step": 488
|
| 10747 |
+
},
|
| 10748 |
+
{
|
| 10749 |
+
"completion_length": 1224.8333333333333,
|
| 10750 |
+
"completions/clipped_ratio": 0.41666666666666663,
|
| 10751 |
+
"completions/max_length": 2000.0,
|
| 10752 |
+
"completions/max_terminated_length": 1414.0,
|
| 10753 |
+
"completions/mean_length": 1224.8333740234375,
|
| 10754 |
+
"completions/mean_terminated_length": 671.1428833007812,
|
| 10755 |
+
"completions/min_length": 205.0,
|
| 10756 |
+
"completions/min_terminated_length": 205.0,
|
| 10757 |
+
"epoch": 2.243119266055046,
|
| 10758 |
+
"frac_reward_zero_std": 0.0,
|
| 10759 |
+
"grad_norm": 0.36838299036026,
|
| 10760 |
+
"kl": 0.00021752280918008182,
|
| 10761 |
+
"learning_rate": 4.265708866531238e-08,
|
| 10762 |
+
"loss": 0.1666,
|
| 10763 |
+
"num_tokens": 15471570.0,
|
| 10764 |
+
"reward": 0.31291672587394714,
|
| 10765 |
+
"reward_std": 0.687069296836853,
|
| 10766 |
+
"rewards/compute_weighted_reward/mean": 0.31291666626930237,
|
| 10767 |
+
"rewards/compute_weighted_reward/std": 1.5205562114715576,
|
| 10768 |
+
"step": 489
|
| 10769 |
+
},
|
| 10770 |
+
{
|
| 10771 |
+
"completion_length": 1431.0833333333333,
|
| 10772 |
+
"completions/clipped_ratio": 0.5416666666666667,
|
| 10773 |
+
"completions/max_length": 2000.0,
|
| 10774 |
+
"completions/max_terminated_length": 1937.0,
|
| 10775 |
+
"completions/mean_length": 1431.0833740234375,
|
| 10776 |
+
"completions/mean_terminated_length": 758.727294921875,
|
| 10777 |
+
"completions/min_length": 281.0,
|
| 10778 |
+
"completions/min_terminated_length": 281.0,
|
| 10779 |
+
"epoch": 2.2477064220183487,
|
| 10780 |
+
"frac_reward_zero_std": 0.0,
|
| 10781 |
+
"grad_norm": 0.47285446524620056,
|
| 10782 |
+
"kl": 0.0003024169091077056,
|
| 10783 |
+
"learning_rate": 4.2615855249089863e-08,
|
| 10784 |
+
"loss": 0.0715,
|
| 10785 |
+
"num_tokens": 15511040.0,
|
| 10786 |
+
"reward": -0.021666675806045532,
|
| 10787 |
+
"reward_std": 0.8609253764152527,
|
| 10788 |
+
"rewards/compute_weighted_reward/mean": -0.021666666492819786,
|
| 10789 |
+
"rewards/compute_weighted_reward/std": 1.4374394416809082,
|
| 10790 |
+
"step": 490
|
| 10791 |
}
|
| 10792 |
],
|
| 10793 |
"logging_steps": 1,
|
| 10794 |
"max_steps": 1500,
|
| 10795 |
+
"num_input_tokens_seen": 15511040,
|
| 10796 |
"num_train_epochs": 7,
|
| 10797 |
"save_steps": 10,
|
| 10798 |
"stateful_callbacks": {
|