Instructions to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.3 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.3 with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("fc91/CQoT_SFT_merged_16bit_lora-Llama-3.2-3B-Instruct-v.DR-GRPO_7.3", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Unsloth Desktop
Training in progress, step 490, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 194563400
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:214d4c92e201996ef5000afb14d16c3a2cb53eba45c80910a30250ccf9fb671e
|
| 3 |
size 194563400
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 100256339
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2b5a77bd4fe37d8b4c68f7fc3e886c3e283d1129eb3d38b56b63c170b448ff35
|
| 3 |
size 100256339
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14645
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4a993517fcad6b88249c3736cbc047e0728eb92a48102fcb394563ea5ea4fdaf
|
| 3 |
size 14645
|
last-checkpoint/scaler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1383
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:631b854a8072c273e62f84ce9d54f2ab354ec712a745e46ebbee521a3aed2150
|
| 3 |
size 1383
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0085eb8f8087f7a151dc3cbff0d73149538e7fd3411aef06647570da27d3da03
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch":
|
| 6 |
"eval_steps": 500,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -10568,11 +10568,231 @@
|
|
| 10568 |
"rewards/compute_weighted_reward/mean": 0.9749999642372131,
|
| 10569 |
"rewards/compute_weighted_reward/std": 1.6546036005020142,
|
| 10570 |
"step": 480
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10571 |
}
|
| 10572 |
],
|
| 10573 |
"logging_steps": 1,
|
| 10574 |
"max_steps": 500,
|
| 10575 |
-
"num_input_tokens_seen":
|
| 10576 |
"num_train_epochs": 56,
|
| 10577 |
"save_steps": 10,
|
| 10578 |
"stateful_callbacks": {
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 54.44444444444444,
|
| 6 |
"eval_steps": 500,
|
| 7 |
+
"global_step": 490,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 10568 |
"rewards/compute_weighted_reward/mean": 0.9749999642372131,
|
| 10569 |
"rewards/compute_weighted_reward/std": 1.6546036005020142,
|
| 10570 |
"step": 480
|
| 10571 |
+
},
|
| 10572 |
+
{
|
| 10573 |
+
"completion_length": 1400.0,
|
| 10574 |
+
"completions/clipped_ratio": 0.25,
|
| 10575 |
+
"completions/max_length": 1400.0,
|
| 10576 |
+
"completions/max_terminated_length": 1201.0,
|
| 10577 |
+
"completions/mean_length": 800.6666870117188,
|
| 10578 |
+
"completions/mean_terminated_length": 600.888916015625,
|
| 10579 |
+
"completions/min_length": 272.0,
|
| 10580 |
+
"completions/min_terminated_length": 272.0,
|
| 10581 |
+
"epoch": 53.44444444444444,
|
| 10582 |
+
"frac_reward_zero_std": 0.0,
|
| 10583 |
+
"grad_norm": 6.093226432800293,
|
| 10584 |
+
"kl": 0.011718678055331111,
|
| 10585 |
+
"learning_rate": 1.2863734927012094e-09,
|
| 10586 |
+
"loss": 0.0117,
|
| 10587 |
+
"num_tokens": 13162020.0,
|
| 10588 |
+
"reward": 0.6654167175292969,
|
| 10589 |
+
"reward_std": 0.9684948921203613,
|
| 10590 |
+
"rewards/compute_weighted_reward/mean": 0.6654166579246521,
|
| 10591 |
+
"rewards/compute_weighted_reward/std": 1.245635747909546,
|
| 10592 |
+
"step": 481
|
| 10593 |
+
},
|
| 10594 |
+
{
|
| 10595 |
+
"completion_length": 1400.0,
|
| 10596 |
+
"completions/clipped_ratio": 0.29166666666666663,
|
| 10597 |
+
"completions/max_length": 1400.0,
|
| 10598 |
+
"completions/max_terminated_length": 1368.0,
|
| 10599 |
+
"completions/mean_length": 931.625,
|
| 10600 |
+
"completions/mean_terminated_length": 738.7647094726562,
|
| 10601 |
+
"completions/min_length": 344.0,
|
| 10602 |
+
"completions/min_terminated_length": 344.0,
|
| 10603 |
+
"epoch": 53.55555555555556,
|
| 10604 |
+
"frac_reward_zero_std": 0.0,
|
| 10605 |
+
"grad_norm": 3.9402177333831787,
|
| 10606 |
+
"kl": 0.007335595786571503,
|
| 10607 |
+
"learning_rate": 1.1771618553447215e-09,
|
| 10608 |
+
"loss": 0.0073,
|
| 10609 |
+
"num_tokens": 13193259.0,
|
| 10610 |
+
"reward": 0.056666698306798935,
|
| 10611 |
+
"reward_std": 1.6200555562973022,
|
| 10612 |
+
"rewards/compute_weighted_reward/mean": 0.05666669085621834,
|
| 10613 |
+
"rewards/compute_weighted_reward/std": 1.5791513919830322,
|
| 10614 |
+
"step": 482
|
| 10615 |
+
},
|
| 10616 |
+
{
|
| 10617 |
+
"completion_length": 1400.0,
|
| 10618 |
+
"completions/clipped_ratio": 0.04166666666666663,
|
| 10619 |
+
"completions/max_length": 1400.0,
|
| 10620 |
+
"completions/max_terminated_length": 1345.0,
|
| 10621 |
+
"completions/mean_length": 604.5416870117188,
|
| 10622 |
+
"completions/mean_terminated_length": 569.95654296875,
|
| 10623 |
+
"completions/min_length": 256.0,
|
| 10624 |
+
"completions/min_terminated_length": 256.0,
|
| 10625 |
+
"epoch": 53.666666666666664,
|
| 10626 |
+
"frac_reward_zero_std": 0.25,
|
| 10627 |
+
"grad_norm": 5.48298978805542,
|
| 10628 |
+
"kl": 0.00782405596692115,
|
| 10629 |
+
"learning_rate": 1.0727667037011667e-09,
|
| 10630 |
+
"loss": 0.0078,
|
| 10631 |
+
"num_tokens": 13215070.0,
|
| 10632 |
+
"reward": 1.693333387374878,
|
| 10633 |
+
"reward_std": 0.9139549732208252,
|
| 10634 |
+
"rewards/compute_weighted_reward/mean": 1.6933332681655884,
|
| 10635 |
+
"rewards/compute_weighted_reward/std": 1.196791648864746,
|
| 10636 |
+
"step": 483
|
| 10637 |
+
},
|
| 10638 |
+
{
|
| 10639 |
+
"completion_length": 1400.0,
|
| 10640 |
+
"completions/clipped_ratio": 0.20833333333333337,
|
| 10641 |
+
"completions/max_length": 1400.0,
|
| 10642 |
+
"completions/max_terminated_length": 1268.0,
|
| 10643 |
+
"completions/mean_length": 825.7916870117188,
|
| 10644 |
+
"completions/mean_terminated_length": 674.6842041015625,
|
| 10645 |
+
"completions/min_length": 275.0,
|
| 10646 |
+
"completions/min_terminated_length": 275.0,
|
| 10647 |
+
"epoch": 53.77777777777778,
|
| 10648 |
+
"frac_reward_zero_std": 0.0,
|
| 10649 |
+
"grad_norm": 3.4163742065429688,
|
| 10650 |
+
"kl": 0.008350545074790716,
|
| 10651 |
+
"learning_rate": 9.731931258429638e-10,
|
| 10652 |
+
"loss": 0.0084,
|
| 10653 |
+
"num_tokens": 13243193.0,
|
| 10654 |
+
"reward": 0.880000114440918,
|
| 10655 |
+
"reward_std": 0.974814772605896,
|
| 10656 |
+
"rewards/compute_weighted_reward/mean": 0.8800000548362732,
|
| 10657 |
+
"rewards/compute_weighted_reward/std": 1.2584704160690308,
|
| 10658 |
+
"step": 484
|
| 10659 |
+
},
|
| 10660 |
+
{
|
| 10661 |
+
"completion_length": 1400.0,
|
| 10662 |
+
"completions/clipped_ratio": 0.08333333333333337,
|
| 10663 |
+
"completions/max_length": 1400.0,
|
| 10664 |
+
"completions/max_terminated_length": 1184.0,
|
| 10665 |
+
"completions/mean_length": 619.4166870117188,
|
| 10666 |
+
"completions/mean_terminated_length": 548.45458984375,
|
| 10667 |
+
"completions/min_length": 200.0,
|
| 10668 |
+
"completions/min_terminated_length": 200.0,
|
| 10669 |
+
"epoch": 53.888888888888886,
|
| 10670 |
+
"frac_reward_zero_std": 0.0,
|
| 10671 |
+
"grad_norm": 3.2675859928131104,
|
| 10672 |
+
"kl": 0.015018837992101908,
|
| 10673 |
+
"learning_rate": 8.784459748458317e-10,
|
| 10674 |
+
"loss": 0.015,
|
| 10675 |
+
"num_tokens": 13265445.0,
|
| 10676 |
+
"reward": 1.5237501859664917,
|
| 10677 |
+
"reward_std": 1.057565689086914,
|
| 10678 |
+
"rewards/compute_weighted_reward/mean": 1.5237499475479126,
|
| 10679 |
+
"rewards/compute_weighted_reward/std": 1.2061611413955688,
|
| 10680 |
+
"step": 485
|
| 10681 |
+
},
|
| 10682 |
+
{
|
| 10683 |
+
"completion_length": 1400.0,
|
| 10684 |
+
"completions/clipped_ratio": 0.45833333333333337,
|
| 10685 |
+
"completions/max_length": 1400.0,
|
| 10686 |
+
"completions/max_terminated_length": 1119.0,
|
| 10687 |
+
"completions/mean_length": 992.7916870117188,
|
| 10688 |
+
"completions/mean_terminated_length": 648.2307739257812,
|
| 10689 |
+
"completions/min_length": 417.0,
|
| 10690 |
+
"completions/min_terminated_length": 417.0,
|
| 10691 |
+
"epoch": 54.0,
|
| 10692 |
+
"frac_reward_zero_std": 0.0,
|
| 10693 |
+
"grad_norm": 4.8129425048828125,
|
| 10694 |
+
"kl": 0.006140718352980912,
|
| 10695 |
+
"learning_rate": 7.885298685522235e-10,
|
| 10696 |
+
"loss": 0.0061,
|
| 10697 |
+
"num_tokens": 13297918.0,
|
| 10698 |
+
"reward": -0.06458334624767303,
|
| 10699 |
+
"reward_std": 0.5508111715316772,
|
| 10700 |
+
"rewards/compute_weighted_reward/mean": -0.06458330154418945,
|
| 10701 |
+
"rewards/compute_weighted_reward/std": 1.783735752105713,
|
| 10702 |
+
"step": 486
|
| 10703 |
+
},
|
| 10704 |
+
{
|
| 10705 |
+
"completion_length": 1400.0,
|
| 10706 |
+
"completions/clipped_ratio": 0.33333333333333337,
|
| 10707 |
+
"completions/max_length": 1400.0,
|
| 10708 |
+
"completions/max_terminated_length": 1363.0,
|
| 10709 |
+
"completions/mean_length": 882.2083740234375,
|
| 10710 |
+
"completions/mean_terminated_length": 623.3125,
|
| 10711 |
+
"completions/min_length": 284.0,
|
| 10712 |
+
"completions/min_terminated_length": 284.0,
|
| 10713 |
+
"epoch": 54.111111111111114,
|
| 10714 |
+
"frac_reward_zero_std": 0.0,
|
| 10715 |
+
"grad_norm": 6.521482467651367,
|
| 10716 |
+
"kl": 0.0026615530950948596,
|
| 10717 |
+
"learning_rate": 7.034491893463057e-10,
|
| 10718 |
+
"loss": 0.0027,
|
| 10719 |
+
"num_tokens": 13326459.0,
|
| 10720 |
+
"reward": 0.5283333659172058,
|
| 10721 |
+
"reward_std": 1.406015157699585,
|
| 10722 |
+
"rewards/compute_weighted_reward/mean": 0.5283333659172058,
|
| 10723 |
+
"rewards/compute_weighted_reward/std": 1.6546525955200195,
|
| 10724 |
+
"step": 487
|
| 10725 |
+
},
|
| 10726 |
+
{
|
| 10727 |
+
"completion_length": 1400.0,
|
| 10728 |
+
"completions/clipped_ratio": 0.25,
|
| 10729 |
+
"completions/max_length": 1400.0,
|
| 10730 |
+
"completions/max_terminated_length": 1293.0,
|
| 10731 |
+
"completions/mean_length": 823.4583740234375,
|
| 10732 |
+
"completions/mean_terminated_length": 631.2777709960938,
|
| 10733 |
+
"completions/min_length": 342.0,
|
| 10734 |
+
"completions/min_terminated_length": 342.0,
|
| 10735 |
+
"epoch": 54.22222222222222,
|
| 10736 |
+
"frac_reward_zero_std": 0.0,
|
| 10737 |
+
"grad_norm": 13.027168273925781,
|
| 10738 |
+
"kl": 0.016307485639117658,
|
| 10739 |
+
"learning_rate": 6.23208083940363e-10,
|
| 10740 |
+
"loss": 0.0163,
|
| 10741 |
+
"num_tokens": 13355240.0,
|
| 10742 |
+
"reward": 0.9591667056083679,
|
| 10743 |
+
"reward_std": 1.122135877609253,
|
| 10744 |
+
"rewards/compute_weighted_reward/mean": 0.9591667056083679,
|
| 10745 |
+
"rewards/compute_weighted_reward/std": 1.348125696182251,
|
| 10746 |
+
"step": 488
|
| 10747 |
+
},
|
| 10748 |
+
{
|
| 10749 |
+
"completion_length": 1400.0,
|
| 10750 |
+
"completions/clipped_ratio": 0.04166666666666663,
|
| 10751 |
+
"completions/max_length": 1400.0,
|
| 10752 |
+
"completions/max_terminated_length": 1327.0,
|
| 10753 |
+
"completions/mean_length": 658.2083740234375,
|
| 10754 |
+
"completions/mean_terminated_length": 625.95654296875,
|
| 10755 |
+
"completions/min_length": 262.0,
|
| 10756 |
+
"completions/min_terminated_length": 262.0,
|
| 10757 |
+
"epoch": 54.333333333333336,
|
| 10758 |
+
"frac_reward_zero_std": 0.5,
|
| 10759 |
+
"grad_norm": 5.863112926483154,
|
| 10760 |
+
"kl": 0.009161067078821361,
|
| 10761 |
+
"learning_rate": 5.47810463172671e-10,
|
| 10762 |
+
"loss": 0.0092,
|
| 10763 |
+
"num_tokens": 13378183.0,
|
| 10764 |
+
"reward": 1.6708335876464844,
|
| 10765 |
+
"reward_std": 0.5133538246154785,
|
| 10766 |
+
"rewards/compute_weighted_reward/mean": 1.6708334684371948,
|
| 10767 |
+
"rewards/compute_weighted_reward/std": 1.1477991342544556,
|
| 10768 |
+
"step": 489
|
| 10769 |
+
},
|
| 10770 |
+
{
|
| 10771 |
+
"completion_length": 1400.0,
|
| 10772 |
+
"completions/clipped_ratio": 0.29166666666666663,
|
| 10773 |
+
"completions/max_length": 1400.0,
|
| 10774 |
+
"completions/max_terminated_length": 1060.0,
|
| 10775 |
+
"completions/mean_length": 854.4583740234375,
|
| 10776 |
+
"completions/mean_terminated_length": 629.8235473632812,
|
| 10777 |
+
"completions/min_length": 146.0,
|
| 10778 |
+
"completions/min_terminated_length": 146.0,
|
| 10779 |
+
"epoch": 54.44444444444444,
|
| 10780 |
+
"frac_reward_zero_std": 0.0,
|
| 10781 |
+
"grad_norm": 8.846403121948242,
|
| 10782 |
+
"kl": 0.008679779479280114,
|
| 10783 |
+
"learning_rate": 4.772600018168815e-10,
|
| 10784 |
+
"loss": 0.0087,
|
| 10785 |
+
"num_tokens": 13406868.0,
|
| 10786 |
+
"reward": 0.15416672825813293,
|
| 10787 |
+
"reward_std": 1.329119086265564,
|
| 10788 |
+
"rewards/compute_weighted_reward/mean": 0.15416669845581055,
|
| 10789 |
+
"rewards/compute_weighted_reward/std": 1.5062823295593262,
|
| 10790 |
+
"step": 490
|
| 10791 |
}
|
| 10792 |
],
|
| 10793 |
"logging_steps": 1,
|
| 10794 |
"max_steps": 500,
|
| 10795 |
+
"num_input_tokens_seen": 13406868,
|
| 10796 |
"num_train_epochs": 56,
|
| 10797 |
"save_steps": 10,
|
| 10798 |
"stateful_callbacks": {
|