Instructions to use yuyanghu06/lecungpt with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use yuyanghu06/lecungpt with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("togethercomputer/Qwen3.5-4B") model = PeftModel.from_pretrained(base_model, "yuyanghu06/lecungpt") - Notebooks
- Google Colab
- Kaggle
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 3.0, | |
| "eval_steps": 117, | |
| "global_step": 117, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.02564102564102564, | |
| "grad_norm": 0.20494528114795685, | |
| "learning_rate": 1e-05, | |
| "loss": 2.944, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.05128205128205128, | |
| "grad_norm": 0.564369261264801, | |
| "learning_rate": 9.998197638354428e-06, | |
| "loss": 2.7139, | |
| "step": 2 | |
| }, | |
| { | |
| "epoch": 0.07692307692307693, | |
| "grad_norm": 0.2275543510913849, | |
| "learning_rate": 9.992791852820709e-06, | |
| "loss": 3.002, | |
| "step": 3 | |
| }, | |
| { | |
| "epoch": 0.10256410256410256, | |
| "grad_norm": 0.1662287712097168, | |
| "learning_rate": 9.983786540671052e-06, | |
| "loss": 2.6937, | |
| "step": 4 | |
| }, | |
| { | |
| "epoch": 0.1282051282051282, | |
| "grad_norm": 0.30666545033454895, | |
| "learning_rate": 9.971188194237141e-06, | |
| "loss": 2.5144, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.15384615384615385, | |
| "grad_norm": 0.22469215095043182, | |
| "learning_rate": 9.955005896229543e-06, | |
| "loss": 2.5842, | |
| "step": 6 | |
| }, | |
| { | |
| "epoch": 0.1794871794871795, | |
| "grad_norm": 0.2725427448749542, | |
| "learning_rate": 9.935251313189564e-06, | |
| "loss": 2.9155, | |
| "step": 7 | |
| }, | |
| { | |
| "epoch": 0.20512820512820512, | |
| "grad_norm": 0.19864390790462494, | |
| "learning_rate": 9.911938687078324e-06, | |
| "loss": 2.6917, | |
| "step": 8 | |
| }, | |
| { | |
| "epoch": 0.23076923076923078, | |
| "grad_norm": 0.15495699644088745, | |
| "learning_rate": 9.885084825009085e-06, | |
| "loss": 2.3191, | |
| "step": 9 | |
| }, | |
| { | |
| "epoch": 0.2564102564102564, | |
| "grad_norm": 0.7309098839759827, | |
| "learning_rate": 9.854709087130261e-06, | |
| "loss": 3.166, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.28205128205128205, | |
| "grad_norm": 0.2202342301607132, | |
| "learning_rate": 9.820833372667813e-06, | |
| "loss": 3.0812, | |
| "step": 11 | |
| }, | |
| { | |
| "epoch": 0.3076923076923077, | |
| "grad_norm": 0.45190778374671936, | |
| "learning_rate": 9.783482104137127e-06, | |
| "loss": 2.6462, | |
| "step": 12 | |
| }, | |
| { | |
| "epoch": 0.3333333333333333, | |
| "grad_norm": 0.2022932767868042, | |
| "learning_rate": 9.742682209735727e-06, | |
| "loss": 2.9554, | |
| "step": 13 | |
| }, | |
| { | |
| "epoch": 0.358974358974359, | |
| "grad_norm": 0.17690043151378632, | |
| "learning_rate": 9.698463103929542e-06, | |
| "loss": 2.5946, | |
| "step": 14 | |
| }, | |
| { | |
| "epoch": 0.38461538461538464, | |
| "grad_norm": 0.20987562835216522, | |
| "learning_rate": 9.650856666246693e-06, | |
| "loss": 2.4706, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.41025641025641024, | |
| "grad_norm": 0.19785286486148834, | |
| "learning_rate": 9.599897218294122e-06, | |
| "loss": 2.5767, | |
| "step": 16 | |
| }, | |
| { | |
| "epoch": 0.4358974358974359, | |
| "grad_norm": 0.26647791266441345, | |
| "learning_rate": 9.54562149901362e-06, | |
| "loss": 3.1842, | |
| "step": 17 | |
| }, | |
| { | |
| "epoch": 0.46153846153846156, | |
| "grad_norm": 0.14869138598442078, | |
| "learning_rate": 9.488068638195072e-06, | |
| "loss": 2.7924, | |
| "step": 18 | |
| }, | |
| { | |
| "epoch": 0.48717948717948717, | |
| "grad_norm": 0.18388637900352478, | |
| "learning_rate": 9.427280128266049e-06, | |
| "loss": 2.779, | |
| "step": 19 | |
| }, | |
| { | |
| "epoch": 0.5128205128205128, | |
| "grad_norm": 0.16335400938987732, | |
| "learning_rate": 9.363299794378072e-06, | |
| "loss": 2.9099, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.5384615384615384, | |
| "grad_norm": 0.3611254096031189, | |
| "learning_rate": 9.296173762811084e-06, | |
| "loss": 2.7484, | |
| "step": 21 | |
| }, | |
| { | |
| "epoch": 0.5641025641025641, | |
| "grad_norm": 0.22266459465026855, | |
| "learning_rate": 9.225950427718974e-06, | |
| "loss": 2.6731, | |
| "step": 22 | |
| }, | |
| { | |
| "epoch": 0.5897435897435898, | |
| "grad_norm": 0.2212408483028412, | |
| "learning_rate": 9.152680416240059e-06, | |
| "loss": 2.5159, | |
| "step": 23 | |
| }, | |
| { | |
| "epoch": 0.6153846153846154, | |
| "grad_norm": 0.24422743916511536, | |
| "learning_rate": 9.076416551997721e-06, | |
| "loss": 3.0238, | |
| "step": 24 | |
| }, | |
| { | |
| "epoch": 0.6410256410256411, | |
| "grad_norm": 0.17641104757785797, | |
| "learning_rate": 8.997213817017508e-06, | |
| "loss": 2.7511, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.6666666666666666, | |
| "grad_norm": 0.17786549031734467, | |
| "learning_rate": 8.915129312088112e-06, | |
| "loss": 2.8246, | |
| "step": 26 | |
| }, | |
| { | |
| "epoch": 0.6923076923076923, | |
| "grad_norm": 0.19077053666114807, | |
| "learning_rate": 8.83022221559489e-06, | |
| "loss": 2.8039, | |
| "step": 27 | |
| }, | |
| { | |
| "epoch": 0.717948717948718, | |
| "grad_norm": 0.16474156081676483, | |
| "learning_rate": 8.742553740855507e-06, | |
| "loss": 2.9055, | |
| "step": 28 | |
| }, | |
| { | |
| "epoch": 0.7435897435897436, | |
| "grad_norm": 0.16694943606853485, | |
| "learning_rate": 8.652187091988516e-06, | |
| "loss": 3.0026, | |
| "step": 29 | |
| }, | |
| { | |
| "epoch": 0.7692307692307693, | |
| "grad_norm": 0.18305574357509613, | |
| "learning_rate": 8.559187418346703e-06, | |
| "loss": 3.05, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.7948717948717948, | |
| "grad_norm": 0.44106462597846985, | |
| "learning_rate": 8.463621767547998e-06, | |
| "loss": 2.3594, | |
| "step": 31 | |
| }, | |
| { | |
| "epoch": 0.8205128205128205, | |
| "grad_norm": 0.13298189640045166, | |
| "learning_rate": 8.36555903713785e-06, | |
| "loss": 2.7032, | |
| "step": 32 | |
| }, | |
| { | |
| "epoch": 0.8461538461538461, | |
| "grad_norm": 0.16964633762836456, | |
| "learning_rate": 8.265069924917925e-06, | |
| "loss": 3.0049, | |
| "step": 33 | |
| }, | |
| { | |
| "epoch": 0.8717948717948718, | |
| "grad_norm": 0.21558941900730133, | |
| "learning_rate": 8.162226877976886e-06, | |
| "loss": 2.5803, | |
| "step": 34 | |
| }, | |
| { | |
| "epoch": 0.8974358974358975, | |
| "grad_norm": 0.1849488765001297, | |
| "learning_rate": 8.057104040460062e-06, | |
| "loss": 2.2861, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 0.9230769230769231, | |
| "grad_norm": 0.16992613673210144, | |
| "learning_rate": 7.949777200115617e-06, | |
| "loss": 2.5638, | |
| "step": 36 | |
| }, | |
| { | |
| "epoch": 0.9487179487179487, | |
| "grad_norm": 0.1811276227235794, | |
| "learning_rate": 7.84032373365578e-06, | |
| "loss": 2.7225, | |
| "step": 37 | |
| }, | |
| { | |
| "epoch": 0.9743589743589743, | |
| "grad_norm": 0.14251349866390228, | |
| "learning_rate": 7.728822550972523e-06, | |
| "loss": 2.5935, | |
| "step": 38 | |
| }, | |
| { | |
| "epoch": 1.0, | |
| "grad_norm": 0.20674623548984528, | |
| "learning_rate": 7.615354038247889e-06, | |
| "loss": 2.3489, | |
| "step": 39 | |
| }, | |
| { | |
| "epoch": 1.0256410256410255, | |
| "grad_norm": 0.16418299078941345, | |
| "learning_rate": 7.500000000000001e-06, | |
| "loss": 2.8806, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 1.0512820512820513, | |
| "grad_norm": 0.42056992650032043, | |
| "learning_rate": 7.382843600106539e-06, | |
| "loss": 2.5225, | |
| "step": 41 | |
| }, | |
| { | |
| "epoch": 1.0769230769230769, | |
| "grad_norm": 0.1651465892791748, | |
| "learning_rate": 7.263969301848188e-06, | |
| "loss": 2.9385, | |
| "step": 42 | |
| }, | |
| { | |
| "epoch": 1.1025641025641026, | |
| "grad_norm": 0.11612721532583237, | |
| "learning_rate": 7.143462807015271e-06, | |
| "loss": 2.65, | |
| "step": 43 | |
| }, | |
| { | |
| "epoch": 1.1282051282051282, | |
| "grad_norm": 0.22792533040046692, | |
| "learning_rate": 7.021410994121525e-06, | |
| "loss": 2.4115, | |
| "step": 44 | |
| }, | |
| { | |
| "epoch": 1.1538461538461537, | |
| "grad_norm": 0.18955646455287933, | |
| "learning_rate": 6.897901855769483e-06, | |
| "loss": 2.5088, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 1.1794871794871795, | |
| "grad_norm": 0.2061670422554016, | |
| "learning_rate": 6.773024435212678e-06, | |
| "loss": 2.8308, | |
| "step": 46 | |
| }, | |
| { | |
| "epoch": 1.205128205128205, | |
| "grad_norm": 0.1557702273130417, | |
| "learning_rate": 6.646868762160399e-06, | |
| "loss": 2.6351, | |
| "step": 47 | |
| }, | |
| { | |
| "epoch": 1.2307692307692308, | |
| "grad_norm": 0.13028891384601593, | |
| "learning_rate": 6.519525787871235e-06, | |
| "loss": 2.2725, | |
| "step": 48 | |
| }, | |
| { | |
| "epoch": 1.2564102564102564, | |
| "grad_norm": 0.5919710993766785, | |
| "learning_rate": 6.391087319582264e-06, | |
| "loss": 2.9043, | |
| "step": 49 | |
| }, | |
| { | |
| "epoch": 1.282051282051282, | |
| "grad_norm": 0.16268998384475708, | |
| "learning_rate": 6.261645954321109e-06, | |
| "loss": 3.0056, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 1.3076923076923077, | |
| "grad_norm": 0.3303690552711487, | |
| "learning_rate": 6.131295012148613e-06, | |
| "loss": 2.4932, | |
| "step": 51 | |
| }, | |
| { | |
| "epoch": 1.3333333333333333, | |
| "grad_norm": 0.140819251537323, | |
| "learning_rate": 6.000128468880223e-06, | |
| "loss": 2.8989, | |
| "step": 52 | |
| }, | |
| { | |
| "epoch": 1.358974358974359, | |
| "grad_norm": 0.13831716775894165, | |
| "learning_rate": 5.8682408883346535e-06, | |
| "loss": 2.5417, | |
| "step": 53 | |
| }, | |
| { | |
| "epoch": 1.3846153846153846, | |
| "grad_norm": 0.18448969721794128, | |
| "learning_rate": 5.735727354158581e-06, | |
| "loss": 2.4042, | |
| "step": 54 | |
| }, | |
| { | |
| "epoch": 1.4102564102564101, | |
| "grad_norm": 0.1544794738292694, | |
| "learning_rate": 5.6026834012766155e-06, | |
| "loss": 2.5227, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 1.435897435897436, | |
| "grad_norm": 0.22304964065551758, | |
| "learning_rate": 5.469204947015897e-06, | |
| "loss": 3.0919, | |
| "step": 56 | |
| }, | |
| { | |
| "epoch": 1.4615384615384617, | |
| "grad_norm": 0.1189192682504654, | |
| "learning_rate": 5.335388221955012e-06, | |
| "loss": 2.7538, | |
| "step": 57 | |
| }, | |
| { | |
| "epoch": 1.4871794871794872, | |
| "grad_norm": 0.14262886345386505, | |
| "learning_rate": 5.201329700547077e-06, | |
| "loss": 2.7141, | |
| "step": 58 | |
| }, | |
| { | |
| "epoch": 1.5128205128205128, | |
| "grad_norm": 0.12737467885017395, | |
| "learning_rate": 5.067126031566988e-06, | |
| "loss": 2.8652, | |
| "step": 59 | |
| }, | |
| { | |
| "epoch": 1.5384615384615383, | |
| "grad_norm": 0.2574975788593292, | |
| "learning_rate": 4.932873968433014e-06, | |
| "loss": 2.6555, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 1.564102564102564, | |
| "grad_norm": 0.18790772557258606, | |
| "learning_rate": 4.798670299452926e-06, | |
| "loss": 2.6082, | |
| "step": 61 | |
| }, | |
| { | |
| "epoch": 1.5897435897435899, | |
| "grad_norm": 0.17647786438465118, | |
| "learning_rate": 4.664611778044988e-06, | |
| "loss": 2.4594, | |
| "step": 62 | |
| }, | |
| { | |
| "epoch": 1.6153846153846154, | |
| "grad_norm": 0.1943756341934204, | |
| "learning_rate": 4.530795052984104e-06, | |
| "loss": 2.9589, | |
| "step": 63 | |
| }, | |
| { | |
| "epoch": 1.641025641025641, | |
| "grad_norm": 0.15042220056056976, | |
| "learning_rate": 4.397316598723385e-06, | |
| "loss": 2.7027, | |
| "step": 64 | |
| }, | |
| { | |
| "epoch": 1.6666666666666665, | |
| "grad_norm": 0.1804312914609909, | |
| "learning_rate": 4.264272645841419e-06, | |
| "loss": 2.7935, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 1.6923076923076923, | |
| "grad_norm": 0.1711619645357132, | |
| "learning_rate": 4.131759111665349e-06, | |
| "loss": 2.7634, | |
| "step": 66 | |
| }, | |
| { | |
| "epoch": 1.717948717948718, | |
| "grad_norm": 0.13611891865730286, | |
| "learning_rate": 3.999871531119779e-06, | |
| "loss": 2.8654, | |
| "step": 67 | |
| }, | |
| { | |
| "epoch": 1.7435897435897436, | |
| "grad_norm": 0.13859815895557404, | |
| "learning_rate": 3.86870498785139e-06, | |
| "loss": 2.9677, | |
| "step": 68 | |
| }, | |
| { | |
| "epoch": 1.7692307692307692, | |
| "grad_norm": 0.17095635831356049, | |
| "learning_rate": 3.7383540456788915e-06, | |
| "loss": 3.006, | |
| "step": 69 | |
| }, | |
| { | |
| "epoch": 1.7948717948717947, | |
| "grad_norm": 0.33051353693008423, | |
| "learning_rate": 3.6089126804177373e-06, | |
| "loss": 2.2646, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 1.8205128205128205, | |
| "grad_norm": 0.12325972318649292, | |
| "learning_rate": 3.480474212128766e-06, | |
| "loss": 2.6746, | |
| "step": 71 | |
| }, | |
| { | |
| "epoch": 1.8461538461538463, | |
| "grad_norm": 0.15464168787002563, | |
| "learning_rate": 3.3531312378396026e-06, | |
| "loss": 2.9761, | |
| "step": 72 | |
| }, | |
| { | |
| "epoch": 1.8717948717948718, | |
| "grad_norm": 0.1984647959470749, | |
| "learning_rate": 3.226975564787322e-06, | |
| "loss": 2.5415, | |
| "step": 73 | |
| }, | |
| { | |
| "epoch": 1.8974358974358974, | |
| "grad_norm": 0.17000555992126465, | |
| "learning_rate": 3.1020981442305187e-06, | |
| "loss": 2.2454, | |
| "step": 74 | |
| }, | |
| { | |
| "epoch": 1.9230769230769231, | |
| "grad_norm": 0.15753290057182312, | |
| "learning_rate": 2.978589005878476e-06, | |
| "loss": 2.535, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 1.9487179487179487, | |
| "grad_norm": 0.16940346360206604, | |
| "learning_rate": 2.8565371929847286e-06, | |
| "loss": 2.6881, | |
| "step": 76 | |
| }, | |
| { | |
| "epoch": 1.9743589743589745, | |
| "grad_norm": 0.13065537810325623, | |
| "learning_rate": 2.736030698151815e-06, | |
| "loss": 2.5689, | |
| "step": 77 | |
| }, | |
| { | |
| "epoch": 2.0, | |
| "grad_norm": 0.19400492310523987, | |
| "learning_rate": 2.6171563998934605e-06, | |
| "loss": 2.3126, | |
| "step": 78 | |
| }, | |
| { | |
| "epoch": 2.0256410256410255, | |
| "grad_norm": 0.1678439974784851, | |
| "learning_rate": 2.5000000000000015e-06, | |
| "loss": 2.8647, | |
| "step": 79 | |
| }, | |
| { | |
| "epoch": 2.051282051282051, | |
| "grad_norm": 0.36615338921546936, | |
| "learning_rate": 2.384645961752113e-06, | |
| "loss": 2.4473, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 2.076923076923077, | |
| "grad_norm": 0.15688776969909668, | |
| "learning_rate": 2.2711774490274767e-06, | |
| "loss": 2.9054, | |
| "step": 81 | |
| }, | |
| { | |
| "epoch": 2.1025641025641026, | |
| "grad_norm": 0.10982992500066757, | |
| "learning_rate": 2.159676266344222e-06, | |
| "loss": 2.6324, | |
| "step": 82 | |
| }, | |
| { | |
| "epoch": 2.128205128205128, | |
| "grad_norm": 0.21072708070278168, | |
| "learning_rate": 2.050222799884387e-06, | |
| "loss": 2.3846, | |
| "step": 83 | |
| }, | |
| { | |
| "epoch": 2.1538461538461537, | |
| "grad_norm": 0.18832428753376007, | |
| "learning_rate": 1.942895959539939e-06, | |
| "loss": 2.4825, | |
| "step": 84 | |
| }, | |
| { | |
| "epoch": 2.1794871794871793, | |
| "grad_norm": 0.20066532492637634, | |
| "learning_rate": 1.8377731220231144e-06, | |
| "loss": 2.7871, | |
| "step": 85 | |
| }, | |
| { | |
| "epoch": 2.2051282051282053, | |
| "grad_norm": 0.16142620146274567, | |
| "learning_rate": 1.7349300750820758e-06, | |
| "loss": 2.6139, | |
| "step": 86 | |
| }, | |
| { | |
| "epoch": 2.230769230769231, | |
| "grad_norm": 0.1329987347126007, | |
| "learning_rate": 1.6344409628621482e-06, | |
| "loss": 2.2679, | |
| "step": 87 | |
| }, | |
| { | |
| "epoch": 2.2564102564102564, | |
| "grad_norm": 0.48261162638664246, | |
| "learning_rate": 1.5363782324520033e-06, | |
| "loss": 2.8223, | |
| "step": 88 | |
| }, | |
| { | |
| "epoch": 2.282051282051282, | |
| "grad_norm": 0.16861321032047272, | |
| "learning_rate": 1.4408125816532981e-06, | |
| "loss": 2.9918, | |
| "step": 89 | |
| }, | |
| { | |
| "epoch": 2.3076923076923075, | |
| "grad_norm": 0.30975234508514404, | |
| "learning_rate": 1.347812908011485e-06, | |
| "loss": 2.4583, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 2.3333333333333335, | |
| "grad_norm": 0.5936061143875122, | |
| "learning_rate": 1.257446259144494e-06, | |
| "loss": 2.8887, | |
| "step": 91 | |
| }, | |
| { | |
| "epoch": 2.358974358974359, | |
| "grad_norm": 0.14251480996608734, | |
| "learning_rate": 1.1697777844051105e-06, | |
| "loss": 2.5273, | |
| "step": 92 | |
| }, | |
| { | |
| "epoch": 2.3846153846153846, | |
| "grad_norm": 0.1809547394514084, | |
| "learning_rate": 1.0848706879118893e-06, | |
| "loss": 2.3871, | |
| "step": 93 | |
| }, | |
| { | |
| "epoch": 2.41025641025641, | |
| "grad_norm": 0.1559205800294876, | |
| "learning_rate": 1.0027861829824953e-06, | |
| "loss": 2.5071, | |
| "step": 94 | |
| }, | |
| { | |
| "epoch": 2.435897435897436, | |
| "grad_norm": 0.22356708347797394, | |
| "learning_rate": 9.235834480022788e-07, | |
| "loss": 3.0696, | |
| "step": 95 | |
| }, | |
| { | |
| "epoch": 2.4615384615384617, | |
| "grad_norm": 0.11708559095859528, | |
| "learning_rate": 8.473195837599419e-07, | |
| "loss": 2.741, | |
| "step": 96 | |
| }, | |
| { | |
| "epoch": 2.4871794871794872, | |
| "grad_norm": 0.13850131630897522, | |
| "learning_rate": 7.740495722810271e-07, | |
| "loss": 2.7068, | |
| "step": 97 | |
| }, | |
| { | |
| "epoch": 2.5128205128205128, | |
| "grad_norm": 0.127314954996109, | |
| "learning_rate": 7.03826237188916e-07, | |
| "loss": 2.8583, | |
| "step": 98 | |
| }, | |
| { | |
| "epoch": 2.5384615384615383, | |
| "grad_norm": 0.2526121437549591, | |
| "learning_rate": 6.367002056219285e-07, | |
| "loss": 2.6353, | |
| "step": 99 | |
| }, | |
| { | |
| "epoch": 2.564102564102564, | |
| "grad_norm": 0.18439951539039612, | |
| "learning_rate": 5.727198717339511e-07, | |
| "loss": 2.5881, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 2.58974358974359, | |
| "grad_norm": 0.17252342402935028, | |
| "learning_rate": 5.119313618049309e-07, | |
| "loss": 2.4458, | |
| "step": 101 | |
| }, | |
| { | |
| "epoch": 2.6153846153846154, | |
| "grad_norm": 0.19350160658359528, | |
| "learning_rate": 4.54378500986381e-07, | |
| "loss": 2.9525, | |
| "step": 102 | |
| }, | |
| { | |
| "epoch": 2.641025641025641, | |
| "grad_norm": 0.15075629949569702, | |
| "learning_rate": 4.001027817058789e-07, | |
| "loss": 2.7002, | |
| "step": 103 | |
| }, | |
| { | |
| "epoch": 2.6666666666666665, | |
| "grad_norm": 0.18015150725841522, | |
| "learning_rate": 3.49143333753309e-07, | |
| "loss": 2.79, | |
| "step": 104 | |
| }, | |
| { | |
| "epoch": 2.6923076923076925, | |
| "grad_norm": 0.17058293521404266, | |
| "learning_rate": 3.015368960704584e-07, | |
| "loss": 2.7456, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 2.717948717948718, | |
| "grad_norm": 0.1371268630027771, | |
| "learning_rate": 2.573177902642726e-07, | |
| "loss": 2.8645, | |
| "step": 106 | |
| }, | |
| { | |
| "epoch": 2.7435897435897436, | |
| "grad_norm": 0.1372060924768448, | |
| "learning_rate": 2.1651789586287442e-07, | |
| "loss": 2.9625, | |
| "step": 107 | |
| }, | |
| { | |
| "epoch": 2.769230769230769, | |
| "grad_norm": 0.17000600695610046, | |
| "learning_rate": 1.7916662733218848e-07, | |
| "loss": 3.005, | |
| "step": 108 | |
| }, | |
| { | |
| "epoch": 2.7948717948717947, | |
| "grad_norm": 0.3287239074707031, | |
| "learning_rate": 1.4529091286973994e-07, | |
| "loss": 2.2515, | |
| "step": 109 | |
| }, | |
| { | |
| "epoch": 2.8205128205128203, | |
| "grad_norm": 0.12300807982683182, | |
| "learning_rate": 1.1491517499091498e-07, | |
| "loss": 2.6743, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 2.8461538461538463, | |
| "grad_norm": 0.15311472117900848, | |
| "learning_rate": 8.80613129216762e-08, | |
| "loss": 2.9753, | |
| "step": 111 | |
| }, | |
| { | |
| "epoch": 2.871794871794872, | |
| "grad_norm": 0.20104099810123444, | |
| "learning_rate": 6.474868681043578e-08, | |
| "loss": 2.5298, | |
| "step": 112 | |
| }, | |
| { | |
| "epoch": 2.8974358974358974, | |
| "grad_norm": 0.17030330002307892, | |
| "learning_rate": 4.499410377045765e-08, | |
| "loss": 2.2434, | |
| "step": 113 | |
| }, | |
| { | |
| "epoch": 2.9230769230769234, | |
| "grad_norm": 0.1580314338207245, | |
| "learning_rate": 2.8811805762860578e-08, | |
| "loss": 2.5331, | |
| "step": 114 | |
| }, | |
| { | |
| "epoch": 2.948717948717949, | |
| "grad_norm": 0.1715579628944397, | |
| "learning_rate": 1.6213459328950355e-08, | |
| "loss": 2.688, | |
| "step": 115 | |
| }, | |
| { | |
| "epoch": 2.9743589743589745, | |
| "grad_norm": 0.1321800947189331, | |
| "learning_rate": 7.2081471792911914e-09, | |
| "loss": 2.5686, | |
| "step": 116 | |
| }, | |
| { | |
| "epoch": 3.0, | |
| "grad_norm": 0.1928524672985077, | |
| "learning_rate": 1.8023616455731253e-09, | |
| "loss": 2.3083, | |
| "step": 117 | |
| }, | |
| { | |
| "epoch": 3.0, | |
| "eval_loss": 3.5, | |
| "eval_runtime": 10.2576, | |
| "eval_samples_per_second": 0.097, | |
| "eval_steps_per_second": 0.097, | |
| "step": 117 | |
| } | |
| ], | |
| "logging_steps": 1.0, | |
| "max_steps": 117, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 3, | |
| "save_steps": 0, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 2.029898068446413e+16, | |
| "train_batch_size": 1, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |