Instructions to use yeok/qwen-2.5-1.5B-instruct-sft-lora-countdown-deepseek-5k with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use yeok/qwen-2.5-1.5B-instruct-sft-lora-countdown-deepseek-5k with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("yeok/qwen-2.5-1.5B-instruct-sft-lora-countdown-deepseek-5k", device_map="auto") - Notebooks
- Google Colab
- Kaggle
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.9964699949571356, | |
| "eval_steps": 500, | |
| "global_step": 247, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.004034291477559254, | |
| "grad_norm": 0.10618364810943604, | |
| "learning_rate": 8.000000000000001e-06, | |
| "loss": 0.8761, | |
| "mean_token_accuracy": 0.7515642121434212, | |
| "step": 1 | |
| }, | |
| { | |
| "epoch": 0.020171457387796268, | |
| "grad_norm": 0.1208970844745636, | |
| "learning_rate": 4e-05, | |
| "loss": 0.9213, | |
| "mean_token_accuracy": 0.7374346107244492, | |
| "step": 5 | |
| }, | |
| { | |
| "epoch": 0.040342914775592535, | |
| "grad_norm": 0.09196259081363678, | |
| "learning_rate": 8e-05, | |
| "loss": 0.8914, | |
| "mean_token_accuracy": 0.7460352770984173, | |
| "step": 10 | |
| }, | |
| { | |
| "epoch": 0.060514372163388806, | |
| "grad_norm": 0.10933861136436462, | |
| "learning_rate": 0.00012, | |
| "loss": 0.8602, | |
| "mean_token_accuracy": 0.7503265753388405, | |
| "step": 15 | |
| }, | |
| { | |
| "epoch": 0.08068582955118507, | |
| "grad_norm": 0.11680758744478226, | |
| "learning_rate": 0.00016, | |
| "loss": 0.8439, | |
| "mean_token_accuracy": 0.7499923676252365, | |
| "step": 20 | |
| }, | |
| { | |
| "epoch": 0.10085728693898134, | |
| "grad_norm": 0.0951286181807518, | |
| "learning_rate": 0.0002, | |
| "loss": 0.7772, | |
| "mean_token_accuracy": 0.7655725114047527, | |
| "step": 25 | |
| }, | |
| { | |
| "epoch": 0.12102874432677761, | |
| "grad_norm": 0.08053945004940033, | |
| "learning_rate": 0.00019974977965945, | |
| "loss": 0.7238, | |
| "mean_token_accuracy": 0.7779346205294132, | |
| "step": 30 | |
| }, | |
| { | |
| "epoch": 0.14120020171457387, | |
| "grad_norm": 0.08780888468027115, | |
| "learning_rate": 0.00019900037084217637, | |
| "loss": 0.6863, | |
| "mean_token_accuracy": 0.7845150113105774, | |
| "step": 35 | |
| }, | |
| { | |
| "epoch": 0.16137165910237014, | |
| "grad_norm": 0.08427116274833679, | |
| "learning_rate": 0.00019775552389476864, | |
| "loss": 0.6315, | |
| "mean_token_accuracy": 0.7988936014473438, | |
| "step": 40 | |
| }, | |
| { | |
| "epoch": 0.1815431164901664, | |
| "grad_norm": 0.08954589813947678, | |
| "learning_rate": 0.00019602146853776894, | |
| "loss": 0.617, | |
| "mean_token_accuracy": 0.8002350099384785, | |
| "step": 45 | |
| }, | |
| { | |
| "epoch": 0.20171457387796268, | |
| "grad_norm": 0.09293227642774582, | |
| "learning_rate": 0.0001938068826896166, | |
| "loss": 0.5883, | |
| "mean_token_accuracy": 0.8079691044986248, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.22188603126575895, | |
| "grad_norm": 0.09921843558549881, | |
| "learning_rate": 0.0001911228490388136, | |
| "loss": 0.5817, | |
| "mean_token_accuracy": 0.8088557526469231, | |
| "step": 55 | |
| }, | |
| { | |
| "epoch": 0.24205748865355523, | |
| "grad_norm": 0.11710495501756668, | |
| "learning_rate": 0.00018798279958164295, | |
| "loss": 0.5737, | |
| "mean_token_accuracy": 0.809910261631012, | |
| "step": 60 | |
| }, | |
| { | |
| "epoch": 0.2622289460413515, | |
| "grad_norm": 0.10755322873592377, | |
| "learning_rate": 0.00018440244840299506, | |
| "loss": 0.567, | |
| "mean_token_accuracy": 0.8104779615998268, | |
| "step": 65 | |
| }, | |
| { | |
| "epoch": 0.28240040342914774, | |
| "grad_norm": 0.12730592489242554, | |
| "learning_rate": 0.00018039971303669407, | |
| "loss": 0.5546, | |
| "mean_token_accuracy": 0.8143862128257752, | |
| "step": 70 | |
| }, | |
| { | |
| "epoch": 0.30257186081694404, | |
| "grad_norm": 0.12127107381820679, | |
| "learning_rate": 0.00017599462479886974, | |
| "loss": 0.5522, | |
| "mean_token_accuracy": 0.8144258864223957, | |
| "step": 75 | |
| }, | |
| { | |
| "epoch": 0.3227433182047403, | |
| "grad_norm": 0.11108188331127167, | |
| "learning_rate": 0.00017120922854310257, | |
| "loss": 0.5305, | |
| "mean_token_accuracy": 0.8214686810970306, | |
| "step": 80 | |
| }, | |
| { | |
| "epoch": 0.3429147755925366, | |
| "grad_norm": 0.11264903098344803, | |
| "learning_rate": 0.00016606747233900815, | |
| "loss": 0.5398, | |
| "mean_token_accuracy": 0.8181388042867184, | |
| "step": 85 | |
| }, | |
| { | |
| "epoch": 0.3630862329803328, | |
| "grad_norm": 0.1377757340669632, | |
| "learning_rate": 0.00016059508762635482, | |
| "loss": 0.532, | |
| "mean_token_accuracy": 0.8195081010460854, | |
| "step": 90 | |
| }, | |
| { | |
| "epoch": 0.3832576903681291, | |
| "grad_norm": 0.12057465314865112, | |
| "learning_rate": 0.00015481946044447099, | |
| "loss": 0.5238, | |
| "mean_token_accuracy": 0.8215388789772987, | |
| "step": 95 | |
| }, | |
| { | |
| "epoch": 0.40342914775592537, | |
| "grad_norm": 0.11411843448877335, | |
| "learning_rate": 0.00014876949438136347, | |
| "loss": 0.5285, | |
| "mean_token_accuracy": 0.819812896847725, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.4236006051437216, | |
| "grad_norm": 0.12218334525823593, | |
| "learning_rate": 0.0001424754659284048, | |
| "loss": 0.5178, | |
| "mean_token_accuracy": 0.8232251830399037, | |
| "step": 105 | |
| }, | |
| { | |
| "epoch": 0.4437720625315179, | |
| "grad_norm": 0.12559793889522552, | |
| "learning_rate": 0.0001359688729644536, | |
| "loss": 0.5269, | |
| "mean_token_accuracy": 0.8201104834675789, | |
| "step": 110 | |
| }, | |
| { | |
| "epoch": 0.46394351991931415, | |
| "grad_norm": 0.14769871532917023, | |
| "learning_rate": 0.00012928227712765504, | |
| "loss": 0.5203, | |
| "mean_token_accuracy": 0.8221889816224575, | |
| "step": 115 | |
| }, | |
| { | |
| "epoch": 0.48411497730711045, | |
| "grad_norm": 0.1480661779642105, | |
| "learning_rate": 0.00012244914086375724, | |
| "loss": 0.523, | |
| "mean_token_accuracy": 0.8210825860500336, | |
| "step": 120 | |
| }, | |
| { | |
| "epoch": 0.5042864346949067, | |
| "grad_norm": 0.14564430713653564, | |
| "learning_rate": 0.00011550365996641979, | |
| "loss": 0.5074, | |
| "mean_token_accuracy": 0.8264268651604653, | |
| "step": 125 | |
| }, | |
| { | |
| "epoch": 0.524457892082703, | |
| "grad_norm": 0.13270238041877747, | |
| "learning_rate": 0.00010848059244755093, | |
| "loss": 0.5092, | |
| "mean_token_accuracy": 0.8253403089940547, | |
| "step": 130 | |
| }, | |
| { | |
| "epoch": 0.5446293494704992, | |
| "grad_norm": 0.11992616951465607, | |
| "learning_rate": 0.00010141508459407623, | |
| "loss": 0.5012, | |
| "mean_token_accuracy": 0.8276812911033631, | |
| "step": 135 | |
| }, | |
| { | |
| "epoch": 0.5648008068582955, | |
| "grad_norm": 0.14025390148162842, | |
| "learning_rate": 9.434249508162076e-05, | |
| "loss": 0.5128, | |
| "mean_token_accuracy": 0.8234357766807079, | |
| "step": 140 | |
| }, | |
| { | |
| "epoch": 0.5849722642460918, | |
| "grad_norm": 0.11215393245220184, | |
| "learning_rate": 8.729821802531212e-05, | |
| "loss": 0.5133, | |
| "mean_token_accuracy": 0.8236173823475837, | |
| "step": 145 | |
| }, | |
| { | |
| "epoch": 0.6051437216338881, | |
| "grad_norm": 0.11705436557531357, | |
| "learning_rate": 8.031750585322947e-05, | |
| "loss": 0.5053, | |
| "mean_token_accuracy": 0.8257141962647438, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.6253151790216843, | |
| "grad_norm": 0.12025938928127289, | |
| "learning_rate": 7.343529288891239e-05, | |
| "loss": 0.4955, | |
| "mean_token_accuracy": 0.8292319044470787, | |
| "step": 155 | |
| }, | |
| { | |
| "epoch": 0.6454866364094806, | |
| "grad_norm": 0.11226729303598404, | |
| "learning_rate": 6.668602052579424e-05, | |
| "loss": 0.4939, | |
| "mean_token_accuracy": 0.8295842953026294, | |
| "step": 160 | |
| }, | |
| { | |
| "epoch": 0.6656580937972768, | |
| "grad_norm": 0.10537487268447876, | |
| "learning_rate": 6.010346486845837e-05, | |
| "loss": 0.4907, | |
| "mean_token_accuracy": 0.8306830637156963, | |
| "step": 165 | |
| }, | |
| { | |
| "epoch": 0.6858295511850732, | |
| "grad_norm": 0.1326971799135208, | |
| "learning_rate": 5.372056770327013e-05, | |
| "loss": 0.4994, | |
| "mean_token_accuracy": 0.8274142310023308, | |
| "step": 170 | |
| }, | |
| { | |
| "epoch": 0.7060010085728694, | |
| "grad_norm": 0.0996076837182045, | |
| "learning_rate": 4.756927164427685e-05, | |
| "loss": 0.4973, | |
| "mean_token_accuracy": 0.8285726100206375, | |
| "step": 175 | |
| }, | |
| { | |
| "epoch": 0.7261724659606656, | |
| "grad_norm": 0.10915440320968628, | |
| "learning_rate": 4.168036027937267e-05, | |
| "loss": 0.4987, | |
| "mean_token_accuracy": 0.8276034623384476, | |
| "step": 180 | |
| }, | |
| { | |
| "epoch": 0.7463439233484619, | |
| "grad_norm": 0.10033515840768814, | |
| "learning_rate": 3.6083304116701535e-05, | |
| "loss": 0.4946, | |
| "mean_token_accuracy": 0.8296850137412548, | |
| "step": 185 | |
| }, | |
| { | |
| "epoch": 0.7665153807362582, | |
| "grad_norm": 0.10720834881067276, | |
| "learning_rate": 3.080611310224539e-05, | |
| "loss": 0.4952, | |
| "mean_token_accuracy": 0.8289067208766937, | |
| "step": 190 | |
| }, | |
| { | |
| "epoch": 0.7866868381240545, | |
| "grad_norm": 0.11353600770235062, | |
| "learning_rate": 2.587519644666001e-05, | |
| "loss": 0.4978, | |
| "mean_token_accuracy": 0.8272387310862541, | |
| "step": 195 | |
| }, | |
| { | |
| "epoch": 0.8068582955118507, | |
| "grad_norm": 0.09966889768838882, | |
| "learning_rate": 2.1315230462840985e-05, | |
| "loss": 0.4993, | |
| "mean_token_accuracy": 0.8278293192386628, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.827029752899647, | |
| "grad_norm": 0.10758362710475922, | |
| "learning_rate": 1.7149035075615794e-05, | |
| "loss": 0.5055, | |
| "mean_token_accuracy": 0.8247546002268791, | |
| "step": 205 | |
| }, | |
| { | |
| "epoch": 0.8472012102874432, | |
| "grad_norm": 0.09960389137268066, | |
| "learning_rate": 1.339745962155613e-05, | |
| "loss": 0.4822, | |
| "mean_token_accuracy": 0.8327905587852001, | |
| "step": 210 | |
| }, | |
| { | |
| "epoch": 0.8673726676752396, | |
| "grad_norm": 0.10244999825954437, | |
| "learning_rate": 1.0079278510416313e-05, | |
| "loss": 0.4894, | |
| "mean_token_accuracy": 0.8305548712611198, | |
| "step": 215 | |
| }, | |
| { | |
| "epoch": 0.8875441250630358, | |
| "grad_norm": 0.09608753025531769, | |
| "learning_rate": 7.211097270349066e-06, | |
| "loss": 0.4998, | |
| "mean_token_accuracy": 0.8276750639081001, | |
| "step": 220 | |
| }, | |
| { | |
| "epoch": 0.9077155824508321, | |
| "grad_norm": 0.09114421159029007, | |
| "learning_rate": 4.807269447087348e-06, | |
| "loss": 0.4985, | |
| "mean_token_accuracy": 0.827449332177639, | |
| "step": 225 | |
| }, | |
| { | |
| "epoch": 0.9278870398386283, | |
| "grad_norm": 0.09354618191719055, | |
| "learning_rate": 2.8798247729623806e-06, | |
| "loss": 0.493, | |
| "mean_token_accuracy": 0.8297582663595676, | |
| "step": 230 | |
| }, | |
| { | |
| "epoch": 0.9480584972264247, | |
| "grad_norm": 0.09092767536640167, | |
| "learning_rate": 1.4384089652291543e-06, | |
| "loss": 0.4887, | |
| "mean_token_accuracy": 0.8308509282767773, | |
| "step": 235 | |
| }, | |
| { | |
| "epoch": 0.9682299546142209, | |
| "grad_norm": 0.09328487515449524, | |
| "learning_rate": 4.902354549733978e-07, | |
| "loss": 0.4923, | |
| "mean_token_accuracy": 0.8295751377940178, | |
| "step": 240 | |
| }, | |
| { | |
| "epoch": 0.9884014120020171, | |
| "grad_norm": 0.09508884698152542, | |
| "learning_rate": 4.0049288167842705e-08, | |
| "loss": 0.5046, | |
| "mean_token_accuracy": 0.825819493830204, | |
| "step": 245 | |
| }, | |
| { | |
| "epoch": 0.9964699949571356, | |
| "mean_token_accuracy": 0.825654674321413, | |
| "step": 247, | |
| "total_flos": 2.5792410000818176e+17, | |
| "train_loss": 0.5621715438993353, | |
| "train_runtime": 8702.0895, | |
| "train_samples_per_second": 0.456, | |
| "train_steps_per_second": 0.028 | |
| } | |
| ], | |
| "logging_steps": 5, | |
| "max_steps": 247, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 1, | |
| "save_steps": 100, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 2.5792410000818176e+17, | |
| "train_batch_size": 1, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |