sidbaines commited on
Commit
d3974e5
·
verified ·
1 Parent(s): 9b33af4

Upload 4b control midtraining artifacts for 20260815T010433Z-ctl2

Browse files
Files changed (29) hide show
  1. .gitattributes +1 -0
  2. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/axolotl.rendered.yaml +53 -0
  3. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/final_checkpoint_files.json +70 -0
  4. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/mix_manifest.json +61 -0
  5. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/post_warmup_checkpoint_files.json +70 -0
  6. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/result.json +1806 -0
  7. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/step31_checkpoint_files.json +70 -0
  8. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/step62_checkpoint_files.json +70 -0
  9. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/step93_checkpoint_files.json +70 -0
  10. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/train.log +515 -0
  11. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/trainer_state.final.json +1770 -0
  12. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/training_plan.json +12 -0
  13. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/training_provenance.json +120 -0
  14. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/training_started.json +8 -0
  15. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/training_trace.jsonl +124 -0
  16. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/data/control_mix.jsonl +3 -0
  17. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/data/control_mix_manifest.json +61 -0
  18. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/data/control_source_order.jsonl +0 -0
  19. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/environment.json +30 -0
  20. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/git_head.json +12 -0
  21. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/git_status.json +11 -0
  22. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/nvidia_smi_full.json +9 -0
  23. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/nvidia_smi_query.json +10 -0
  24. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/pip_freeze.json +11 -0
  25. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/uname.json +9 -0
  26. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/events.jsonl +17 -0
  27. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/payload_files.json +114 -0
  28. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/run.log +0 -0
  29. runs/20260815T010433Z-ctl2/midtrain/control/artifacts/run_manifest.json +1873 -0
.gitattributes CHANGED
@@ -52,3 +52,4 @@ midtrain_4epoch/control/checkpoint-31/tokenizer.json filter=lfs diff=lfs merge=l
52
  midtrain_4epoch/control/checkpoint-62/tokenizer.json filter=lfs diff=lfs merge=lfs -text
53
  midtrain_4epoch/control/checkpoint-93/tokenizer.json filter=lfs diff=lfs merge=lfs -text
54
  midtrain_4epoch/control/checkpoint-124/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
52
  midtrain_4epoch/control/checkpoint-62/tokenizer.json filter=lfs diff=lfs merge=lfs -text
53
  midtrain_4epoch/control/checkpoint-93/tokenizer.json filter=lfs diff=lfs merge=lfs -text
54
  midtrain_4epoch/control/checkpoint-124/tokenizer.json filter=lfs diff=lfs merge=lfs -text
55
+ runs/20260815T010433Z-ctl2/midtrain/control/artifacts/data/control_mix.jsonl filter=lfs diff=lfs merge=lfs -text
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/axolotl.rendered.yaml ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ base_model: /root/.cache/huggingface/hub/models--unsloth--gemma-3-4b-pt/snapshots/52aba93981c6ad7712b030eb6dd496ece1d279d6
2
+ trust_remote_code: false
3
+ plugins:
4
+ - axolotl.integrations.liger.LigerPlugin
5
+ - scimt.train.axolotl_plugins.CheckpointSchedulePlugin
6
+ liger_fused_linear_cross_entropy: true
7
+ liger_rope: true
8
+ liger_rms_norm: true
9
+ liger_glu_activation: true
10
+ datasets:
11
+ - path: /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/mix_control
12
+ type: completion
13
+ field: text
14
+ dataset_prepared_path: /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/prepared
15
+ dataset_processes: 16
16
+ sequence_len: 8192
17
+ sample_packing: true
18
+ pad_to_sequence_len: true
19
+ bf16: true
20
+ tf32: true
21
+ flash_attention: true
22
+ gradient_checkpointing: true
23
+ micro_batch_size: 1
24
+ gradient_accumulation_steps: 16
25
+ num_epochs: 4
26
+ max_steps: 124
27
+ optimizer: adamw_torch_fused
28
+ learning_rate: 1.0e-05
29
+ weight_decay: 0.01
30
+ max_grad_norm: 1.0
31
+ lr_scheduler: cosine
32
+ cosine_min_lr_ratio: 0.1
33
+ warmup_ratio: 0.03
34
+ fsdp_version: 2
35
+ fsdp_config:
36
+ offload_params: false
37
+ cpu_ram_efficient_loading: true
38
+ auto_wrap_policy: TRANSFORMER_BASED_WRAP
39
+ transformer_layer_cls_to_wrap: Gemma3DecoderLayer
40
+ state_dict_type: FULL_STATE_DICT
41
+ reshard_after_forward: true
42
+ logging_steps: 1
43
+ save_strategy: 'no'
44
+ save_only_model: false
45
+ save_total_limit: 6
46
+ checkpoint_schedule:
47
+ - 4
48
+ - 31
49
+ - 62
50
+ - 93
51
+ - 124
52
+ seed: 314159
53
+ output_dir: /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/final_checkpoint_files.json ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "files": {
3
+ "checkpoint_hydration.json": {
4
+ "sha256": "013724bbb267ccb3c2ddc9da489fd8d47c9113c651d0522ac130df7590e7f5a5",
5
+ "size": 498
6
+ },
7
+ "config.json": {
8
+ "sha256": "cae7138fe10c5856878f5b6948848ea069a862e0dd1c5db8ac6b2aa095821ef3",
9
+ "size": 2908
10
+ },
11
+ "generation_config.json": {
12
+ "sha256": "18e36cbffc12cbb19db8878f1e4a83e722841b1c55cdf8e0eebb2782c74f2b5b",
13
+ "size": 209
14
+ },
15
+ "model.safetensors": {
16
+ "sha256": "5b1122a66ca15cde2a25652a32b100269770939c17a29ac66b7bb702197c90c8",
17
+ "size": 9942783064
18
+ },
19
+ "optimizer.bin": {
20
+ "sha256": "874f0a88cf3d709c0a0cb16c717c32588a441ff37d044a4629407ba9176dcec5",
21
+ "size": 15521485995
22
+ },
23
+ "preprocessor_config.json": {
24
+ "sha256": "f688d6bb20c5017601c4011de7ca656da8485b540b05013efdaf986c0fcc918d",
25
+ "size": 570
26
+ },
27
+ "processor_config.json": {
28
+ "sha256": "3ffd5f11778dc73e2b69b3c00535e4121e1badf7018136263cd17b5b34fbaa53",
29
+ "size": 70
30
+ },
31
+ "pytorch_model_fsdp.bin": {
32
+ "sha256": "d102be5799386729b1ff6a6cc0552b1ce9edfabf1c7754c77e5e166a0935c744",
33
+ "size": 9942970813
34
+ },
35
+ "rng_state_0.pth": {
36
+ "sha256": "f401510bc676ef14da9d2d9c8d2c35fb729b0db03f001f73e801d0ef57065917",
37
+ "size": 14917
38
+ },
39
+ "rng_state_1.pth": {
40
+ "sha256": "a46bd41bd8c2d675e0e03d42f8324244eed0f995ba38c14a166da85f8bc77ab7",
41
+ "size": 14917
42
+ },
43
+ "scheduler.pt": {
44
+ "sha256": "6ce074f8ca381e21781aac530131b23d6fa450fad781af3488ba3002fe1af9c4",
45
+ "size": 1465
46
+ },
47
+ "tokenizer.json": {
48
+ "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726",
49
+ "size": 33384567
50
+ },
51
+ "tokenizer_config.json": {
52
+ "sha256": "6cd6abcca758e52fb87f65912303f92e33fe71e32529443fc77fb334c3ab3429",
53
+ "size": 745
54
+ },
55
+ "tokens_state.json": {
56
+ "sha256": "2152c7242956d8e5d329642d8875a4f49d198521bd45e0ccc34f9f7fdec10dba",
57
+ "size": 42
58
+ },
59
+ "trainer_state.json": {
60
+ "sha256": "61cc4b722d7a64ae370c1e05f6dcab6129ac297b6970c90b45a494ab0d98c0a8",
61
+ "size": 54464
62
+ },
63
+ "training_args.bin": {
64
+ "sha256": "49b7e135b93c89482921bf6a0b88ccbe2ff2e542dd069b24abb6ffe8b61947b6",
65
+ "size": 7377
66
+ }
67
+ },
68
+ "root": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-124",
69
+ "tree_sha256": "273b74b00d6b1f866268318bbcc6badc3e5059aa1060a624840c4c1851af67db"
70
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/mix_manifest.json ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "arm": "control",
3
+ "docs": 11387,
4
+ "filler_manifest": {
5
+ "all_shards_order_sha256": "fbd27dcd107799286f3b24a208c617b50dc812c4fb7c95050b246486647ed2f3",
6
+ "budget": 8000000,
7
+ "docs": 11387,
8
+ "opened_shards": [
9
+ "data/ingredient1-olmocr_science_pdfs-high_quality-crime_law-2e13/crime_law_p030_shard_00004457.jsonl.zst",
10
+ "data/ingredient1-wiki_to_rcqa-part1/00009_f325.jsonl.zst",
11
+ "data/ingredient2-wiki_to_rcqa_part2/00046_f187.jsonl.zst",
12
+ "data/ingredient1-common_crawl-high-quality_19_science_math_and_technology/shard_00000558.jsonl.zst",
13
+ "data/ingredient2-dolmino-math/dolmino_math_tinyGSM-MIND_2students_tiny_gsm_inline_part117.000000.jsonl.jsonl.zst",
14
+ "data/ingredient2-wiki_to_rcqa_part1/00007_f60.jsonl.zst",
15
+ "data/ingredient2-olmocr_science_pdfs-high_quality-science_tech-length_2e12/science_tech_p050_shard_00001495.jsonl.zst",
16
+ "data/ingredient2-common_crawl-high-quality_20_sports_and_fitness/shard_00000448.jsonl.zst",
17
+ "data/ingredient1-dolmino-math/dolmino_math_mathcoder2-synthmath_m-a-p_Matrix_filtered-math_book_math.0003.0245.jsonl.jsonl.zst",
18
+ "data/ingredient2-common_crawl-high-quality_19_religion/shard_00000206.jsonl.zst",
19
+ "data/ingredient2-common_crawl-high-quality_19_health/shard_00000244.jsonl.zst",
20
+ "data/ingredient2-common_crawl-high-quality_20_social_life/shard_00000541.jsonl.zst",
21
+ "data/ingredient1-wiki_to_rcqa-part1/00020_f124.jsonl.zst",
22
+ "data/ingredient2-common_crawl-high-quality_20_fashion_and_beauty/shard_00000119.jsonl.zst",
23
+ "data/ingredient2-wiki_to_rcqa_part2/00060_f63.jsonl.zst",
24
+ "data/ingredient1-wiki_to_rcqa-part1/00014_f172.jsonl.zst",
25
+ "data/ingredient2-common_crawl-high-quality_19_entertainment/shard_00001008.jsonl.zst",
26
+ "data/ingredient2-common_crawl-high-quality_20_education_and_jobs/shard_00000154.jsonl.zst",
27
+ "data/ingredient2-general_reasoning_mix/train-00184-of-00278.jsonl.zst",
28
+ "data/ingredient1-wiki_to_rcqa-part1/00000_f313.jsonl.zst"
29
+ ],
30
+ "ordered_rows_sha256": "a852f50e44ec8814f74b15e0f9e0aebebb01a7141e11e2c9b027fe292164bb12",
31
+ "repo": "allenai/dolma3_dolmino_mix-100B-1125",
32
+ "revision": "f23aa129fda8335ba9760057bcc1f0c02f3d068b",
33
+ "seed": 42,
34
+ "shuffle_buffer": 10000,
35
+ "tokens": 8002382
36
+ },
37
+ "gate2_reference_digests": {
38
+ "jsonl_sha256": "de2c2c62e12ab0714ca3d7149d18865d8287b603893c52d082844cc8ac5a57e0",
39
+ "observed_ordered_rows_gate2_scheme": "5fc226629c0c253c179550aa362a56e89a4fa943d2747e1012264db87d50f1e6",
40
+ "ordered_rows_sha256": "a852f50e44ec8814f74b15e0f9e0aebebb01a7141e11e2c9b027fe292164bb12"
41
+ },
42
+ "jsonl_sha256": "de2c2c62e12ab0714ca3d7149d18865d8287b603893c52d082844cc8ac5a57e0",
43
+ "per_source": {
44
+ "filler": {
45
+ "docs": 11387,
46
+ "tokens": 8002382
47
+ }
48
+ },
49
+ "prefix_replay": {
50
+ "docs": 6085,
51
+ "ordered_rows_sha256": "819f35334706f6cd942ef3af31c927f3fcd986e3a3107372b461046d30ff02a9",
52
+ "tokens": 4001953
53
+ },
54
+ "seed": 42,
55
+ "source_order_sha256": "abdc46436ddad18f6aff9fada41c2c01df7f7501dfa9eecbe03ce49356e1324b",
56
+ "tokenizer": {
57
+ "repo": "unsloth/gemma-3-4b-pt",
58
+ "revision": "52aba93981c6ad7712b030eb6dd496ece1d279d6"
59
+ },
60
+ "total_tokens": 8002382
61
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/post_warmup_checkpoint_files.json ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "files": {
3
+ "checkpoint_hydration.json": {
4
+ "sha256": "013724bbb267ccb3c2ddc9da489fd8d47c9113c651d0522ac130df7590e7f5a5",
5
+ "size": 498
6
+ },
7
+ "config.json": {
8
+ "sha256": "cae7138fe10c5856878f5b6948848ea069a862e0dd1c5db8ac6b2aa095821ef3",
9
+ "size": 2908
10
+ },
11
+ "generation_config.json": {
12
+ "sha256": "18e36cbffc12cbb19db8878f1e4a83e722841b1c55cdf8e0eebb2782c74f2b5b",
13
+ "size": 209
14
+ },
15
+ "model.safetensors": {
16
+ "sha256": "08bd44005c38e419708f6553cc4438d5cc6cea78d45f84406574c9a79c864b0c",
17
+ "size": 9942783064
18
+ },
19
+ "optimizer.bin": {
20
+ "sha256": "3b1f67916d0c62d034d558128d90d6e5d79f68736da78f67c202e1385adddc52",
21
+ "size": 15521485995
22
+ },
23
+ "preprocessor_config.json": {
24
+ "sha256": "f688d6bb20c5017601c4011de7ca656da8485b540b05013efdaf986c0fcc918d",
25
+ "size": 570
26
+ },
27
+ "processor_config.json": {
28
+ "sha256": "3ffd5f11778dc73e2b69b3c00535e4121e1badf7018136263cd17b5b34fbaa53",
29
+ "size": 70
30
+ },
31
+ "pytorch_model_fsdp.bin": {
32
+ "sha256": "bb3a582cde25aca94fe7de7543cf30b822503b1ba720ce1b384c5429b8819044",
33
+ "size": 9942970813
34
+ },
35
+ "rng_state_0.pth": {
36
+ "sha256": "d6ae654ed628e46a69a3185a2008a760ecb6fa76a6ed8ea36091fbf51572a9dc",
37
+ "size": 14917
38
+ },
39
+ "rng_state_1.pth": {
40
+ "sha256": "aca90864f21237f3b2ef7719c284655f031e849a610ff91881a6d9e258037405",
41
+ "size": 14917
42
+ },
43
+ "scheduler.pt": {
44
+ "sha256": "38c65cfa85ef7ad1327f57a033ff1edbdf6287abd82c355b083790fd55ccdf73",
45
+ "size": 1465
46
+ },
47
+ "tokenizer.json": {
48
+ "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726",
49
+ "size": 33384567
50
+ },
51
+ "tokenizer_config.json": {
52
+ "sha256": "6cd6abcca758e52fb87f65912303f92e33fe71e32529443fc77fb334c3ab3429",
53
+ "size": 745
54
+ },
55
+ "tokens_state.json": {
56
+ "sha256": "25bfaf5769fe53cee4f7a251a821eaa1a028be4e5aeb6b722a03d9c1d46b2f8a",
57
+ "size": 40
58
+ },
59
+ "trainer_state.json": {
60
+ "sha256": "abe41f4f6d54d6167a6853763782fb6acc5b54b87b7319c3c679299314989b26",
61
+ "size": 2469
62
+ },
63
+ "training_args.bin": {
64
+ "sha256": "49b7e135b93c89482921bf6a0b88ccbe2ff2e542dd069b24abb6ffe8b61947b6",
65
+ "size": 7377
66
+ }
67
+ },
68
+ "root": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-4",
69
+ "tree_sha256": "51f1edcfa4ca103a92fdbbd53e4d1bdc0f5264eba3ba1aa60621cab20c603f31"
70
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/result.json ADDED
@@ -0,0 +1,1806 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "arm": "control",
3
+ "checkpoints": {
4
+ "final": {
5
+ "commit_oid": "9b33af4fd0103fe07e05732cbac4779bee2be97f",
6
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/9b33af4fd0103fe07e05732cbac4779bee2be97f",
7
+ "files": 16,
8
+ "local_name": "checkpoint-124",
9
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-124",
10
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
11
+ "tree_sha256": "273b74b00d6b1f866268318bbcc6badc3e5059aa1060a624840c4c1851af67db",
12
+ "verified_at": "2026-08-15T01:54:09+00:00"
13
+ },
14
+ "post_warmup": {
15
+ "commit_oid": "c0a5c82e20dd90872dd2e1ba0a06f92f98f08f03",
16
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/c0a5c82e20dd90872dd2e1ba0a06f92f98f08f03",
17
+ "files": 16,
18
+ "local_name": "checkpoint-4",
19
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-4",
20
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
21
+ "tree_sha256": "51f1edcfa4ca103a92fdbbd53e4d1bdc0f5264eba3ba1aa60621cab20c603f31",
22
+ "verified_at": "2026-08-15T01:43:20+00:00"
23
+ },
24
+ "step31": {
25
+ "commit_oid": "f225df41f35617335440fd45c89adf20e61ebb66",
26
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/f225df41f35617335440fd45c89adf20e61ebb66",
27
+ "files": 16,
28
+ "local_name": "checkpoint-31",
29
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-31",
30
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
31
+ "tree_sha256": "a34acde4e2a622a6ffbfd1912d8058fcf471be768537d690a769966a5f3f7fc7",
32
+ "verified_at": "2026-08-15T01:46:04+00:00"
33
+ },
34
+ "step62": {
35
+ "commit_oid": "3082a28661539aeac56366a2b203758242038186",
36
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/3082a28661539aeac56366a2b203758242038186",
37
+ "files": 16,
38
+ "local_name": "checkpoint-62",
39
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-62",
40
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
41
+ "tree_sha256": "cf7d6d4ebe35d33b9cd1d58a5701b993c146887dbc45b8fae8095dc555416c5f",
42
+ "verified_at": "2026-08-15T01:48:46+00:00"
43
+ },
44
+ "step93": {
45
+ "commit_oid": "fdc198123a2bc0720cbb48bfedd0ae393775fee4",
46
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/fdc198123a2bc0720cbb48bfedd0ae393775fee4",
47
+ "files": 16,
48
+ "local_name": "checkpoint-93",
49
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-93",
50
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
51
+ "tree_sha256": "010cb434ceba9690be1a5f05cce520e922a532129314029c4d032c0eeb8762ba",
52
+ "verified_at": "2026-08-15T01:51:27+00:00"
53
+ }
54
+ },
55
+ "completed_at": "2026-08-15T01:54:09+00:00",
56
+ "elapsed_seconds": 2761.146,
57
+ "expected_optimizer_steps": 124,
58
+ "loss": {
59
+ "first_loss": 1.6685791015625,
60
+ "global_step": 124,
61
+ "last_loss": 1.245849609375,
62
+ "log_history": [
63
+ {
64
+ "epoch": 0.0326530612244898,
65
+ "grad_norm": 3.96875,
66
+ "learning_rate": 0.0,
67
+ "loss": 1.6685791015625,
68
+ "memory/device_reserved (GiB)": 26.08,
69
+ "memory/max_active (GiB)": 20.03,
70
+ "memory/max_allocated (GiB)": 20.03,
71
+ "ppl": 5.30463,
72
+ "step": 1,
73
+ "tokens/total": 262144,
74
+ "tokens/train_per_sec_per_gpu": 471.78,
75
+ "tokens/trainable": 261922
76
+ },
77
+ {
78
+ "epoch": 0.0653061224489796,
79
+ "grad_norm": 4.21875,
80
+ "learning_rate": 3.3333333333333333e-06,
81
+ "loss": 1.5614013671875,
82
+ "memory/device_reserved (GiB)": 33.29,
83
+ "memory/max_active (GiB)": 27.26,
84
+ "memory/max_allocated (GiB)": 27.26,
85
+ "ppl": 4.76549,
86
+ "step": 2,
87
+ "tokens/total": 524288,
88
+ "tokens/train_per_sec_per_gpu": 597.28,
89
+ "tokens/trainable": 523573
90
+ },
91
+ {
92
+ "epoch": 0.09795918367346938,
93
+ "grad_norm": 3.40625,
94
+ "learning_rate": 6.666666666666667e-06,
95
+ "loss": 1.5543212890625,
96
+ "memory/device_reserved (GiB)": 33.29,
97
+ "memory/max_active (GiB)": 27.26,
98
+ "memory/max_allocated (GiB)": 27.26,
99
+ "ppl": 4.73187,
100
+ "step": 3,
101
+ "tokens/total": 786432,
102
+ "tokens/train_per_sec_per_gpu": 606.45,
103
+ "tokens/trainable": 785311
104
+ },
105
+ {
106
+ "epoch": 0.1306122448979592,
107
+ "grad_norm": 2.4375,
108
+ "learning_rate": 1e-05,
109
+ "loss": 1.575927734375,
110
+ "memory/device_reserved (GiB)": 33.29,
111
+ "memory/max_active (GiB)": 27.26,
112
+ "memory/max_allocated (GiB)": 27.26,
113
+ "ppl": 4.83523,
114
+ "step": 4,
115
+ "tokens/total": 1048576,
116
+ "tokens/train_per_sec_per_gpu": 609.43,
117
+ "tokens/trainable": 1047069
118
+ },
119
+ {
120
+ "epoch": 0.16326530612244897,
121
+ "grad_norm": 2.921875,
122
+ "learning_rate": 9.998483343865806e-06,
123
+ "loss": 1.529052734375,
124
+ "memory/device_reserved (GiB)": 33.29,
125
+ "memory/max_active (GiB)": 27.26,
126
+ "memory/max_allocated (GiB)": 27.26,
127
+ "ppl": 4.6138,
128
+ "step": 5,
129
+ "tokens/total": 1310720,
130
+ "tokens/train_per_sec_per_gpu": 604.52,
131
+ "tokens/trainable": 1308722
132
+ },
133
+ {
134
+ "epoch": 0.19591836734693877,
135
+ "grad_norm": 2.625,
136
+ "learning_rate": 9.993934397794704e-06,
137
+ "loss": 1.39306640625,
138
+ "memory/device_reserved (GiB)": 33.29,
139
+ "memory/max_active (GiB)": 27.26,
140
+ "memory/max_allocated (GiB)": 27.26,
141
+ "ppl": 4.02718,
142
+ "step": 6,
143
+ "tokens/total": 1572864,
144
+ "tokens/train_per_sec_per_gpu": 607.08,
145
+ "tokens/trainable": 1570345
146
+ },
147
+ {
148
+ "epoch": 0.22857142857142856,
149
+ "grad_norm": 1.734375,
150
+ "learning_rate": 9.986356228092011e-06,
151
+ "loss": 1.4849853515625,
152
+ "memory/device_reserved (GiB)": 33.29,
153
+ "memory/max_active (GiB)": 27.26,
154
+ "memory/max_allocated (GiB)": 27.26,
155
+ "ppl": 4.4149,
156
+ "step": 7,
157
+ "tokens/total": 1835008,
158
+ "tokens/train_per_sec_per_gpu": 605.44,
159
+ "tokens/trainable": 1832202
160
+ },
161
+ {
162
+ "epoch": 0.2612244897959184,
163
+ "grad_norm": 1.4296875,
164
+ "learning_rate": 9.975753942969978e-06,
165
+ "loss": 1.58740234375,
166
+ "memory/device_reserved (GiB)": 33.29,
167
+ "memory/max_active (GiB)": 27.26,
168
+ "memory/max_allocated (GiB)": 27.26,
169
+ "ppl": 4.89103,
170
+ "step": 8,
171
+ "tokens/total": 2097152,
172
+ "tokens/train_per_sec_per_gpu": 604.81,
173
+ "tokens/trainable": 2093962
174
+ },
175
+ {
176
+ "epoch": 0.2938775510204082,
177
+ "grad_norm": 1.46875,
178
+ "learning_rate": 9.962134689104498e-06,
179
+ "loss": 1.34619140625,
180
+ "memory/device_reserved (GiB)": 33.29,
181
+ "memory/max_active (GiB)": 27.26,
182
+ "memory/max_allocated (GiB)": 27.26,
183
+ "ppl": 3.84276,
184
+ "step": 9,
185
+ "tokens/total": 2359296,
186
+ "tokens/train_per_sec_per_gpu": 607.46,
187
+ "tokens/trainable": 2355679
188
+ },
189
+ {
190
+ "epoch": 0.32653061224489793,
191
+ "grad_norm": 1.25,
192
+ "learning_rate": 9.945507646817764e-06,
193
+ "loss": 1.47705078125,
194
+ "memory/device_reserved (GiB)": 33.29,
195
+ "memory/max_active (GiB)": 27.26,
196
+ "memory/max_allocated (GiB)": 27.26,
197
+ "ppl": 4.38001,
198
+ "step": 10,
199
+ "tokens/total": 2621440,
200
+ "tokens/train_per_sec_per_gpu": 606.8,
201
+ "tokens/trainable": 2617308
202
+ },
203
+ {
204
+ "epoch": 0.35918367346938773,
205
+ "grad_norm": 1.171875,
206
+ "learning_rate": 9.925884023890072e-06,
207
+ "loss": 1.4583740234375,
208
+ "memory/device_reserved (GiB)": 33.29,
209
+ "memory/max_active (GiB)": 27.26,
210
+ "memory/max_allocated (GiB)": 27.26,
211
+ "ppl": 4.29896,
212
+ "step": 11,
213
+ "tokens/total": 2883584,
214
+ "tokens/train_per_sec_per_gpu": 604.71,
215
+ "tokens/trainable": 2878962
216
+ },
217
+ {
218
+ "epoch": 0.39183673469387753,
219
+ "grad_norm": 1.203125,
220
+ "learning_rate": 9.903277048005017e-06,
221
+ "loss": 1.5108642578125,
222
+ "memory/device_reserved (GiB)": 33.29,
223
+ "memory/max_active (GiB)": 27.26,
224
+ "memory/max_allocated (GiB)": 27.26,
225
+ "ppl": 4.53064,
226
+ "step": 12,
227
+ "tokens/total": 3145728,
228
+ "tokens/train_per_sec_per_gpu": 603.72,
229
+ "tokens/trainable": 3140634
230
+ },
231
+ {
232
+ "epoch": 0.42448979591836733,
233
+ "grad_norm": 1.0234375,
234
+ "learning_rate": 9.877701957833113e-06,
235
+ "loss": 1.3990478515625,
236
+ "memory/device_reserved (GiB)": 33.29,
237
+ "memory/max_active (GiB)": 27.26,
238
+ "memory/max_allocated (GiB)": 27.26,
239
+ "ppl": 4.05134,
240
+ "step": 13,
241
+ "tokens/total": 3407872,
242
+ "tokens/train_per_sec_per_gpu": 604.61,
243
+ "tokens/trainable": 3402298
244
+ },
245
+ {
246
+ "epoch": 0.45714285714285713,
247
+ "grad_norm": 0.9375,
248
+ "learning_rate": 9.849175992759867e-06,
249
+ "loss": 1.32666015625,
250
+ "memory/device_reserved (GiB)": 33.29,
251
+ "memory/max_active (GiB)": 27.26,
252
+ "memory/max_allocated (GiB)": 27.26,
253
+ "ppl": 3.76844,
254
+ "step": 14,
255
+ "tokens/total": 3670016,
256
+ "tokens/train_per_sec_per_gpu": 605.87,
257
+ "tokens/trainable": 3663962
258
+ },
259
+ {
260
+ "epoch": 0.4897959183673469,
261
+ "grad_norm": 1.7734375,
262
+ "learning_rate": 9.81771838126524e-06,
263
+ "loss": 1.46875,
264
+ "memory/device_reserved (GiB)": 33.29,
265
+ "memory/max_active (GiB)": 27.26,
266
+ "memory/max_allocated (GiB)": 27.26,
267
+ "ppl": 4.3438,
268
+ "step": 15,
269
+ "tokens/total": 3932160,
270
+ "tokens/train_per_sec_per_gpu": 603.17,
271
+ "tokens/trainable": 3925695
272
+ },
273
+ {
274
+ "epoch": 0.5224489795918368,
275
+ "grad_norm": 1.7890625,
276
+ "learning_rate": 9.783350327962313e-06,
277
+ "loss": 1.409912109375,
278
+ "memory/device_reserved (GiB)": 33.29,
279
+ "memory/max_active (GiB)": 27.26,
280
+ "memory/max_allocated (GiB)": 27.26,
281
+ "ppl": 4.0956,
282
+ "step": 16,
283
+ "tokens/total": 4194304,
284
+ "tokens/train_per_sec_per_gpu": 603.95,
285
+ "tokens/trainable": 4187301
286
+ },
287
+ {
288
+ "epoch": 0.5551020408163265,
289
+ "grad_norm": 3.03125,
290
+ "learning_rate": 9.74609499930392e-06,
291
+ "loss": 1.401123046875,
292
+ "memory/device_reserved (GiB)": 33.29,
293
+ "memory/max_active (GiB)": 27.26,
294
+ "memory/max_allocated (GiB)": 27.26,
295
+ "ppl": 4.05976,
296
+ "step": 17,
297
+ "tokens/total": 4456448,
298
+ "tokens/train_per_sec_per_gpu": 603.72,
299
+ "tokens/trainable": 4449140
300
+ },
301
+ {
302
+ "epoch": 0.5877551020408164,
303
+ "grad_norm": 1.015625,
304
+ "learning_rate": 9.70597750796683e-06,
305
+ "loss": 1.476318359375,
306
+ "memory/device_reserved (GiB)": 33.29,
307
+ "memory/max_active (GiB)": 27.26,
308
+ "memory/max_allocated (GiB)": 27.26,
309
+ "ppl": 4.3768,
310
+ "step": 18,
311
+ "tokens/total": 4718592,
312
+ "tokens/train_per_sec_per_gpu": 604.17,
313
+ "tokens/trainable": 4710654
314
+ },
315
+ {
316
+ "epoch": 0.6204081632653061,
317
+ "grad_norm": 0.9453125,
318
+ "learning_rate": 9.663024895924078e-06,
319
+ "loss": 1.2203369140625,
320
+ "memory/device_reserved (GiB)": 33.29,
321
+ "memory/max_active (GiB)": 27.26,
322
+ "memory/max_allocated (GiB)": 27.26,
323
+ "ppl": 3.38833,
324
+ "step": 19,
325
+ "tokens/total": 4980736,
326
+ "tokens/train_per_sec_per_gpu": 605.11,
327
+ "tokens/trainable": 4972132
328
+ },
329
+ {
330
+ "epoch": 0.6530612244897959,
331
+ "grad_norm": 1.0078125,
332
+ "learning_rate": 9.61726611621679e-06,
333
+ "loss": 1.451171875,
334
+ "memory/device_reserved (GiB)": 33.29,
335
+ "memory/max_active (GiB)": 27.26,
336
+ "memory/max_allocated (GiB)": 27.26,
337
+ "ppl": 4.26811,
338
+ "step": 20,
339
+ "tokens/total": 5242880,
340
+ "tokens/train_per_sec_per_gpu": 603.4,
341
+ "tokens/trainable": 5233845
342
+ },
343
+ {
344
+ "epoch": 0.6857142857142857,
345
+ "grad_norm": 0.83203125,
346
+ "learning_rate": 9.568732013437827e-06,
347
+ "loss": 1.3369140625,
348
+ "memory/device_reserved (GiB)": 33.29,
349
+ "memory/max_active (GiB)": 27.26,
350
+ "memory/max_allocated (GiB)": 27.26,
351
+ "ppl": 3.80728,
352
+ "step": 21,
353
+ "tokens/total": 5505024,
354
+ "tokens/train_per_sec_per_gpu": 604.73,
355
+ "tokens/trainable": 5495423
356
+ },
357
+ {
358
+ "epoch": 0.7183673469387755,
359
+ "grad_norm": 0.953125,
360
+ "learning_rate": 9.517455302940388e-06,
361
+ "loss": 1.4034423828125,
362
+ "memory/device_reserved (GiB)": 33.29,
363
+ "memory/max_active (GiB)": 27.26,
364
+ "memory/max_allocated (GiB)": 27.26,
365
+ "ppl": 4.06918,
366
+ "step": 22,
367
+ "tokens/total": 5767168,
368
+ "tokens/train_per_sec_per_gpu": 605.55,
369
+ "tokens/trainable": 5756905
370
+ },
371
+ {
372
+ "epoch": 0.7510204081632653,
373
+ "grad_norm": 0.78125,
374
+ "learning_rate": 9.46347054878559e-06,
375
+ "loss": 1.3974609375,
376
+ "memory/device_reserved (GiB)": 33.29,
377
+ "memory/max_active (GiB)": 27.26,
378
+ "memory/max_allocated (GiB)": 27.26,
379
+ "ppl": 4.04492,
380
+ "step": 23,
381
+ "tokens/total": 6029312,
382
+ "tokens/train_per_sec_per_gpu": 601.68,
383
+ "tokens/trainable": 6018394
384
+ },
385
+ {
386
+ "epoch": 0.7836734693877551,
387
+ "grad_norm": 0.84375,
388
+ "learning_rate": 9.406814140443898e-06,
389
+ "loss": 1.446533203125,
390
+ "memory/device_reserved (GiB)": 33.29,
391
+ "memory/max_active (GiB)": 27.26,
392
+ "memory/max_allocated (GiB)": 27.26,
393
+ "ppl": 4.24836,
394
+ "step": 24,
395
+ "tokens/total": 6291456,
396
+ "tokens/train_per_sec_per_gpu": 602.65,
397
+ "tokens/trainable": 6279921
398
+ },
399
+ {
400
+ "epoch": 0.8163265306122449,
401
+ "grad_norm": 0.94140625,
402
+ "learning_rate": 9.347524268266092e-06,
403
+ "loss": 1.303955078125,
404
+ "memory/device_reserved (GiB)": 33.29,
405
+ "memory/max_active (GiB)": 27.26,
406
+ "memory/max_allocated (GiB)": 27.26,
407
+ "ppl": 3.68384,
408
+ "step": 25,
409
+ "tokens/total": 6553600,
410
+ "tokens/train_per_sec_per_gpu": 605.26,
411
+ "tokens/trainable": 6541489
412
+ },
413
+ {
414
+ "epoch": 0.8489795918367347,
415
+ "grad_norm": 0.9453125,
416
+ "learning_rate": 9.285640897740316e-06,
417
+ "loss": 1.4931640625,
418
+ "memory/device_reserved (GiB)": 33.29,
419
+ "memory/max_active (GiB)": 27.26,
420
+ "memory/max_allocated (GiB)": 27.26,
421
+ "ppl": 4.45116,
422
+ "step": 26,
423
+ "tokens/total": 6815744,
424
+ "tokens/train_per_sec_per_gpu": 603.79,
425
+ "tokens/trainable": 6803233
426
+ },
427
+ {
428
+ "epoch": 0.8816326530612245,
429
+ "grad_norm": 0.953125,
430
+ "learning_rate": 9.22120574255258e-06,
431
+ "loss": 1.2637939453125,
432
+ "memory/device_reserved (GiB)": 33.29,
433
+ "memory/max_active (GiB)": 27.26,
434
+ "memory/max_allocated (GiB)": 27.26,
435
+ "ppl": 3.53882,
436
+ "step": 27,
437
+ "tokens/total": 7077888,
438
+ "tokens/train_per_sec_per_gpu": 578.56,
439
+ "tokens/trainable": 7064799
440
+ },
441
+ {
442
+ "epoch": 0.9142857142857143,
443
+ "grad_norm": 0.75390625,
444
+ "learning_rate": 9.154262236468826e-06,
445
+ "loss": 1.222412109375,
446
+ "memory/device_reserved (GiB)": 33.29,
447
+ "memory/max_active (GiB)": 27.26,
448
+ "memory/max_allocated (GiB)": 27.26,
449
+ "ppl": 3.39537,
450
+ "step": 28,
451
+ "tokens/total": 7340032,
452
+ "tokens/train_per_sec_per_gpu": 605.17,
453
+ "tokens/trainable": 7326396
454
+ },
455
+ {
456
+ "epoch": 0.9469387755102041,
457
+ "grad_norm": 0.83984375,
458
+ "learning_rate": 9.084855504057562e-06,
459
+ "loss": 1.306884765625,
460
+ "memory/device_reserved (GiB)": 33.29,
461
+ "memory/max_active (GiB)": 27.26,
462
+ "memory/max_allocated (GiB)": 27.26,
463
+ "ppl": 3.69465,
464
+ "step": 29,
465
+ "tokens/total": 7602176,
466
+ "tokens/train_per_sec_per_gpu": 606.42,
467
+ "tokens/trainable": 7587940
468
+ },
469
+ {
470
+ "epoch": 0.9795918367346939,
471
+ "grad_norm": 0.83984375,
472
+ "learning_rate": 9.013032330272777e-06,
473
+ "loss": 1.3385009765625,
474
+ "memory/device_reserved (GiB)": 33.29,
475
+ "memory/max_active (GiB)": 27.26,
476
+ "memory/max_allocated (GiB)": 27.26,
477
+ "ppl": 3.81332,
478
+ "step": 30,
479
+ "tokens/total": 7864320,
480
+ "tokens/train_per_sec_per_gpu": 606.34,
481
+ "tokens/trainable": 7849389
482
+ },
483
+ {
484
+ "epoch": 1.0,
485
+ "grad_norm": 0.9375,
486
+ "learning_rate": 8.938841128917622e-06,
487
+ "loss": 1.3125,
488
+ "memory/device_reserved (GiB)": 33.29,
489
+ "memory/max_active (GiB)": 27.26,
490
+ "memory/max_allocated (GiB)": 27.26,
491
+ "ppl": 3.71545,
492
+ "step": 31,
493
+ "tokens/total": 8028160,
494
+ "tokens/train_per_sec_per_gpu": 912.97,
495
+ "tokens/trainable": 8012073
496
+ },
497
+ {
498
+ "epoch": 1.0326530612244897,
499
+ "grad_norm": 0.87109375,
500
+ "learning_rate": 8.86233191001016e-06,
501
+ "loss": 1.4200439453125,
502
+ "memory/device_reserved (GiB)": 33.29,
503
+ "memory/max_active (GiB)": 27.26,
504
+ "memory/max_allocated (GiB)": 27.26,
505
+ "ppl": 4.1373,
506
+ "step": 32,
507
+ "tokens/total": 8290304,
508
+ "tokens/train_per_sec_per_gpu": 584.93,
509
+ "tokens/trainable": 8273995
510
+ },
511
+ {
512
+ "epoch": 1.0653061224489795,
513
+ "grad_norm": 0.890625,
514
+ "learning_rate": 8.783556246073135e-06,
515
+ "loss": 1.3094482421875,
516
+ "memory/device_reserved (GiB)": 33.29,
517
+ "memory/max_active (GiB)": 27.26,
518
+ "memory/max_allocated (GiB)": 27.26,
519
+ "ppl": 3.70413,
520
+ "step": 33,
521
+ "tokens/total": 8552448,
522
+ "tokens/train_per_sec_per_gpu": 607.05,
523
+ "tokens/trainable": 8535646
524
+ },
525
+ {
526
+ "epoch": 1.0979591836734695,
527
+ "grad_norm": 0.9375,
528
+ "learning_rate": 8.702567237370521e-06,
529
+ "loss": 1.317138671875,
530
+ "memory/device_reserved (GiB)": 33.29,
531
+ "memory/max_active (GiB)": 27.26,
532
+ "memory/max_allocated (GiB)": 27.26,
533
+ "ppl": 3.73273,
534
+ "step": 34,
535
+ "tokens/total": 8814592,
536
+ "tokens/train_per_sec_per_gpu": 605.99,
537
+ "tokens/trainable": 8797384
538
+ },
539
+ {
540
+ "epoch": 1.1306122448979592,
541
+ "grad_norm": 0.80078125,
542
+ "learning_rate": 8.619419476114251e-06,
543
+ "loss": 1.3460693359375,
544
+ "memory/device_reserved (GiB)": 33.29,
545
+ "memory/max_active (GiB)": 27.26,
546
+ "memory/max_allocated (GiB)": 27.26,
547
+ "ppl": 3.84229,
548
+ "step": 35,
549
+ "tokens/total": 9076736,
550
+ "tokens/train_per_sec_per_gpu": 606.76,
551
+ "tokens/trainable": 9059142
552
+ },
553
+ {
554
+ "epoch": 1.163265306122449,
555
+ "grad_norm": 0.83984375,
556
+ "learning_rate": 8.534169009665282e-06,
557
+ "loss": 1.353759765625,
558
+ "memory/device_reserved (GiB)": 33.29,
559
+ "memory/max_active (GiB)": 27.26,
560
+ "memory/max_allocated (GiB)": 27.26,
561
+ "ppl": 3.87196,
562
+ "step": 36,
563
+ "tokens/total": 9338880,
564
+ "tokens/train_per_sec_per_gpu": 607.47,
565
+ "tokens/trainable": 9320795
566
+ },
567
+ {
568
+ "epoch": 1.1959183673469387,
569
+ "grad_norm": 0.80859375,
570
+ "learning_rate": 8.446873302753783e-06,
571
+ "loss": 1.2401123046875,
572
+ "memory/device_reserved (GiB)": 33.29,
573
+ "memory/max_active (GiB)": 27.26,
574
+ "memory/max_allocated (GiB)": 27.26,
575
+ "ppl": 3.456,
576
+ "step": 37,
577
+ "tokens/total": 9601024,
578
+ "tokens/train_per_sec_per_gpu": 605.46,
579
+ "tokens/trainable": 9582418
580
+ },
581
+ {
582
+ "epoch": 1.2285714285714286,
583
+ "grad_norm": 0.765625,
584
+ "learning_rate": 8.357591198743923e-06,
585
+ "loss": 1.343017578125,
586
+ "memory/device_reserved (GiB)": 33.29,
587
+ "memory/max_active (GiB)": 27.26,
588
+ "memory/max_allocated (GiB)": 27.26,
589
+ "ppl": 3.83059,
590
+ "step": 38,
591
+ "tokens/total": 9863168,
592
+ "tokens/train_per_sec_per_gpu": 603.72,
593
+ "tokens/trainable": 9844275
594
+ },
595
+ {
596
+ "epoch": 1.2612244897959184,
597
+ "grad_norm": 0.80078125,
598
+ "learning_rate": 8.266382879969356e-06,
599
+ "loss": 1.466552734375,
600
+ "memory/device_reserved (GiB)": 33.29,
601
+ "memory/max_active (GiB)": 27.26,
602
+ "memory/max_allocated (GiB)": 27.26,
603
+ "ppl": 4.33427,
604
+ "step": 39,
605
+ "tokens/total": 10125312,
606
+ "tokens/train_per_sec_per_gpu": 604.71,
607
+ "tokens/trainable": 10106035
608
+ },
609
+ {
610
+ "epoch": 1.2938775510204081,
611
+ "grad_norm": 0.7578125,
612
+ "learning_rate": 8.17330982716615e-06,
613
+ "loss": 1.2305908203125,
614
+ "memory/device_reserved (GiB)": 33.29,
615
+ "memory/max_active (GiB)": 27.26,
616
+ "memory/max_allocated (GiB)": 27.26,
617
+ "ppl": 3.42325,
618
+ "step": 40,
619
+ "tokens/total": 10387456,
620
+ "tokens/train_per_sec_per_gpu": 607.6,
621
+ "tokens/trainable": 10367752
622
+ },
623
+ {
624
+ "epoch": 1.3265306122448979,
625
+ "grad_norm": 0.875,
626
+ "learning_rate": 8.078434778030511e-06,
627
+ "loss": 1.3778076171875,
628
+ "memory/device_reserved (GiB)": 33.29,
629
+ "memory/max_active (GiB)": 27.26,
630
+ "memory/max_allocated (GiB)": 27.26,
631
+ "ppl": 3.9662,
632
+ "step": 41,
633
+ "tokens/total": 10649600,
634
+ "tokens/train_per_sec_per_gpu": 605.02,
635
+ "tokens/trainable": 10629381
636
+ },
637
+ {
638
+ "epoch": 1.3591836734693876,
639
+ "grad_norm": 0.765625,
640
+ "learning_rate": 7.981821684929218e-06,
641
+ "loss": 1.362060546875,
642
+ "memory/device_reserved (GiB)": 33.29,
643
+ "memory/max_active (GiB)": 27.26,
644
+ "memory/max_allocated (GiB)": 27.26,
645
+ "ppl": 3.90423,
646
+ "step": 42,
647
+ "tokens/total": 10911744,
648
+ "tokens/train_per_sec_per_gpu": 603.78,
649
+ "tokens/trainable": 10891035
650
+ },
651
+ {
652
+ "epoch": 1.3918367346938776,
653
+ "grad_norm": 0.8046875,
654
+ "learning_rate": 7.883535671791294e-06,
655
+ "loss": 1.420166015625,
656
+ "memory/device_reserved (GiB)": 33.29,
657
+ "memory/max_active (GiB)": 27.26,
658
+ "memory/max_allocated (GiB)": 27.26,
659
+ "ppl": 4.13781,
660
+ "step": 43,
661
+ "tokens/total": 11173888,
662
+ "tokens/train_per_sec_per_gpu": 604.13,
663
+ "tokens/trainable": 11152707
664
+ },
665
+ {
666
+ "epoch": 1.4244897959183673,
667
+ "grad_norm": 0.83984375,
668
+ "learning_rate": 7.783642990209951e-06,
669
+ "loss": 1.31005859375,
670
+ "memory/device_reserved (GiB)": 33.29,
671
+ "memory/max_active (GiB)": 27.26,
672
+ "memory/max_allocated (GiB)": 27.26,
673
+ "ppl": 3.70639,
674
+ "step": 44,
675
+ "tokens/total": 11436032,
676
+ "tokens/train_per_sec_per_gpu": 605.21,
677
+ "tokens/trainable": 11414371
678
+ },
679
+ {
680
+ "epoch": 1.457142857142857,
681
+ "grad_norm": 0.77734375,
682
+ "learning_rate": 7.682210974784426e-06,
683
+ "loss": 1.2432861328125,
684
+ "memory/device_reserved (GiB)": 33.29,
685
+ "memory/max_active (GiB)": 27.26,
686
+ "memory/max_allocated (GiB)": 27.26,
687
+ "ppl": 3.46699,
688
+ "step": 45,
689
+ "tokens/total": 11698176,
690
+ "tokens/train_per_sec_per_gpu": 605.6,
691
+ "tokens/trainable": 11676035
692
+ },
693
+ {
694
+ "epoch": 1.489795918367347,
695
+ "grad_norm": 0.85546875,
696
+ "learning_rate": 7.579307997731783e-06,
697
+ "loss": 1.4068603515625,
698
+ "memory/device_reserved (GiB)": 33.29,
699
+ "memory/max_active (GiB)": 27.26,
700
+ "memory/max_allocated (GiB)": 27.26,
701
+ "ppl": 4.08312,
702
+ "step": 46,
703
+ "tokens/total": 11960320,
704
+ "tokens/train_per_sec_per_gpu": 603.01,
705
+ "tokens/trainable": 11937768
706
+ },
707
+ {
708
+ "epoch": 1.5224489795918368,
709
+ "grad_norm": 0.83984375,
710
+ "learning_rate": 7.475003422799302e-06,
711
+ "loss": 1.3421630859375,
712
+ "memory/device_reserved (GiB)": 33.29,
713
+ "memory/max_active (GiB)": 27.26,
714
+ "memory/max_allocated (GiB)": 27.26,
715
+ "ppl": 3.82731,
716
+ "step": 47,
717
+ "tokens/total": 12222464,
718
+ "tokens/train_per_sec_per_gpu": 603.3,
719
+ "tokens/trainable": 12199374
720
+ },
721
+ {
722
+ "epoch": 1.5551020408163265,
723
+ "grad_norm": 0.8828125,
724
+ "learning_rate": 7.36936755850849e-06,
725
+ "loss": 1.3466796875,
726
+ "memory/device_reserved (GiB)": 33.29,
727
+ "memory/max_active (GiB)": 27.26,
728
+ "memory/max_allocated (GiB)": 27.26,
729
+ "ppl": 3.84464,
730
+ "step": 48,
731
+ "tokens/total": 12484608,
732
+ "tokens/train_per_sec_per_gpu": 605.0,
733
+ "tokens/trainable": 12461213
734
+ },
735
+ {
736
+ "epoch": 1.5877551020408163,
737
+ "grad_norm": 0.81640625,
738
+ "learning_rate": 7.2624716107622675e-06,
739
+ "loss": 1.404296875,
740
+ "memory/device_reserved (GiB)": 33.29,
741
+ "memory/max_active (GiB)": 27.26,
742
+ "memory/max_allocated (GiB)": 27.26,
743
+ "ppl": 4.07266,
744
+ "step": 49,
745
+ "tokens/total": 12746752,
746
+ "tokens/train_per_sec_per_gpu": 604.6,
747
+ "tokens/trainable": 12722727
748
+ },
749
+ {
750
+ "epoch": 1.620408163265306,
751
+ "grad_norm": 0.73046875,
752
+ "learning_rate": 7.154387634847241e-06,
753
+ "loss": 1.1488037109375,
754
+ "memory/device_reserved (GiB)": 33.29,
755
+ "memory/max_active (GiB)": 27.26,
756
+ "memory/max_allocated (GiB)": 27.26,
757
+ "ppl": 3.15442,
758
+ "step": 50,
759
+ "tokens/total": 13008896,
760
+ "tokens/train_per_sec_per_gpu": 605.31,
761
+ "tokens/trainable": 12984205
762
+ },
763
+ {
764
+ "epoch": 1.6530612244897958,
765
+ "grad_norm": 0.79296875,
766
+ "learning_rate": 7.045188486863449e-06,
767
+ "loss": 1.38818359375,
768
+ "memory/device_reserved (GiB)": 33.29,
769
+ "memory/max_active (GiB)": 27.26,
770
+ "memory/max_allocated (GiB)": 27.26,
771
+ "ppl": 4.00756,
772
+ "step": 51,
773
+ "tokens/total": 13271040,
774
+ "tokens/train_per_sec_per_gpu": 603.17,
775
+ "tokens/trainable": 13245918
776
+ },
777
+ {
778
+ "epoch": 1.6857142857142857,
779
+ "grad_norm": 0.75,
780
+ "learning_rate": 6.9349477746142846e-06,
781
+ "loss": 1.2738037109375,
782
+ "memory/device_reserved (GiB)": 33.29,
783
+ "memory/max_active (GiB)": 27.26,
784
+ "memory/max_allocated (GiB)": 27.26,
785
+ "ppl": 3.57442,
786
+ "step": 52,
787
+ "tokens/total": 13533184,
788
+ "tokens/train_per_sec_per_gpu": 606.57,
789
+ "tokens/trainable": 13507496
790
+ },
791
+ {
792
+ "epoch": 1.7183673469387755,
793
+ "grad_norm": 0.84375,
794
+ "learning_rate": 6.823739807989734e-06,
795
+ "loss": 1.3431396484375,
796
+ "memory/device_reserved (GiB)": 33.29,
797
+ "memory/max_active (GiB)": 27.26,
798
+ "memory/max_allocated (GiB)": 27.26,
799
+ "ppl": 3.83105,
800
+ "step": 53,
801
+ "tokens/total": 13795328,
802
+ "tokens/train_per_sec_per_gpu": 606.26,
803
+ "tokens/trainable": 13768978
804
+ },
805
+ {
806
+ "epoch": 1.7510204081632654,
807
+ "grad_norm": 0.7265625,
808
+ "learning_rate": 6.7116395488763565e-06,
809
+ "loss": 1.3397216796875,
810
+ "memory/device_reserved (GiB)": 33.29,
811
+ "memory/max_active (GiB)": 27.26,
812
+ "memory/max_allocated (GiB)": 27.26,
813
+ "ppl": 3.81798,
814
+ "step": 54,
815
+ "tokens/total": 14057472,
816
+ "tokens/train_per_sec_per_gpu": 602.75,
817
+ "tokens/trainable": 14030467
818
+ },
819
+ {
820
+ "epoch": 1.7836734693877552,
821
+ "grad_norm": 0.7265625,
822
+ "learning_rate": 6.598722560627761e-06,
823
+ "loss": 1.3912353515625,
824
+ "memory/device_reserved (GiB)": 33.29,
825
+ "memory/max_active (GiB)": 27.26,
826
+ "memory/max_allocated (GiB)": 27.26,
827
+ "ppl": 4.01981,
828
+ "step": 55,
829
+ "tokens/total": 14319616,
830
+ "tokens/train_per_sec_per_gpu": 571.54,
831
+ "tokens/trainable": 14291994
832
+ },
833
+ {
834
+ "epoch": 1.816326530612245,
835
+ "grad_norm": 0.8515625,
836
+ "learning_rate": 6.485064957129677e-06,
837
+ "loss": 1.2479248046875,
838
+ "memory/device_reserved (GiB)": 33.29,
839
+ "memory/max_active (GiB)": 27.26,
840
+ "memory/max_allocated (GiB)": 27.26,
841
+ "ppl": 3.48311,
842
+ "step": 56,
843
+ "tokens/total": 14581760,
844
+ "tokens/train_per_sec_per_gpu": 606.07,
845
+ "tokens/trainable": 14553562
846
+ },
847
+ {
848
+ "epoch": 1.8489795918367347,
849
+ "grad_norm": 0.82421875,
850
+ "learning_rate": 6.370743351493899e-06,
851
+ "loss": 1.4447021484375,
852
+ "memory/device_reserved (GiB)": 33.29,
853
+ "memory/max_active (GiB)": 27.26,
854
+ "memory/max_allocated (GiB)": 27.26,
855
+ "ppl": 4.24059,
856
+ "step": 57,
857
+ "tokens/total": 14843904,
858
+ "tokens/train_per_sec_per_gpu": 604.39,
859
+ "tokens/trainable": 14815306
860
+ },
861
+ {
862
+ "epoch": 1.8816326530612244,
863
+ "grad_norm": 0.75390625,
864
+ "learning_rate": 6.255834804415742e-06,
865
+ "loss": 1.2120361328125,
866
+ "memory/device_reserved (GiB)": 33.29,
867
+ "memory/max_active (GiB)": 27.26,
868
+ "memory/max_allocated (GiB)": 27.26,
869
+ "ppl": 3.36032,
870
+ "step": 58,
871
+ "tokens/total": 15106048,
872
+ "tokens/train_per_sec_per_gpu": 606.32,
873
+ "tokens/trainable": 15076872
874
+ },
875
+ {
876
+ "epoch": 1.9142857142857141,
877
+ "grad_norm": 0.83203125,
878
+ "learning_rate": 6.140416772229785e-06,
879
+ "loss": 1.175048828125,
880
+ "memory/device_reserved (GiB)": 33.29,
881
+ "memory/max_active (GiB)": 27.26,
882
+ "memory/max_allocated (GiB)": 27.26,
883
+ "ppl": 3.2383,
884
+ "step": 59,
885
+ "tokens/total": 15368192,
886
+ "tokens/train_per_sec_per_gpu": 604.24,
887
+ "tokens/trainable": 15338469
888
+ },
889
+ {
890
+ "epoch": 1.9469387755102041,
891
+ "grad_norm": 0.71875,
892
+ "learning_rate": 6.0245670546989165e-06,
893
+ "loss": 1.2623291015625,
894
+ "memory/device_reserved (GiB)": 33.29,
895
+ "memory/max_active (GiB)": 27.26,
896
+ "memory/max_allocated (GiB)": 27.26,
897
+ "ppl": 3.53364,
898
+ "step": 60,
899
+ "tokens/total": 15630336,
900
+ "tokens/train_per_sec_per_gpu": 606.62,
901
+ "tokens/trainable": 15600013
902
+ },
903
+ {
904
+ "epoch": 1.9795918367346939,
905
+ "grad_norm": 0.7578125,
906
+ "learning_rate": 5.908363742571915e-06,
907
+ "loss": 1.291015625,
908
+ "memory/device_reserved (GiB)": 33.29,
909
+ "memory/max_active (GiB)": 27.26,
910
+ "memory/max_allocated (GiB)": 27.26,
911
+ "ppl": 3.63648,
912
+ "step": 61,
913
+ "tokens/total": 15892480,
914
+ "tokens/train_per_sec_per_gpu": 606.98,
915
+ "tokens/trainable": 15861462
916
+ },
917
+ {
918
+ "epoch": 2.0,
919
+ "grad_norm": 1.0078125,
920
+ "learning_rate": 5.791885164944844e-06,
921
+ "loss": 1.26220703125,
922
+ "memory/device_reserved (GiB)": 33.29,
923
+ "memory/max_active (GiB)": 27.26,
924
+ "memory/max_allocated (GiB)": 27.26,
925
+ "ppl": 3.53321,
926
+ "step": 62,
927
+ "tokens/total": 16056320,
928
+ "tokens/train_per_sec_per_gpu": 920.2,
929
+ "tokens/trainable": 16024146
930
+ },
931
+ {
932
+ "epoch": 2.0326530612244897,
933
+ "grad_norm": 0.81640625,
934
+ "learning_rate": 5.67520983646182e-06,
935
+ "loss": 1.37841796875,
936
+ "memory/device_reserved (GiB)": 33.29,
937
+ "memory/max_active (GiB)": 27.26,
938
+ "memory/max_allocated (GiB)": 27.26,
939
+ "ppl": 3.96862,
940
+ "step": 63,
941
+ "tokens/total": 16318464,
942
+ "tokens/train_per_sec_per_gpu": 585.03,
943
+ "tokens/trainable": 16286068
944
+ },
945
+ {
946
+ "epoch": 2.0653061224489795,
947
+ "grad_norm": 1.109375,
948
+ "learning_rate": 5.5584164043906895e-06,
949
+ "loss": 1.271728515625,
950
+ "memory/device_reserved (GiB)": 33.29,
951
+ "memory/max_active (GiB)": 27.26,
952
+ "memory/max_allocated (GiB)": 27.26,
953
+ "ppl": 3.56701,
954
+ "step": 64,
955
+ "tokens/total": 16580608,
956
+ "tokens/train_per_sec_per_gpu": 607.21,
957
+ "tokens/trainable": 16547719
958
+ },
959
+ {
960
+ "epoch": 2.0979591836734692,
961
+ "grad_norm": 1.0390625,
962
+ "learning_rate": 5.441583595609312e-06,
963
+ "loss": 1.28076171875,
964
+ "memory/device_reserved (GiB)": 33.29,
965
+ "memory/max_active (GiB)": 27.26,
966
+ "memory/max_allocated (GiB)": 27.26,
967
+ "ppl": 3.59938,
968
+ "step": 65,
969
+ "tokens/total": 16842752,
970
+ "tokens/train_per_sec_per_gpu": 605.49,
971
+ "tokens/trainable": 16809456
972
+ },
973
+ {
974
+ "epoch": 2.130612244897959,
975
+ "grad_norm": 0.86328125,
976
+ "learning_rate": 5.324790163538181e-06,
977
+ "loss": 1.3126220703125,
978
+ "memory/device_reserved (GiB)": 33.29,
979
+ "memory/max_active (GiB)": 27.26,
980
+ "memory/max_allocated (GiB)": 27.26,
981
+ "ppl": 3.7159,
982
+ "step": 66,
983
+ "tokens/total": 17104896,
984
+ "tokens/train_per_sec_per_gpu": 606.91,
985
+ "tokens/trainable": 17071218
986
+ },
987
+ {
988
+ "epoch": 2.163265306122449,
989
+ "grad_norm": 0.84765625,
990
+ "learning_rate": 5.208114835055157e-06,
991
+ "loss": 1.32037353515625,
992
+ "memory/device_reserved (GiB)": 33.29,
993
+ "memory/max_active (GiB)": 27.26,
994
+ "memory/max_allocated (GiB)": 27.26,
995
+ "ppl": 3.74482,
996
+ "step": 67,
997
+ "tokens/total": 17367040,
998
+ "tokens/train_per_sec_per_gpu": 607.11,
999
+ "tokens/trainable": 17332876
1000
+ },
1001
+ {
1002
+ "epoch": 2.195918367346939,
1003
+ "grad_norm": 0.72265625,
1004
+ "learning_rate": 5.0916362574280864e-06,
1005
+ "loss": 1.2105712890625,
1006
+ "memory/device_reserved (GiB)": 33.29,
1007
+ "memory/max_active (GiB)": 27.26,
1008
+ "memory/max_allocated (GiB)": 27.26,
1009
+ "ppl": 3.3554,
1010
+ "step": 68,
1011
+ "tokens/total": 17629184,
1012
+ "tokens/train_per_sec_per_gpu": 605.15,
1013
+ "tokens/trainable": 17594500
1014
+ },
1015
+ {
1016
+ "epoch": 2.2285714285714286,
1017
+ "grad_norm": 0.8515625,
1018
+ "learning_rate": 4.975432945301085e-06,
1019
+ "loss": 1.3128662109375,
1020
+ "memory/device_reserved (GiB)": 33.29,
1021
+ "memory/max_active (GiB)": 27.26,
1022
+ "memory/max_allocated (GiB)": 27.26,
1023
+ "ppl": 3.71681,
1024
+ "step": 69,
1025
+ "tokens/total": 17891328,
1026
+ "tokens/train_per_sec_per_gpu": 603.11,
1027
+ "tokens/trainable": 17856356
1028
+ },
1029
+ {
1030
+ "epoch": 2.2612244897959184,
1031
+ "grad_norm": 0.78515625,
1032
+ "learning_rate": 4.859583227770218e-06,
1033
+ "loss": 1.4384765625,
1034
+ "memory/device_reserved (GiB)": 33.29,
1035
+ "memory/max_active (GiB)": 27.26,
1036
+ "memory/max_allocated (GiB)": 27.26,
1037
+ "ppl": 4.21427,
1038
+ "step": 70,
1039
+ "tokens/total": 18153472,
1040
+ "tokens/train_per_sec_per_gpu": 604.33,
1041
+ "tokens/trainable": 18118114
1042
+ },
1043
+ {
1044
+ "epoch": 2.293877551020408,
1045
+ "grad_norm": 0.80859375,
1046
+ "learning_rate": 4.744165195584258e-06,
1047
+ "loss": 1.203125,
1048
+ "memory/device_reserved (GiB)": 33.29,
1049
+ "memory/max_active (GiB)": 27.26,
1050
+ "memory/max_allocated (GiB)": 27.26,
1051
+ "ppl": 3.33051,
1052
+ "step": 71,
1053
+ "tokens/total": 18415616,
1054
+ "tokens/train_per_sec_per_gpu": 606.6,
1055
+ "tokens/trainable": 18379828
1056
+ },
1057
+ {
1058
+ "epoch": 2.326530612244898,
1059
+ "grad_norm": 0.82421875,
1060
+ "learning_rate": 4.6292566485061015e-06,
1061
+ "loss": 1.352294921875,
1062
+ "memory/device_reserved (GiB)": 33.29,
1063
+ "memory/max_active (GiB)": 27.26,
1064
+ "memory/max_allocated (GiB)": 27.26,
1065
+ "ppl": 3.86629,
1066
+ "step": 72,
1067
+ "tokens/total": 18677760,
1068
+ "tokens/train_per_sec_per_gpu": 606.39,
1069
+ "tokens/trainable": 18641452
1070
+ },
1071
+ {
1072
+ "epoch": 2.3591836734693876,
1073
+ "grad_norm": 0.78125,
1074
+ "learning_rate": 4.514935042870324e-06,
1075
+ "loss": 1.3382568359375,
1076
+ "memory/device_reserved (GiB)": 33.29,
1077
+ "memory/max_active (GiB)": 27.26,
1078
+ "memory/max_allocated (GiB)": 27.26,
1079
+ "ppl": 3.81239,
1080
+ "step": 73,
1081
+ "tokens/total": 18939904,
1082
+ "tokens/train_per_sec_per_gpu": 602.92,
1083
+ "tokens/trainable": 18903104
1084
+ },
1085
+ {
1086
+ "epoch": 2.3918367346938774,
1087
+ "grad_norm": 0.78125,
1088
+ "learning_rate": 4.40127743937224e-06,
1089
+ "loss": 1.396728515625,
1090
+ "memory/device_reserved (GiB)": 33.29,
1091
+ "memory/max_active (GiB)": 27.26,
1092
+ "memory/max_allocated (GiB)": 27.26,
1093
+ "ppl": 4.04196,
1094
+ "step": 74,
1095
+ "tokens/total": 19202048,
1096
+ "tokens/train_per_sec_per_gpu": 603.33,
1097
+ "tokens/trainable": 19164778
1098
+ },
1099
+ {
1100
+ "epoch": 2.424489795918367,
1101
+ "grad_norm": 0.71875,
1102
+ "learning_rate": 4.288360451123646e-06,
1103
+ "loss": 1.287841796875,
1104
+ "memory/device_reserved (GiB)": 33.29,
1105
+ "memory/max_active (GiB)": 27.26,
1106
+ "memory/max_allocated (GiB)": 27.26,
1107
+ "ppl": 3.62495,
1108
+ "step": 75,
1109
+ "tokens/total": 19464192,
1110
+ "tokens/train_per_sec_per_gpu": 603.18,
1111
+ "tokens/trainable": 19426440
1112
+ },
1113
+ {
1114
+ "epoch": 2.4571428571428573,
1115
+ "grad_norm": 0.81640625,
1116
+ "learning_rate": 4.1762601920102675e-06,
1117
+ "loss": 1.22216796875,
1118
+ "memory/device_reserved (GiB)": 33.29,
1119
+ "memory/max_active (GiB)": 27.26,
1120
+ "memory/max_allocated (GiB)": 27.26,
1121
+ "ppl": 3.39454,
1122
+ "step": 76,
1123
+ "tokens/total": 19726336,
1124
+ "tokens/train_per_sec_per_gpu": 606.31,
1125
+ "tokens/trainable": 19688104
1126
+ },
1127
+ {
1128
+ "epoch": 2.489795918367347,
1129
+ "grad_norm": 0.76171875,
1130
+ "learning_rate": 4.065052225385717e-06,
1131
+ "loss": 1.3858642578125,
1132
+ "memory/device_reserved (GiB)": 33.29,
1133
+ "memory/max_active (GiB)": 27.26,
1134
+ "memory/max_allocated (GiB)": 27.26,
1135
+ "ppl": 3.99828,
1136
+ "step": 77,
1137
+ "tokens/total": 19988480,
1138
+ "tokens/train_per_sec_per_gpu": 603.6,
1139
+ "tokens/trainable": 19949836
1140
+ },
1141
+ {
1142
+ "epoch": 2.522448979591837,
1143
+ "grad_norm": 0.74609375,
1144
+ "learning_rate": 3.954811513136554e-06,
1145
+ "loss": 1.32080078125,
1146
+ "memory/device_reserved (GiB)": 33.29,
1147
+ "memory/max_active (GiB)": 27.26,
1148
+ "memory/max_allocated (GiB)": 27.26,
1149
+ "ppl": 3.74642,
1150
+ "step": 78,
1151
+ "tokens/total": 20250624,
1152
+ "tokens/train_per_sec_per_gpu": 603.42,
1153
+ "tokens/trainable": 20211444
1154
+ },
1155
+ {
1156
+ "epoch": 2.5551020408163265,
1157
+ "grad_norm": 1.015625,
1158
+ "learning_rate": 3.84561236515276e-06,
1159
+ "loss": 1.3240966796875,
1160
+ "memory/device_reserved (GiB)": 33.29,
1161
+ "memory/max_active (GiB)": 27.26,
1162
+ "memory/max_allocated (GiB)": 27.26,
1163
+ "ppl": 3.75879,
1164
+ "step": 79,
1165
+ "tokens/total": 20512768,
1166
+ "tokens/train_per_sec_per_gpu": 603.23,
1167
+ "tokens/trainable": 20473284
1168
+ },
1169
+ {
1170
+ "epoch": 2.5877551020408163,
1171
+ "grad_norm": 0.703125,
1172
+ "learning_rate": 3.7375283892377344e-06,
1173
+ "loss": 1.385986328125,
1174
+ "memory/device_reserved (GiB)": 33.29,
1175
+ "memory/max_active (GiB)": 27.26,
1176
+ "memory/max_allocated (GiB)": 27.26,
1177
+ "ppl": 3.99877,
1178
+ "step": 80,
1179
+ "tokens/total": 20774912,
1180
+ "tokens/train_per_sec_per_gpu": 604.17,
1181
+ "tokens/trainable": 20734804
1182
+ },
1183
+ {
1184
+ "epoch": 2.620408163265306,
1185
+ "grad_norm": 0.7109375,
1186
+ "learning_rate": 3.630632441491512e-06,
1187
+ "loss": 1.130859375,
1188
+ "memory/device_reserved (GiB)": 33.29,
1189
+ "memory/max_active (GiB)": 27.26,
1190
+ "memory/max_allocated (GiB)": 27.26,
1191
+ "ppl": 3.09832,
1192
+ "step": 81,
1193
+ "tokens/total": 21037056,
1194
+ "tokens/train_per_sec_per_gpu": 606.12,
1195
+ "tokens/trainable": 20996280
1196
+ },
1197
+ {
1198
+ "epoch": 2.6530612244897958,
1199
+ "grad_norm": 0.85546875,
1200
+ "learning_rate": 3.5249965772007e-06,
1201
+ "loss": 1.3719482421875,
1202
+ "memory/device_reserved (GiB)": 33.29,
1203
+ "memory/max_active (GiB)": 27.26,
1204
+ "memory/max_allocated (GiB)": 27.26,
1205
+ "ppl": 3.94303,
1206
+ "step": 82,
1207
+ "tokens/total": 21299200,
1208
+ "tokens/train_per_sec_per_gpu": 603.84,
1209
+ "tokens/trainable": 21257992
1210
+ },
1211
+ {
1212
+ "epoch": 2.685714285714286,
1213
+ "grad_norm": 0.703125,
1214
+ "learning_rate": 3.4206920022682173e-06,
1215
+ "loss": 1.2578125,
1216
+ "memory/device_reserved (GiB)": 33.29,
1217
+ "memory/max_active (GiB)": 27.26,
1218
+ "memory/max_allocated (GiB)": 27.26,
1219
+ "ppl": 3.51772,
1220
+ "step": 83,
1221
+ "tokens/total": 21561344,
1222
+ "tokens/train_per_sec_per_gpu": 605.93,
1223
+ "tokens/trainable": 21519576
1224
+ },
1225
+ {
1226
+ "epoch": 2.7183673469387752,
1227
+ "grad_norm": 0.85546875,
1228
+ "learning_rate": 3.3177890252155755e-06,
1229
+ "loss": 1.3260498046875,
1230
+ "memory/device_reserved (GiB)": 33.29,
1231
+ "memory/max_active (GiB)": 27.26,
1232
+ "memory/max_allocated (GiB)": 27.26,
1233
+ "ppl": 3.76614,
1234
+ "step": 84,
1235
+ "tokens/total": 21823488,
1236
+ "tokens/train_per_sec_per_gpu": 604.99,
1237
+ "tokens/trainable": 21781056
1238
+ },
1239
+ {
1240
+ "epoch": 2.7510204081632654,
1241
+ "grad_norm": 0.8203125,
1242
+ "learning_rate": 3.2163570097900497e-06,
1243
+ "loss": 1.324951171875,
1244
+ "memory/device_reserved (GiB)": 33.29,
1245
+ "memory/max_active (GiB)": 27.26,
1246
+ "memory/max_allocated (GiB)": 27.26,
1247
+ "ppl": 3.762,
1248
+ "step": 85,
1249
+ "tokens/total": 22085632,
1250
+ "tokens/train_per_sec_per_gpu": 585.4,
1251
+ "tokens/trainable": 22042546
1252
+ },
1253
+ {
1254
+ "epoch": 2.783673469387755,
1255
+ "grad_norm": 0.73828125,
1256
+ "learning_rate": 3.116464328208708e-06,
1257
+ "loss": 1.3780517578125,
1258
+ "memory/device_reserved (GiB)": 33.29,
1259
+ "memory/max_active (GiB)": 27.26,
1260
+ "memory/max_allocated (GiB)": 27.26,
1261
+ "ppl": 3.96717,
1262
+ "step": 86,
1263
+ "tokens/total": 22347776,
1264
+ "tokens/train_per_sec_per_gpu": 586.69,
1265
+ "tokens/trainable": 22304072
1266
+ },
1267
+ {
1268
+ "epoch": 2.816326530612245,
1269
+ "grad_norm": 0.80859375,
1270
+ "learning_rate": 3.0181783150707827e-06,
1271
+ "loss": 1.23388671875,
1272
+ "memory/device_reserved (GiB)": 33.29,
1273
+ "memory/max_active (GiB)": 27.26,
1274
+ "memory/max_allocated (GiB)": 27.26,
1275
+ "ppl": 3.43455,
1276
+ "step": 87,
1277
+ "tokens/total": 22609920,
1278
+ "tokens/train_per_sec_per_gpu": 605.15,
1279
+ "tokens/trainable": 22565640
1280
+ },
1281
+ {
1282
+ "epoch": 2.8489795918367347,
1283
+ "grad_norm": 0.79296875,
1284
+ "learning_rate": 2.921565221969492e-06,
1285
+ "loss": 1.430908203125,
1286
+ "memory/device_reserved (GiB)": 33.29,
1287
+ "memory/max_active (GiB)": 27.26,
1288
+ "memory/max_allocated (GiB)": 27.26,
1289
+ "ppl": 4.1825,
1290
+ "step": 88,
1291
+ "tokens/total": 22872064,
1292
+ "tokens/train_per_sec_per_gpu": 603.72,
1293
+ "tokens/trainable": 22827386
1294
+ },
1295
+ {
1296
+ "epoch": 2.8816326530612244,
1297
+ "grad_norm": 0.74609375,
1298
+ "learning_rate": 2.8266901728338526e-06,
1299
+ "loss": 1.19873046875,
1300
+ "memory/device_reserved (GiB)": 33.29,
1301
+ "memory/max_active (GiB)": 27.26,
1302
+ "memory/max_allocated (GiB)": 27.26,
1303
+ "ppl": 3.3159,
1304
+ "step": 89,
1305
+ "tokens/total": 23134208,
1306
+ "tokens/train_per_sec_per_gpu": 606.04,
1307
+ "tokens/trainable": 23088952
1308
+ },
1309
+ {
1310
+ "epoch": 2.914285714285714,
1311
+ "grad_norm": 0.6953125,
1312
+ "learning_rate": 2.7336171200306467e-06,
1313
+ "loss": 1.1632080078125,
1314
+ "memory/device_reserved (GiB)": 33.29,
1315
+ "memory/max_active (GiB)": 27.26,
1316
+ "memory/max_allocated (GiB)": 27.26,
1317
+ "ppl": 3.20018,
1318
+ "step": 90,
1319
+ "tokens/total": 23396352,
1320
+ "tokens/train_per_sec_per_gpu": 604.92,
1321
+ "tokens/trainable": 23350548
1322
+ },
1323
+ {
1324
+ "epoch": 2.946938775510204,
1325
+ "grad_norm": 0.875,
1326
+ "learning_rate": 2.6424088012560766e-06,
1327
+ "loss": 1.2506103515625,
1328
+ "memory/device_reserved (GiB)": 33.29,
1329
+ "memory/max_active (GiB)": 27.26,
1330
+ "memory/max_allocated (GiB)": 27.26,
1331
+ "ppl": 3.49247,
1332
+ "step": 91,
1333
+ "tokens/total": 23658496,
1334
+ "tokens/train_per_sec_per_gpu": 605.58,
1335
+ "tokens/trainable": 23612092
1336
+ },
1337
+ {
1338
+ "epoch": 2.979591836734694,
1339
+ "grad_norm": 0.859375,
1340
+ "learning_rate": 2.5531266972462176e-06,
1341
+ "loss": 1.2801513671875,
1342
+ "memory/device_reserved (GiB)": 33.29,
1343
+ "memory/max_active (GiB)": 27.26,
1344
+ "memory/max_allocated (GiB)": 27.26,
1345
+ "ppl": 3.59718,
1346
+ "step": 92,
1347
+ "tokens/total": 23920640,
1348
+ "tokens/train_per_sec_per_gpu": 606.4,
1349
+ "tokens/trainable": 23873538
1350
+ },
1351
+ {
1352
+ "epoch": 3.0,
1353
+ "grad_norm": 0.94140625,
1354
+ "learning_rate": 2.4658309903347196e-06,
1355
+ "loss": 1.248046875,
1356
+ "memory/device_reserved (GiB)": 33.29,
1357
+ "memory/max_active (GiB)": 27.26,
1358
+ "memory/max_allocated (GiB)": 27.26,
1359
+ "ppl": 3.48353,
1360
+ "step": 93,
1361
+ "tokens/total": 24084480,
1362
+ "tokens/train_per_sec_per_gpu": 922.28,
1363
+ "tokens/trainable": 24036224
1364
+ },
1365
+ {
1366
+ "epoch": 3.0326530612244897,
1367
+ "grad_norm": 0.7109375,
1368
+ "learning_rate": 2.380580523885751e-06,
1369
+ "loss": 1.3673095703125,
1370
+ "memory/device_reserved (GiB)": 33.29,
1371
+ "memory/max_active (GiB)": 27.26,
1372
+ "memory/max_allocated (GiB)": 27.26,
1373
+ "ppl": 3.92478,
1374
+ "step": 94,
1375
+ "tokens/total": 24346624,
1376
+ "tokens/train_per_sec_per_gpu": 586.15,
1377
+ "tokens/trainable": 24298144
1378
+ },
1379
+ {
1380
+ "epoch": 3.0653061224489795,
1381
+ "grad_norm": 0.67578125,
1382
+ "learning_rate": 2.29743276262948e-06,
1383
+ "loss": 1.2625732421875,
1384
+ "memory/device_reserved (GiB)": 33.29,
1385
+ "memory/max_active (GiB)": 27.26,
1386
+ "memory/max_allocated (GiB)": 27.26,
1387
+ "ppl": 3.5345,
1388
+ "step": 95,
1389
+ "tokens/total": 24608768,
1390
+ "tokens/train_per_sec_per_gpu": 607.42,
1391
+ "tokens/trainable": 24559796
1392
+ },
1393
+ {
1394
+ "epoch": 3.0979591836734692,
1395
+ "grad_norm": 0.84375,
1396
+ "learning_rate": 2.2164437539268652e-06,
1397
+ "loss": 1.2720947265625,
1398
+ "memory/device_reserved (GiB)": 33.29,
1399
+ "memory/max_active (GiB)": 27.26,
1400
+ "memory/max_allocated (GiB)": 27.26,
1401
+ "ppl": 3.56832,
1402
+ "step": 96,
1403
+ "tokens/total": 24870912,
1404
+ "tokens/train_per_sec_per_gpu": 605.03,
1405
+ "tokens/trainable": 24821538
1406
+ },
1407
+ {
1408
+ "epoch": 3.130612244897959,
1409
+ "grad_norm": 0.69921875,
1410
+ "learning_rate": 2.1376680899898415e-06,
1411
+ "loss": 1.3033447265625,
1412
+ "memory/device_reserved (GiB)": 33.29,
1413
+ "memory/max_active (GiB)": 27.26,
1414
+ "memory/max_allocated (GiB)": 27.26,
1415
+ "ppl": 3.68159,
1416
+ "step": 97,
1417
+ "tokens/total": 25133056,
1418
+ "tokens/train_per_sec_per_gpu": 607.5,
1419
+ "tokens/trainable": 25083298
1420
+ },
1421
+ {
1422
+ "epoch": 3.163265306122449,
1423
+ "grad_norm": 0.73828125,
1424
+ "learning_rate": 2.0611588710823797e-06,
1425
+ "loss": 1.31024169921875,
1426
+ "memory/device_reserved (GiB)": 33.29,
1427
+ "memory/max_active (GiB)": 27.26,
1428
+ "memory/max_allocated (GiB)": 27.26,
1429
+ "ppl": 3.70707,
1430
+ "step": 98,
1431
+ "tokens/total": 25395200,
1432
+ "tokens/train_per_sec_per_gpu": 607.32,
1433
+ "tokens/trainable": 25344956
1434
+ },
1435
+ {
1436
+ "epoch": 3.195918367346939,
1437
+ "grad_norm": 0.796875,
1438
+ "learning_rate": 1.986967669727224e-06,
1439
+ "loss": 1.20263671875,
1440
+ "memory/device_reserved (GiB)": 33.29,
1441
+ "memory/max_active (GiB)": 27.26,
1442
+ "memory/max_allocated (GiB)": 27.26,
1443
+ "ppl": 3.32888,
1444
+ "step": 99,
1445
+ "tokens/total": 25657344,
1446
+ "tokens/train_per_sec_per_gpu": 604.43,
1447
+ "tokens/trainable": 25606580
1448
+ },
1449
+ {
1450
+ "epoch": 3.2285714285714286,
1451
+ "grad_norm": 0.69921875,
1452
+ "learning_rate": 1.9151444959424383e-06,
1453
+ "loss": 1.3046875,
1454
+ "memory/device_reserved (GiB)": 33.29,
1455
+ "memory/max_active (GiB)": 27.26,
1456
+ "memory/max_allocated (GiB)": 27.26,
1457
+ "ppl": 3.68654,
1458
+ "step": 100,
1459
+ "tokens/total": 25919488,
1460
+ "tokens/train_per_sec_per_gpu": 603.08,
1461
+ "tokens/trainable": 25868436
1462
+ },
1463
+ {
1464
+ "epoch": 3.2612244897959184,
1465
+ "grad_norm": 0.78515625,
1466
+ "learning_rate": 1.8457377635311763e-06,
1467
+ "loss": 1.431396484375,
1468
+ "memory/device_reserved (GiB)": 33.29,
1469
+ "memory/max_active (GiB)": 27.26,
1470
+ "memory/max_allocated (GiB)": 27.26,
1471
+ "ppl": 4.18454,
1472
+ "step": 101,
1473
+ "tokens/total": 26181632,
1474
+ "tokens/train_per_sec_per_gpu": 603.9,
1475
+ "tokens/trainable": 26130194
1476
+ },
1477
+ {
1478
+ "epoch": 3.293877551020408,
1479
+ "grad_norm": 1.1171875,
1480
+ "learning_rate": 1.7787942574474215e-06,
1481
+ "loss": 1.19580078125,
1482
+ "memory/device_reserved (GiB)": 33.29,
1483
+ "memory/max_active (GiB)": 27.26,
1484
+ "memory/max_allocated (GiB)": 27.26,
1485
+ "ppl": 3.3062,
1486
+ "step": 102,
1487
+ "tokens/total": 26443776,
1488
+ "tokens/train_per_sec_per_gpu": 606.74,
1489
+ "tokens/trainable": 26391908
1490
+ },
1491
+ {
1492
+ "epoch": 3.326530612244898,
1493
+ "grad_norm": 0.70703125,
1494
+ "learning_rate": 1.7143591022596846e-06,
1495
+ "loss": 1.34716796875,
1496
+ "memory/device_reserved (GiB)": 33.29,
1497
+ "memory/max_active (GiB)": 27.26,
1498
+ "memory/max_allocated (GiB)": 27.26,
1499
+ "ppl": 3.84652,
1500
+ "step": 103,
1501
+ "tokens/total": 26705920,
1502
+ "tokens/train_per_sec_per_gpu": 606.45,
1503
+ "tokens/trainable": 26653532
1504
+ },
1505
+ {
1506
+ "epoch": 3.3591836734693876,
1507
+ "grad_norm": 0.71484375,
1508
+ "learning_rate": 1.6524757317339102e-06,
1509
+ "loss": 1.3314208984375,
1510
+ "memory/device_reserved (GiB)": 33.29,
1511
+ "memory/max_active (GiB)": 27.26,
1512
+ "memory/max_allocated (GiB)": 27.26,
1513
+ "ppl": 3.78642,
1514
+ "step": 104,
1515
+ "tokens/total": 26968064,
1516
+ "tokens/train_per_sec_per_gpu": 602.49,
1517
+ "tokens/trainable": 26915184
1518
+ },
1519
+ {
1520
+ "epoch": 3.3918367346938774,
1521
+ "grad_norm": 2.484375,
1522
+ "learning_rate": 1.593185859556103e-06,
1523
+ "loss": 1.3916015625,
1524
+ "memory/device_reserved (GiB)": 33.29,
1525
+ "memory/max_active (GiB)": 27.26,
1526
+ "memory/max_allocated (GiB)": 27.26,
1527
+ "ppl": 4.02129,
1528
+ "step": 105,
1529
+ "tokens/total": 27230208,
1530
+ "tokens/train_per_sec_per_gpu": 603.38,
1531
+ "tokens/trainable": 27176858
1532
+ },
1533
+ {
1534
+ "epoch": 3.424489795918367,
1535
+ "grad_norm": 0.71875,
1536
+ "learning_rate": 1.5365294512144114e-06,
1537
+ "loss": 1.282958984375,
1538
+ "memory/device_reserved (GiB)": 33.29,
1539
+ "memory/max_active (GiB)": 27.26,
1540
+ "memory/max_allocated (GiB)": 27.26,
1541
+ "ppl": 3.6073,
1542
+ "step": 106,
1543
+ "tokens/total": 27492352,
1544
+ "tokens/train_per_sec_per_gpu": 604.12,
1545
+ "tokens/trainable": 27438520
1546
+ },
1547
+ {
1548
+ "epoch": 3.4571428571428573,
1549
+ "grad_norm": 0.6640625,
1550
+ "learning_rate": 1.4825446970596136e-06,
1551
+ "loss": 1.2174072265625,
1552
+ "memory/device_reserved (GiB)": 33.29,
1553
+ "memory/max_active (GiB)": 27.26,
1554
+ "memory/max_allocated (GiB)": 27.26,
1555
+ "ppl": 3.37842,
1556
+ "step": 107,
1557
+ "tokens/total": 27754496,
1558
+ "tokens/train_per_sec_per_gpu": 606.17,
1559
+ "tokens/trainable": 27700184
1560
+ },
1561
+ {
1562
+ "epoch": 3.489795918367347,
1563
+ "grad_norm": 0.73046875,
1564
+ "learning_rate": 1.4312679865621742e-06,
1565
+ "loss": 1.380615234375,
1566
+ "memory/device_reserved (GiB)": 33.29,
1567
+ "memory/max_active (GiB)": 27.26,
1568
+ "memory/max_allocated (GiB)": 27.26,
1569
+ "ppl": 3.97735,
1570
+ "step": 108,
1571
+ "tokens/total": 28016640,
1572
+ "tokens/train_per_sec_per_gpu": 602.64,
1573
+ "tokens/trainable": 27961916
1574
+ },
1575
+ {
1576
+ "epoch": 3.522448979591837,
1577
+ "grad_norm": 0.6953125,
1578
+ "learning_rate": 1.382733883783211e-06,
1579
+ "loss": 1.3165283203125,
1580
+ "memory/device_reserved (GiB)": 33.29,
1581
+ "memory/max_active (GiB)": 27.26,
1582
+ "memory/max_allocated (GiB)": 27.26,
1583
+ "ppl": 3.73045,
1584
+ "step": 109,
1585
+ "tokens/total": 28278784,
1586
+ "tokens/train_per_sec_per_gpu": 603.31,
1587
+ "tokens/trainable": 28223524
1588
+ },
1589
+ {
1590
+ "epoch": 3.5551020408163265,
1591
+ "grad_norm": 0.71484375,
1592
+ "learning_rate": 1.3369751040759236e-06,
1593
+ "loss": 1.3199462890625,
1594
+ "memory/device_reserved (GiB)": 33.29,
1595
+ "memory/max_active (GiB)": 27.26,
1596
+ "memory/max_allocated (GiB)": 27.26,
1597
+ "ppl": 3.74322,
1598
+ "step": 110,
1599
+ "tokens/total": 28540928,
1600
+ "tokens/train_per_sec_per_gpu": 602.96,
1601
+ "tokens/trainable": 28485364
1602
+ },
1603
+ {
1604
+ "epoch": 3.5877551020408163,
1605
+ "grad_norm": 0.6953125,
1606
+ "learning_rate": 1.2940224920331707e-06,
1607
+ "loss": 1.3828125,
1608
+ "memory/device_reserved (GiB)": 33.29,
1609
+ "memory/max_active (GiB)": 27.26,
1610
+ "memory/max_allocated (GiB)": 27.26,
1611
+ "ppl": 3.9861,
1612
+ "step": 111,
1613
+ "tokens/total": 28803072,
1614
+ "tokens/train_per_sec_per_gpu": 603.97,
1615
+ "tokens/trainable": 28746884
1616
+ },
1617
+ {
1618
+ "epoch": 3.620408163265306,
1619
+ "grad_norm": 0.6640625,
1620
+ "learning_rate": 1.2539050006960814e-06,
1621
+ "loss": 1.1270751953125,
1622
+ "memory/device_reserved (GiB)": 33.29,
1623
+ "memory/max_active (GiB)": 27.26,
1624
+ "memory/max_allocated (GiB)": 27.26,
1625
+ "ppl": 3.08662,
1626
+ "step": 112,
1627
+ "tokens/total": 29065216,
1628
+ "tokens/train_per_sec_per_gpu": 605.96,
1629
+ "tokens/trainable": 29008360
1630
+ },
1631
+ {
1632
+ "epoch": 3.6530612244897958,
1633
+ "grad_norm": 0.7421875,
1634
+ "learning_rate": 1.2166496720376874e-06,
1635
+ "loss": 1.368408203125,
1636
+ "memory/device_reserved (GiB)": 33.29,
1637
+ "memory/max_active (GiB)": 27.26,
1638
+ "memory/max_allocated (GiB)": 27.26,
1639
+ "ppl": 3.92909,
1640
+ "step": 113,
1641
+ "tokens/total": 29327360,
1642
+ "tokens/train_per_sec_per_gpu": 602.33,
1643
+ "tokens/trainable": 29270072
1644
+ },
1645
+ {
1646
+ "epoch": 3.685714285714286,
1647
+ "grad_norm": 0.67578125,
1648
+ "learning_rate": 1.1822816187347625e-06,
1649
+ "loss": 1.2537841796875,
1650
+ "memory/device_reserved (GiB)": 33.29,
1651
+ "memory/max_active (GiB)": 27.26,
1652
+ "memory/max_allocated (GiB)": 27.26,
1653
+ "ppl": 3.50358,
1654
+ "step": 114,
1655
+ "tokens/total": 29589504,
1656
+ "tokens/train_per_sec_per_gpu": 606.12,
1657
+ "tokens/trainable": 29531656
1658
+ },
1659
+ {
1660
+ "epoch": 3.7183673469387752,
1661
+ "grad_norm": 0.703125,
1662
+ "learning_rate": 1.1508240072401336e-06,
1663
+ "loss": 1.3232421875,
1664
+ "memory/device_reserved (GiB)": 33.29,
1665
+ "memory/max_active (GiB)": 27.26,
1666
+ "memory/max_allocated (GiB)": 27.26,
1667
+ "ppl": 3.75558,
1668
+ "step": 115,
1669
+ "tokens/total": 29851648,
1670
+ "tokens/train_per_sec_per_gpu": 605.25,
1671
+ "tokens/trainable": 29793136
1672
+ },
1673
+ {
1674
+ "epoch": 3.7510204081632654,
1675
+ "grad_norm": 0.70703125,
1676
+ "learning_rate": 1.1222980421668874e-06,
1677
+ "loss": 1.3226318359375,
1678
+ "memory/device_reserved (GiB)": 33.29,
1679
+ "memory/max_active (GiB)": 27.26,
1680
+ "memory/max_allocated (GiB)": 27.26,
1681
+ "ppl": 3.75329,
1682
+ "step": 116,
1683
+ "tokens/total": 30113792,
1684
+ "tokens/train_per_sec_per_gpu": 570.96,
1685
+ "tokens/trainable": 30054626
1686
+ },
1687
+ {
1688
+ "epoch": 3.783673469387755,
1689
+ "grad_norm": 1.328125,
1690
+ "learning_rate": 1.0967229519949833e-06,
1691
+ "loss": 1.3746337890625,
1692
+ "memory/device_reserved (GiB)": 33.29,
1693
+ "memory/max_active (GiB)": 27.26,
1694
+ "memory/max_allocated (GiB)": 27.26,
1695
+ "ppl": 3.95363,
1696
+ "step": 117,
1697
+ "tokens/total": 30375936,
1698
+ "tokens/train_per_sec_per_gpu": 602.33,
1699
+ "tokens/trainable": 30316152
1700
+ },
1701
+ {
1702
+ "epoch": 3.816326530612245,
1703
+ "grad_norm": 0.71875,
1704
+ "learning_rate": 1.0741159761099294e-06,
1705
+ "loss": 1.231201171875,
1706
+ "memory/device_reserved (GiB)": 33.29,
1707
+ "memory/max_active (GiB)": 27.26,
1708
+ "memory/max_allocated (GiB)": 27.26,
1709
+ "ppl": 3.42534,
1710
+ "step": 118,
1711
+ "tokens/total": 30638080,
1712
+ "tokens/train_per_sec_per_gpu": 605.12,
1713
+ "tokens/trainable": 30577720
1714
+ },
1715
+ {
1716
+ "epoch": 3.8489795918367347,
1717
+ "grad_norm": 0.80078125,
1718
+ "learning_rate": 1.054492353182237e-06,
1719
+ "loss": 1.4288330078125,
1720
+ "memory/device_reserved (GiB)": 33.29,
1721
+ "memory/max_active (GiB)": 27.26,
1722
+ "memory/max_allocated (GiB)": 27.26,
1723
+ "ppl": 4.17383,
1724
+ "step": 119,
1725
+ "tokens/total": 30900224,
1726
+ "tokens/train_per_sec_per_gpu": 603.79,
1727
+ "tokens/trainable": 30839466
1728
+ },
1729
+ {
1730
+ "epoch": 3.8816326530612244,
1731
+ "grad_norm": 0.76171875,
1732
+ "learning_rate": 1.0378653108955017e-06,
1733
+ "loss": 1.197021484375,
1734
+ "memory/device_reserved (GiB)": 33.29,
1735
+ "memory/max_active (GiB)": 27.26,
1736
+ "memory/max_allocated (GiB)": 27.26,
1737
+ "ppl": 3.31024,
1738
+ "step": 120,
1739
+ "tokens/total": 31162368,
1740
+ "tokens/train_per_sec_per_gpu": 606.03,
1741
+ "tokens/trainable": 31101032
1742
+ },
1743
+ {
1744
+ "epoch": 3.914285714285714,
1745
+ "grad_norm": 0.671875,
1746
+ "learning_rate": 1.0242460570300241e-06,
1747
+ "loss": 1.1612548828125,
1748
+ "memory/device_reserved (GiB)": 33.29,
1749
+ "memory/max_active (GiB)": 27.26,
1750
+ "memory/max_allocated (GiB)": 27.26,
1751
+ "ppl": 3.19394,
1752
+ "step": 121,
1753
+ "tokens/total": 31424512,
1754
+ "tokens/train_per_sec_per_gpu": 605.01,
1755
+ "tokens/trainable": 31362628
1756
+ },
1757
+ {
1758
+ "epoch": 3.946938775510204,
1759
+ "grad_norm": 0.6796875,
1760
+ "learning_rate": 1.01364377190799e-06,
1761
+ "loss": 1.2498779296875,
1762
+ "memory/device_reserved (GiB)": 33.29,
1763
+ "memory/max_active (GiB)": 27.26,
1764
+ "memory/max_allocated (GiB)": 27.26,
1765
+ "ppl": 3.48992,
1766
+ "step": 122,
1767
+ "tokens/total": 31686656,
1768
+ "tokens/train_per_sec_per_gpu": 606.07,
1769
+ "tokens/trainable": 31624172
1770
+ },
1771
+ {
1772
+ "epoch": 3.979591836734694,
1773
+ "grad_norm": 0.69921875,
1774
+ "learning_rate": 1.0060656022052966e-06,
1775
+ "loss": 1.2774658203125,
1776
+ "memory/device_reserved (GiB)": 33.29,
1777
+ "memory/max_active (GiB)": 27.26,
1778
+ "memory/max_allocated (GiB)": 27.26,
1779
+ "ppl": 3.58754,
1780
+ "step": 123,
1781
+ "tokens/total": 31948800,
1782
+ "tokens/train_per_sec_per_gpu": 605.62,
1783
+ "tokens/trainable": 31885618
1784
+ },
1785
+ {
1786
+ "epoch": 4.0,
1787
+ "grad_norm": 0.86328125,
1788
+ "learning_rate": 1.0015166561341943e-06,
1789
+ "loss": 1.245849609375,
1790
+ "memory/device_reserved (GiB)": 33.29,
1791
+ "memory/max_active (GiB)": 27.26,
1792
+ "memory/max_allocated (GiB)": 27.26,
1793
+ "ppl": 3.47589,
1794
+ "step": 124,
1795
+ "tokens/total": 32112640,
1796
+ "tokens/train_per_sec_per_gpu": 920.1,
1797
+ "tokens/trainable": 32048304
1798
+ }
1799
+ ],
1800
+ "loss_count": 124,
1801
+ "max_loss": 1.6685791015625,
1802
+ "max_steps": 124,
1803
+ "min_loss": 1.1270751953125
1804
+ },
1805
+ "status": "complete"
1806
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/step31_checkpoint_files.json ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "files": {
3
+ "checkpoint_hydration.json": {
4
+ "sha256": "013724bbb267ccb3c2ddc9da489fd8d47c9113c651d0522ac130df7590e7f5a5",
5
+ "size": 498
6
+ },
7
+ "config.json": {
8
+ "sha256": "cae7138fe10c5856878f5b6948848ea069a862e0dd1c5db8ac6b2aa095821ef3",
9
+ "size": 2908
10
+ },
11
+ "generation_config.json": {
12
+ "sha256": "18e36cbffc12cbb19db8878f1e4a83e722841b1c55cdf8e0eebb2782c74f2b5b",
13
+ "size": 209
14
+ },
15
+ "model.safetensors": {
16
+ "sha256": "dd3c131636d0ae27fd627849a5a68ca3779b163b86883f18fcb248ec53168658",
17
+ "size": 9942783064
18
+ },
19
+ "optimizer.bin": {
20
+ "sha256": "45a3449595fb52c7b61b798c758b6f66d3fed9503f8c7b6a95edfacd65c0df90",
21
+ "size": 15521485995
22
+ },
23
+ "preprocessor_config.json": {
24
+ "sha256": "f688d6bb20c5017601c4011de7ca656da8485b540b05013efdaf986c0fcc918d",
25
+ "size": 570
26
+ },
27
+ "processor_config.json": {
28
+ "sha256": "3ffd5f11778dc73e2b69b3c00535e4121e1badf7018136263cd17b5b34fbaa53",
29
+ "size": 70
30
+ },
31
+ "pytorch_model_fsdp.bin": {
32
+ "sha256": "19bb2d1f1f2ab311aaec2f29eb6188695bdbdd1712353d39a68d6785b0958b33",
33
+ "size": 9942970813
34
+ },
35
+ "rng_state_0.pth": {
36
+ "sha256": "d6ae654ed628e46a69a3185a2008a760ecb6fa76a6ed8ea36091fbf51572a9dc",
37
+ "size": 14917
38
+ },
39
+ "rng_state_1.pth": {
40
+ "sha256": "aca90864f21237f3b2ef7719c284655f031e849a610ff91881a6d9e258037405",
41
+ "size": 14917
42
+ },
43
+ "scheduler.pt": {
44
+ "sha256": "974506cd65acc00842301cd5ef4c89783acac813a470db05904c014f45d2ebd5",
45
+ "size": 1465
46
+ },
47
+ "tokenizer.json": {
48
+ "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726",
49
+ "size": 33384567
50
+ },
51
+ "tokenizer_config.json": {
52
+ "sha256": "6cd6abcca758e52fb87f65912303f92e33fe71e32529443fc77fb334c3ab3429",
53
+ "size": 745
54
+ },
55
+ "tokens_state.json": {
56
+ "sha256": "0a7e08a3e1b5b885e38e35917bdd4423904d6abb8334594105e045075ae9547a",
57
+ "size": 40
58
+ },
59
+ "trainer_state.json": {
60
+ "sha256": "9805a596267b40cb17744e5d61eab1a25de5c22f697744fc92daba85d2ef163d",
61
+ "size": 14076
62
+ },
63
+ "training_args.bin": {
64
+ "sha256": "49b7e135b93c89482921bf6a0b88ccbe2ff2e542dd069b24abb6ffe8b61947b6",
65
+ "size": 7377
66
+ }
67
+ },
68
+ "root": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-31",
69
+ "tree_sha256": "a34acde4e2a622a6ffbfd1912d8058fcf471be768537d690a769966a5f3f7fc7"
70
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/step62_checkpoint_files.json ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "files": {
3
+ "checkpoint_hydration.json": {
4
+ "sha256": "013724bbb267ccb3c2ddc9da489fd8d47c9113c651d0522ac130df7590e7f5a5",
5
+ "size": 498
6
+ },
7
+ "config.json": {
8
+ "sha256": "cae7138fe10c5856878f5b6948848ea069a862e0dd1c5db8ac6b2aa095821ef3",
9
+ "size": 2908
10
+ },
11
+ "generation_config.json": {
12
+ "sha256": "18e36cbffc12cbb19db8878f1e4a83e722841b1c55cdf8e0eebb2782c74f2b5b",
13
+ "size": 209
14
+ },
15
+ "model.safetensors": {
16
+ "sha256": "3a8b720cbbe714586b36d9f84ebb2d4019e4c01a419b07cecb17496572747abb",
17
+ "size": 9942783064
18
+ },
19
+ "optimizer.bin": {
20
+ "sha256": "915f72905698794ab7dcc8491393d4e9447fe18a7d826d621960b072d2a6025a",
21
+ "size": 15521485995
22
+ },
23
+ "preprocessor_config.json": {
24
+ "sha256": "f688d6bb20c5017601c4011de7ca656da8485b540b05013efdaf986c0fcc918d",
25
+ "size": 570
26
+ },
27
+ "processor_config.json": {
28
+ "sha256": "3ffd5f11778dc73e2b69b3c00535e4121e1badf7018136263cd17b5b34fbaa53",
29
+ "size": 70
30
+ },
31
+ "pytorch_model_fsdp.bin": {
32
+ "sha256": "1040c96d961ca68300eb84da93889a1f5a8c4a744b042de380448d93799c2167",
33
+ "size": 9942970813
34
+ },
35
+ "rng_state_0.pth": {
36
+ "sha256": "76eeb4414f7c5c99c10487c996a4b97cbc076d9c557e466c209e0810dc91dfce",
37
+ "size": 14917
38
+ },
39
+ "rng_state_1.pth": {
40
+ "sha256": "e12a34119c58a3b7ba0f13d2fd9f5b4a7383dfdf5d2db77191ef8ca4de62c11f",
41
+ "size": 14917
42
+ },
43
+ "scheduler.pt": {
44
+ "sha256": "e78011ef1490d84f0bd6e204ed4a67a024825dfebe6c22983f85f6543f9b0beb",
45
+ "size": 1465
46
+ },
47
+ "tokenizer.json": {
48
+ "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726",
49
+ "size": 33384567
50
+ },
51
+ "tokenizer_config.json": {
52
+ "sha256": "6cd6abcca758e52fb87f65912303f92e33fe71e32529443fc77fb334c3ab3429",
53
+ "size": 745
54
+ },
55
+ "tokens_state.json": {
56
+ "sha256": "8f612b77e39ab9c586bdd943229da84aa2a307e3d17f97256762c913bcbe79b5",
57
+ "size": 42
58
+ },
59
+ "trainer_state.json": {
60
+ "sha256": "352547e3c44afa9fa92aaa5570e15fb17a1181c34d94965b906f518e3e2a7852",
61
+ "size": 27519
62
+ },
63
+ "training_args.bin": {
64
+ "sha256": "49b7e135b93c89482921bf6a0b88ccbe2ff2e542dd069b24abb6ffe8b61947b6",
65
+ "size": 7377
66
+ }
67
+ },
68
+ "root": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-62",
69
+ "tree_sha256": "cf7d6d4ebe35d33b9cd1d58a5701b993c146887dbc45b8fae8095dc555416c5f"
70
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/step93_checkpoint_files.json ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "files": {
3
+ "checkpoint_hydration.json": {
4
+ "sha256": "013724bbb267ccb3c2ddc9da489fd8d47c9113c651d0522ac130df7590e7f5a5",
5
+ "size": 498
6
+ },
7
+ "config.json": {
8
+ "sha256": "cae7138fe10c5856878f5b6948848ea069a862e0dd1c5db8ac6b2aa095821ef3",
9
+ "size": 2908
10
+ },
11
+ "generation_config.json": {
12
+ "sha256": "18e36cbffc12cbb19db8878f1e4a83e722841b1c55cdf8e0eebb2782c74f2b5b",
13
+ "size": 209
14
+ },
15
+ "model.safetensors": {
16
+ "sha256": "fb7d36606bd9c74c7c3e3f30f347fbdd49fba192b81505cf21dc97ad47275a33",
17
+ "size": 9942783064
18
+ },
19
+ "optimizer.bin": {
20
+ "sha256": "8d8d130561b44b06a3fec8cd1531d70b08c2578ce0bdd44cc217ad0d2c3db36d",
21
+ "size": 15521485995
22
+ },
23
+ "preprocessor_config.json": {
24
+ "sha256": "f688d6bb20c5017601c4011de7ca656da8485b540b05013efdaf986c0fcc918d",
25
+ "size": 570
26
+ },
27
+ "processor_config.json": {
28
+ "sha256": "3ffd5f11778dc73e2b69b3c00535e4121e1badf7018136263cd17b5b34fbaa53",
29
+ "size": 70
30
+ },
31
+ "pytorch_model_fsdp.bin": {
32
+ "sha256": "4aadf8c362e7e9583b35bd283bc6e05b2b0773dfa3b6aba3122b5e575fd280f0",
33
+ "size": 9942970813
34
+ },
35
+ "rng_state_0.pth": {
36
+ "sha256": "5ec230879ac026301443a52d93db3f26eaefc37a838116d18968f2decae92bfe",
37
+ "size": 14917
38
+ },
39
+ "rng_state_1.pth": {
40
+ "sha256": "20daee2211d4e7c28f96e3ac5f3b1bb5eb4e42289190dd5cf6d82b818ffb032c",
41
+ "size": 14917
42
+ },
43
+ "scheduler.pt": {
44
+ "sha256": "74c2a83dabeed514936b61068e71bc25b091ad91ddbc1106943e3b81b398605b",
45
+ "size": 1465
46
+ },
47
+ "tokenizer.json": {
48
+ "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726",
49
+ "size": 33384567
50
+ },
51
+ "tokenizer_config.json": {
52
+ "sha256": "6cd6abcca758e52fb87f65912303f92e33fe71e32529443fc77fb334c3ab3429",
53
+ "size": 745
54
+ },
55
+ "tokens_state.json": {
56
+ "sha256": "3de36e20a179b3da3117e12676f52dc061d60031d845591d355d01b111328a4a",
57
+ "size": 42
58
+ },
59
+ "trainer_state.json": {
60
+ "sha256": "60bf651ba95ad63a7eca179207e356e0c65a35a8dcc9c22a401e9f8677f4c394",
61
+ "size": 40964
62
+ },
63
+ "training_args.bin": {
64
+ "sha256": "49b7e135b93c89482921bf6a0b88ccbe2ff2e542dd069b24abb6ffe8b61947b6",
65
+ "size": 7377
66
+ }
67
+ },
68
+ "root": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-93",
69
+ "tree_sha256": "010cb434ceba9690be1a5f05cce520e922a532129314029c4d032c0eeb8762ba"
70
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/train.log ADDED
@@ -0,0 +1,515 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/124 [00:00<?, ?it/s][transformers] `use_return_dict` is deprecated! Use `return_dict` instead!
 
 
1
  1%| | 1/124 [00:19<39:52, 19.45s/it]
2
 
 
3
  1%| | 1/124 [00:19<39:52, 19.45s/it]
4
  2%|▏ | 2/124 [00:33<32:42, 16.08s/it]
5
 
 
6
  2%|▏ | 2/124 [00:33<32:42, 16.08s/it]
7
  2%|▏ | 3/124 [00:46<30:01, 14.89s/it]
8
 
 
9
  2%|▏ | 3/124 [00:46<30:01, 14.89s/it]
10
  3%|▎ | 4/124 [01:00<28:36, 14.30s/it]
11
 
 
12
  3%|▎ | 4/124 [01:00<28:36, 14.30s/it][2026-08-15 01:11:03,971] [INFO] [axolotl.core.trainers.base] Saving model checkpoint to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-4
 
 
 
 
13
  4%|▍ | 5/124 [01:46<51:01, 25.72s/it]
14
 
 
15
  4%|▍ | 5/124 [01:46<51:01, 25.72s/it]
16
  5%|▍ | 6/124 [01:59<42:23, 21.55s/it]
17
 
 
18
  5%|▍ | 6/124 [01:59<42:23, 21.55s/it]
19
  6%|▌ | 7/124 [02:12<36:54, 18.92s/it]
20
 
 
21
  6%|▌ | 7/124 [02:13<36:54, 18.92s/it]
22
  6%|▋ | 8/124 [02:26<33:15, 17.21s/it]
23
 
 
24
  6%|▋ | 8/124 [02:26<33:15, 17.21s/it]
25
  7%|▋ | 9/124 [02:39<30:43, 16.03s/it]
26
 
 
27
  7%|▋ | 9/124 [02:39<30:43, 16.03s/it]
28
  8%|▊ | 10/124 [02:53<28:58, 15.25s/it]
29
 
 
30
  8%|▊ | 10/124 [02:53<28:58, 15.25s/it]
31
  9%|▉ | 11/124 [03:07<27:44, 14.73s/it]
32
 
 
33
  9%|▉ | 11/124 [03:07<27:44, 14.73s/it]
34
  10%|▉ | 12/124 [03:20<26:48, 14.36s/it]
35
 
 
36
  10%|▉ | 12/124 [03:20<26:48, 14.36s/it]
37
  10%|█ | 13/124 [03:34<26:06, 14.11s/it]
38
 
 
39
  10%|█ | 13/124 [03:34<26:06, 14.11s/it]
40
  11%|█▏ | 14/124 [03:47<25:32, 13.93s/it]
41
 
 
42
  11%|█▏ | 14/124 [03:47<25:32, 13.93s/it]
43
  12%|█▏ | 15/124 [04:01<25:06, 13.82s/it]
44
 
 
45
  12%|█▏ | 15/124 [04:01<25:06, 13.82s/it]
46
  13%|█▎ | 16/124 [04:14<24:42, 13.73s/it]
47
 
 
48
  13%|█▎ | 16/124 [04:14<24:42, 13.73s/it]
49
  14%|█▎ | 17/124 [04:28<24:23, 13.68s/it]
50
 
 
51
  14%|█▎ | 17/124 [04:28<24:23, 13.68s/it]
52
  15%|█▍ | 18/124 [04:41<24:04, 13.63s/it]
53
 
 
54
  15%|█▍ | 18/124 [04:41<24:04, 13.63s/it]
55
  15%|█▌ | 19/124 [04:55<23:48, 13.60s/it]
56
 
 
57
  15%|█▌ | 19/124 [04:55<23:48, 13.60s/it]
58
  16%|█▌ | 20/124 [05:08<23:33, 13.59s/it]
59
 
 
60
  16%|█▌ | 20/124 [05:08<23:33, 13.59s/it]
61
  17%|█▋ | 21/124 [05:22<23:17, 13.56s/it]
62
 
 
63
  17%|█▋ | 21/124 [05:22<23:17, 13.56s/it]
64
  18%|█▊ | 22/124 [05:35<23:02, 13.55s/it]
65
 
 
66
  18%|█▊ | 22/124 [05:35<23:02, 13.55s/it]
67
  19%|█▊ | 23/124 [05:49<22:48, 13.55s/it]
68
 
 
69
  19%|█▊ | 23/124 [05:49<22:48, 13.55s/it]
70
  19%|█▉ | 24/124 [06:03<22:35, 13.56s/it]
71
 
 
72
  19%|█▉ | 24/124 [06:03<22:35, 13.56s/it]
73
  20%|██ | 25/124 [06:16<22:20, 13.54s/it]
74
 
 
75
  20%|██ | 25/124 [06:16<22:20, 13.54s/it]
76
  21%|██ | 26/124 [06:30<22:07, 13.54s/it]
77
 
 
78
  21%|██ | 26/124 [06:30<22:07, 13.54s/it]
79
  22%|██▏ | 27/124 [06:44<22:10, 13.72s/it]
80
 
 
81
  22%|██▏ | 27/124 [06:44<22:10, 13.72s/it]
82
  23%|██▎ | 28/124 [06:57<21:50, 13.66s/it]
83
 
 
84
  23%|██▎ | 28/124 [06:57<21:50, 13.66s/it]
85
  23%|██▎ | 29/124 [07:11<21:32, 13.61s/it]
86
 
 
87
  23%|██▎ | 29/124 [07:11<21:32, 13.61s/it]
88
  24%|██▍ | 30/124 [07:24<21:15, 13.57s/it]
89
 
 
90
  24%|██▍ | 30/124 [07:24<21:15, 13.57s/it]
91
  25%|██▌ | 31/124 [07:33<18:44, 12.09s/it]
92
 
 
93
  25%|██▌ | 31/124 [07:33<18:44, 12.09s/it][2026-08-15 01:17:36,195] [INFO] [axolotl.core.trainers.base] Saving model checkpoint to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-31
 
 
 
 
94
  26%|██▌ | 32/124 [08:18<33:48, 22.04s/it]
95
 
 
96
  26%|██▌ | 32/124 [08:18<33:48, 22.04s/it]
97
  27%|██▋ | 33/124 [08:32<29:31, 19.47s/it]
98
 
 
99
  27%|██▋ | 33/124 [08:32<29:31, 19.47s/it]
100
  27%|██▋ | 34/124 [08:45<26:30, 17.67s/it]
101
 
 
102
  27%|██▋ | 34/124 [08:45<26:30, 17.67s/it]
103
  28%|██▊ | 35/124 [08:58<24:20, 16.41s/it]
104
 
 
105
  28%|██▊ | 35/124 [08:58<24:20, 16.41s/it]
106
  29%|██▉ | 36/124 [09:12<22:47, 15.53s/it]
107
 
 
108
  29%|██▉ | 36/124 [09:12<22:47, 15.53s/it]
109
  30%|██▉ | 37/124 [09:25<21:38, 14.92s/it]
110
 
 
111
  30%|██▉ | 37/124 [09:25<21:38, 14.92s/it]
112
  31%|███ | 38/124 [09:39<20:47, 14.51s/it]
113
 
 
114
  31%|███ | 38/124 [09:39<20:47, 14.51s/it]
115
  31%|███▏ | 39/124 [09:53<20:08, 14.22s/it]
116
 
 
117
  31%|███▏ | 39/124 [09:53<20:08, 14.22s/it]
118
  32%|███▏ | 40/124 [10:06<19:34, 13.99s/it]
119
 
 
120
  32%|███▏ | 40/124 [10:06<19:34, 13.99s/it]
121
  33%|███▎ | 41/124 [10:20<19:09, 13.85s/it]
122
 
 
123
  33%|███▎ | 41/124 [10:20<19:09, 13.85s/it]
124
  34%|███▍ | 42/124 [10:33<18:48, 13.77s/it]
125
 
 
126
  34%|███▍ | 42/124 [10:33<18:48, 13.77s/it]
127
  35%|███▍ | 43/124 [10:47<18:28, 13.69s/it]
128
 
 
129
  35%|███▍ | 43/124 [10:47<18:28, 13.69s/it]
130
  35%|███▌ | 44/124 [11:00<18:10, 13.64s/it]
131
 
 
132
  35%|███▌ | 44/124 [11:00<18:10, 13.64s/it]
133
  36%|███▋ | 45/124 [11:14<17:54, 13.61s/it]
134
 
 
135
  36%|███▋ | 45/124 [11:14<17:54, 13.61s/it]
136
  37%|███▋ | 46/124 [11:27<17:40, 13.59s/it]
137
 
 
138
  37%|███▋ | 46/124 [11:27<17:40, 13.59s/it]
139
  38%|███▊ | 47/124 [11:41<17:25, 13.58s/it]
140
 
 
141
  38%|███▊ | 47/124 [11:41<17:25, 13.58s/it]
142
  39%|███▊ | 48/124 [11:54<17:10, 13.56s/it]
143
 
 
144
  39%|███▊ | 48/124 [11:54<17:10, 13.56s/it]
145
  40%|███▉ | 49/124 [12:08<16:55, 13.54s/it]
146
 
 
147
  40%|███▉ | 49/124 [12:08<16:55, 13.54s/it]
148
  40%|████ | 50/124 [12:21<16:42, 13.54s/it]
149
 
 
150
  40%|████ | 50/124 [12:21<16:42, 13.54s/it]
151
  41%|████ | 51/124 [12:35<16:29, 13.55s/it]
152
 
 
153
  41%|████ | 51/124 [12:35<16:29, 13.55s/it]
154
  42%|████▏ | 52/124 [12:48<16:13, 13.52s/it]
155
 
 
156
  42%|████▏ | 52/124 [12:48<16:13, 13.52s/it]
157
  43%|████▎ | 53/124 [13:02<15:59, 13.52s/it]
158
 
 
159
  43%|████▎ | 53/124 [13:02<15:59, 13.52s/it]
160
  44%|████▎ | 54/124 [13:15<15:46, 13.52s/it]
161
 
 
162
  44%|████▎ | 54/124 [13:15<15:46, 13.52s/it]
163
  44%|████▍ | 55/124 [13:30<15:49, 13.76s/it]
164
 
 
165
  44%|████▍ | 55/124 [13:30<15:49, 13.76s/it]
166
  45%|████▌ | 56/124 [13:43<15:29, 13.67s/it]
167
 
 
168
  45%|████▌ | 56/124 [13:43<15:29, 13.67s/it]
169
  46%|████▌ | 57/124 [13:57<15:13, 13.63s/it]
170
 
 
171
  46%|████▌ | 57/124 [13:57<15:13, 13.63s/it]
172
  47%|████▋ | 58/124 [14:10<14:56, 13.59s/it]
173
 
 
174
  47%|████▋ | 58/124 [14:10<14:56, 13.59s/it]
175
  48%|████▊ | 59/124 [14:24<14:42, 13.57s/it]
176
 
 
177
  48%|████▊ | 59/124 [14:24<14:42, 13.57s/it]
178
  48%|████▊ | 60/124 [14:37<14:27, 13.55s/it]
179
 
 
180
  48%|████▊ | 60/124 [14:37<14:27, 13.55s/it]
181
  49%|████▉ | 61/124 [14:51<14:11, 13.52s/it]
182
 
 
183
  49%|████▉ | 61/124 [14:51<14:11, 13.52s/it]
184
  50%|█████ | 62/124 [14:59<12:28, 12.07s/it]
185
 
 
186
  50%|█████ | 62/124 [14:59<12:28, 12.07s/it][2026-08-15 01:25:02,804] [INFO] [axolotl.core.trainers.base] Saving model checkpoint to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-62
 
 
 
 
187
  51%|█████ | 63/124 [15:46<22:50, 22.47s/it]
188
 
 
189
  51%|█████ | 63/124 [15:46<22:50, 22.47s/it]
190
  52%|█████▏ | 64/124 [16:00<19:45, 19.76s/it]
191
 
 
192
  52%|█████▏ | 64/124 [16:00<19:45, 19.76s/it]
193
  52%|█████▏ | 65/124 [16:13<17:35, 17.88s/it]
194
 
 
195
  52%|█████▏ | 65/124 [16:13<17:35, 17.88s/it]
196
  53%|█████▎ | 66/124 [16:27<16:00, 16.56s/it]
197
 
 
198
  53%|█████▎ | 66/124 [16:27<16:00, 16.56s/it]
199
  54%|█████▍ | 67/124 [16:40<14:51, 15.64s/it]
200
 
 
201
  54%|█████▍ | 67/124 [16:40<14:51, 15.64s/it]
202
  55%|█████▍ | 68/124 [16:54<13:59, 15.00s/it]
203
 
 
204
  55%|█████▍ | 68/124 [16:54<13:59, 15.00s/it]
205
  56%|█████▌ | 69/124 [17:07<13:21, 14.57s/it]
206
 
 
207
  56%|█████▌ | 69/124 [17:07<13:21, 14.57s/it]
208
  56%|█████▋ | 70/124 [17:21<12:49, 14.26s/it]
209
 
 
210
  56%|█████▋ | 70/124 [17:21<12:49, 14.26s/it]
211
  57%|█████▋ | 71/124 [17:34<12:23, 14.02s/it]
212
 
 
213
  57%|█████▋ | 71/124 [17:34<12:23, 14.02s/it]
214
  58%|█████▊ | 72/124 [17:48<12:01, 13.87s/it]
215
 
 
216
  58%|█████▊ | 72/124 [17:48<12:01, 13.87s/it]
217
  59%|█████▉ | 73/124 [18:01<11:42, 13.78s/it]
218
 
 
219
  59%|█████▉ | 73/124 [18:01<11:42, 13.78s/it]
220
  60%|█████▉ | 74/124 [18:15<11:25, 13.71s/it]
221
 
 
222
  60%|█���███▉ | 74/124 [18:15<11:25, 13.71s/it]
223
  60%|██████ | 75/124 [18:28<11:09, 13.66s/it]
224
 
 
225
  60%|██████ | 75/124 [18:28<11:09, 13.66s/it]
226
  61%|██████▏ | 76/124 [18:42<10:53, 13.62s/it]
227
 
 
228
  61%|██████▏ | 76/124 [18:42<10:53, 13.62s/it]
229
  62%|██████▏ | 77/124 [18:55<10:39, 13.60s/it]
230
 
 
231
  62%|██████▏ | 77/124 [18:55<10:39, 13.60s/it]
232
  63%|██████▎ | 78/124 [19:09<10:24, 13.58s/it]
233
 
 
234
  63%|██████▎ | 78/124 [19:09<10:24, 13.58s/it]
235
  64%|██████▎ | 79/124 [19:22<10:10, 13.58s/it]
236
 
 
237
  64%|██████▎ | 79/124 [19:22<10:10, 13.58s/it]
238
  65%|██████▍ | 80/124 [19:36<09:56, 13.56s/it]
239
 
 
240
  65%|██████▍ | 80/124 [19:36<09:56, 13.56s/it]
241
  65%|██████▌ | 81/124 [19:49<09:42, 13.55s/it]
242
 
 
243
  65%|██████▌ | 81/124 [19:49<09:42, 13.55s/it]
244
  66%|██████▌ | 82/124 [20:03<09:28, 13.55s/it]
245
 
 
246
  66%|██████▌ | 82/124 [20:03<09:28, 13.55s/it]
247
  67%|██████▋ | 83/124 [20:17<09:14, 13.53s/it]
248
 
 
249
  67%|██████▋ | 83/124 [20:17<09:14, 13.53s/it]
250
  68%|██████▊ | 84/124 [20:30<09:01, 13.53s/it]
251
 
 
252
  68%|██████▊ | 84/124 [20:30<09:01, 13.53s/it]
253
  69%|██████▊ | 85/124 [20:44<08:52, 13.65s/it]
254
 
 
255
  69%|██████▊ | 85/124 [20:44<08:52, 13.65s/it]
256
  69%|██████▉ | 86/124 [20:58<08:41, 13.74s/it]
257
 
 
258
  69%|██████▉ | 86/124 [20:58<08:41, 13.74s/it]
259
  70%|███████ | 87/124 [21:11<08:25, 13.66s/it]
260
 
 
261
  70%|███████ | 87/124 [21:11<08:25, 13.66s/it]
262
  71%|███████ | 88/124 [21:25<08:10, 13.63s/it]
263
 
 
264
  71%|███████ | 88/124 [21:25<08:10, 13.63s/it]
265
  72%|███████▏ | 89/124 [21:38<07:55, 13.59s/it]
266
 
 
267
  72%|███████▏ | 89/124 [21:38<07:55, 13.59s/it]
268
  73%|███████▎ | 90/124 [21:52<07:41, 13.57s/it]
269
 
 
270
  73%|███████▎ | 90/124 [21:52<07:41, 13.57s/it]
271
  73%|███████▎ | 91/124 [22:05<07:27, 13.55s/it]
272
 
 
273
  73%|███████▎ | 91/124 [22:05<07:27, 13.55s/it]
274
  74%|███████▍ | 92/124 [22:19<07:12, 13.53s/it]
275
 
 
276
  74%|███████▍ | 92/124 [22:19<07:12, 13.53s/it]
277
  75%|███████▌ | 93/124 [22:28<06:13, 12.06s/it]
278
 
 
279
  75%|███████▌ | 93/124 [22:28<06:13, 12.06s/it][2026-08-15 01:32:31,116] [INFO] [axolotl.core.trainers.base] Saving model checkpoint to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-93
 
 
 
 
280
  76%|███████▌ | 94/124 [23:15<11:17, 22.57s/it]
281
 
 
282
  76%|███████▌ | 94/124 [23:15<11:17, 22.57s/it]
283
  77%|███████▋ | 95/124 [23:28<09:35, 19.84s/it]
284
 
 
285
  77%|███████▋ | 95/124 [23:28<09:35, 19.84s/it]
286
  77%|███████▋ | 96/124 [23:42<08:22, 17.94s/it]
287
 
 
288
  77%|███████▋ | 96/124 [23:42<08:22, 17.94s/it]
289
  78%|███████▊ | 97/124 [23:55<07:27, 16.59s/it]
290
 
 
291
  78%|███████▊ | 97/124 [23:55<07:27, 16.59s/it]
292
  79%|███████▉ | 98/124 [24:09<06:47, 15.66s/it]
293
 
 
294
  79%|███████▉ | 98/124 [24:09<06:47, 15.66s/it]
295
  80%|███████▉ | 99/124 [24:22<06:15, 15.02s/it]
296
 
 
297
  80%|███████▉ | 99/124 [24:22<06:15, 15.02s/it]
298
  81%|████████ | 100/124 [24:36<05:49, 14.58s/it]
299
 
 
300
  81%|████████ | 100/124 [24:36<05:49, 14.58s/it]
301
  81%|████████▏ | 101/124 [24:49<05:28, 14.27s/it]
302
 
 
303
  81%|████████▏ | 101/124 [24:49<05:28, 14.27s/it]
304
  82%|████████▏ | 102/124 [25:03<05:08, 14.03s/it]
305
 
 
306
  82%|████████▏ | 102/124 [25:03<05:08, 14.03s/it]
307
  83%|████████▎ | 103/124 [25:16<04:51, 13.88s/it]
308
 
 
309
  83%|████████▎ | 103/124 [25:16<04:51, 13.88s/it]
310
  84%|████████▍ | 104/124 [25:30<04:35, 13.79s/it]
311
 
 
312
  84%|██████���█▍ | 104/124 [25:30<04:35, 13.79s/it]
313
  85%|████████▍ | 105/124 [25:43<04:20, 13.71s/it]
314
 
 
315
  85%|████████▍ | 105/124 [25:43<04:20, 13.71s/it]
316
  85%|████████▌ | 106/124 [25:57<04:05, 13.66s/it]
317
 
 
318
  85%|████████▌ | 106/124 [25:57<04:05, 13.66s/it]
319
  86%|████████▋ | 107/124 [26:10<03:51, 13.62s/it]
320
 
 
321
  86%|████████▋ | 107/124 [26:10<03:51, 13.62s/it]
322
  87%|████████▋ | 108/124 [26:24<03:37, 13.61s/it]
323
 
 
324
  87%|████████▋ | 108/124 [26:24<03:37, 13.61s/it]
325
  88%|████████▊ | 109/124 [26:37<03:23, 13.58s/it]
326
 
 
327
  88%|████████▊ | 109/124 [26:37<03:23, 13.58s/it]
328
  89%|████████▊ | 110/124 [26:51<03:10, 13.58s/it]
329
 
 
330
  89%|████████▊ | 110/124 [26:51<03:10, 13.58s/it]
331
  90%|████████▉ | 111/124 [27:05<02:56, 13.56s/it]
332
 
 
333
  90%|████████▉ | 111/124 [27:05<02:56, 13.56s/it]
334
  90%|█████████ | 112/124 [27:18<02:42, 13.55s/it]
335
 
 
336
  90%|█████████ | 112/124 [27:18<02:42, 13.55s/it]
337
  91%|█████████ | 113/124 [27:32<02:29, 13.56s/it]
338
 
 
339
  91%|█████████ | 113/124 [27:32<02:29, 13.56s/it]
340
  92%|█████████▏| 114/124 [27:45<02:15, 13.53s/it]
341
 
 
342
  92%|█████████▏| 114/124 [27:45<02:15, 13.53s/it]
343
  93%|█████████▎| 115/124 [27:59<02:01, 13.53s/it]
344
 
 
345
  93%|█████████▎| 115/124 [27:59<02:01, 13.53s/it]
346
  94%|█████████▎| 116/124 [28:13<01:50, 13.75s/it]
347
 
 
348
  94%|█████████▎| 116/124 [28:13<01:50, 13.75s/it]
349
  94%|█████████▍| 117/124 [28:27<01:35, 13.70s/it]
350
 
 
351
  94%|█████████▍| 117/124 [28:27<01:35, 13.70s/it]
352
  95%|█████████▌| 118/124 [28:40<01:21, 13.64s/it]
353
 
 
354
  95%|█████████▌| 118/124 [28:40<01:21, 13.64s/it]
355
  96%|█████████▌| 119/124 [28:54<01:08, 13.61s/it]
356
 
 
357
  96%|█████████▌| 119/124 [28:54<01:08, 13.61s/it]
358
  97%|█████████▋| 120/124 [29:07<00:54, 13.58s/it]
359
 
 
360
  97%|█████████▋| 120/124 [29:07<00:54, 13.58s/it]
361
  98%|█████████▊| 121/124 [29:21<00:40, 13.56s/it]
362
 
 
363
  98%|█████████▊| 121/124 [29:21<00:40, 13.56s/it]
364
  98%|█████████▊| 122/124 [29:34<00:27, 13.54s/it]
365
 
 
366
  98%|█████████▊| 122/124 [29:34<00:27, 13.54s/it]
367
  99%|█████████▉| 123/124 [29:48<00:13, 13.53s/it]
368
 
 
369
  99%|█████████▉| 123/124 [29:48<00:13, 13.53s/it]
370
 
 
 
 
 
 
371
 
 
 
 
 
 
 
1
+ [2026-08-15 01:08:10,079] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
2
+ warnings.warn(
3
+
4
+ [2026-08-15 01:08:12,327] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/torchao/quantization/quant_api.py:1731: SyntaxWarning: invalid escape sequence '\.'
5
+ """Configuration class for applying different quantization configs to modules or parameters based on their fully qualified names (FQNs).
6
+
7
+ W0815 01:08:12.494000 1096 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
8
+ W0815 01:08:12.558000 1096 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
9
+
10
+ #@@ #@@ @@# @@#
11
+ @@ @@ @@ @@ =@@# @@ #@ =@@#.
12
+ @@ #@@@@@@@@@ @@ #@#@= @@ #@ .=@@
13
+ #@@@@@@@@@@@@@@@@@ =@# @# ##= ## =####=+ @@ =#####+ =#@@###. @@
14
+ @@@@@@@@@@/ +@@/ +@@ #@ =@= #@= @@ =@#+ +#@# @@ =@#+ +#@# #@. @@
15
+ @@@@@@@@@@ ##@@ ##@@ =@# @# =@# @# @@ @@ @@ @@ #@ #@ @@
16
+ @@@@@@@@@@@@@@@@@@@@ #@=+++#@= =@@# @@ @@ @@ @@ #@ #@ @@
17
+ =@#=====@@ =@# @# @@ @@ @@ @@ #@ #@ @@
18
+ @@@@@@@@@@@@@@@@ @@@@ #@ #@= #@= +@@ #@# =@# @@. =@# =@# #@. @@
19
+ =@# @# #@= #@ =#@@@@#= +#@@= +#@@@@#= .##@@+ @@
20
+ @@@@ @@@@@@@@@@@@@@@@
21
+
22
+ The following values were not passed to `accelerate launch` and had defaults used instead:
23
+ `--num_processes` was set to a value of `2`
24
+ More than one GPU was found, enabling multi-GPU training.
25
+ If this was unintended please pass in `--num_processes=1`.
26
+ `--num_machines` was set to a value of `1`
27
+ `--mixed_precision` was set to a value of `'no'`
28
+ `--dynamo_backend` was set to a value of `'no'`
29
+ To avoid this warning pass in values for each of the problematic parameters or run `accelerate config`.
30
+ [2026-08-15 01:08:23,555] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
31
+ warnings.warn(
32
+
33
+ [2026-08-15 01:08:23,558] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
34
+ warnings.warn(
35
+
36
+ W0815 01:08:25.865000 1362 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
37
+ W0815 01:08:25.882000 1362 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
38
+ W0815 01:08:25.958000 1361 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
39
+ W0815 01:08:25.975000 1361 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
40
+ [2026-08-15 01:08:28,364] [INFO] [axolotl.integrations.base] Attempting to load plugin: axolotl.integrations.liger.LigerPlugin
41
+ [2026-08-15 01:08:28,366] [INFO] [axolotl.integrations.base] Plugin loaded successfully: axolotl.integrations.liger.LigerPlugin
42
+ [2026-08-15 01:08:28,366] [INFO] [axolotl.integrations.base] Attempting to load plugin: scimt.train.axolotl_plugins.CheckpointSchedulePlugin
43
+ [2026-08-15 01:08:28,385] [INFO] [axolotl.integrations.base] Plugin loaded successfully: scimt.train.axolotl_plugins.CheckpointSchedulePlugin
44
+ [2026-08-15 01:08:28,431] [WARNING] [axolotl.utils.schemas.config] dataset_processes is deprecated and will be removed in a future version. Please use dataset_num_proc instead.
45
+ [2026-08-15 01:08:28,431] [WARNING] [axolotl.utils.schemas.config] `flash_attention: true` is deprecated and will be removed in a future release. Use `attn_implementation: flash_attention_2` instead.
46
+ [2026-08-15 01:08:28,431] [INFO] [axolotl.utils.schemas.validation] explicitly setting `eval_sample_packing` to match `sample_packing`
47
+ [2026-08-15 01:08:28,431] [WARNING] [axolotl.utils.schemas.validation] Configuring FSDP fields with the `fsdp_` prefix is deprecated. Please omit the `fsdp_` prefix from the any fields in `fsdp_config`.
48
+ [2026-08-15 01:08:28,476] [INFO] [axolotl.cli.config] config:
49
+ {
50
+ "activation_offloading": false,
51
+ "attn_implementation": "flash_attention_2",
52
+ "attn_needs_dtype_cast": true,
53
+ "attn_supports_packing": true,
54
+ "attn_uses_flash_lib": true,
55
+ "axolotl_config_path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/axolotl.yaml",
56
+ "base_model": "/root/.cache/huggingface/hub/models--unsloth--gemma-3-4b-pt/snapshots/52aba93981c6ad7712b030eb6dd496ece1d279d6",
57
+ "base_model_config": "/root/.cache/huggingface/hub/models--unsloth--gemma-3-4b-pt/snapshots/52aba93981c6ad7712b030eb6dd496ece1d279d6",
58
+ "batch_size": 32,
59
+ "bf16": true,
60
+ "capabilities": {
61
+ "bf16": true,
62
+ "compute_capability": "sm_90",
63
+ "fp8": true,
64
+ "n_gpu": 2,
65
+ "n_node": 1,
66
+ "tf32": true
67
+ },
68
+ "checkpoint_schedule": [
69
+ 4,
70
+ 31,
71
+ 62,
72
+ 93,
73
+ 124
74
+ ],
75
+ "context_parallel_size": 1,
76
+ "cosine_min_lr_ratio": 0.1,
77
+ "dataloader_num_workers": 2,
78
+ "dataloader_pin_memory": true,
79
+ "dataloader_prefetch_factor": 256,
80
+ "dataset_num_proc": 16,
81
+ "dataset_prepared_path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/prepared",
82
+ "datasets": [
83
+ {
84
+ "field": "text",
85
+ "message_property_mappings": {
86
+ "content": "content",
87
+ "role": "role"
88
+ },
89
+ "path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/mix_control",
90
+ "trust_remote_code": false,
91
+ "type": "completion"
92
+ }
93
+ ],
94
+ "ddp": true,
95
+ "device": "cuda:0",
96
+ "device_map": {
97
+ "": 0
98
+ },
99
+ "dion_rank_fraction": 1.0,
100
+ "dion_rank_multiple_of": 1,
101
+ "eaft_alpha": 1.0,
102
+ "eaft_k": 20,
103
+ "env_capabilities": {
104
+ "torch_version": "2.12.1"
105
+ },
106
+ "eval_batch_size": 1,
107
+ "eval_causal_lm_metrics": [
108
+ "sacrebleu",
109
+ "comet",
110
+ "ter",
111
+ "chrf"
112
+ ],
113
+ "eval_max_new_tokens": 128,
114
+ "eval_sample_packing": true,
115
+ "eval_table_size": 0,
116
+ "experimental_skip_move_to_device": true,
117
+ "fp16": false,
118
+ "fsdp_config": {
119
+ "auto_wrap_policy": "TRANSFORMER_BASED_WRAP",
120
+ "cpu_ram_efficient_loading": true,
121
+ "fsdp_version": 2,
122
+ "offload_params": false,
123
+ "reshard_after_forward": true,
124
+ "state_dict_type": "FULL_STATE_DICT",
125
+ "transformer_layer_cls_to_wrap": "Gemma3DecoderLayer"
126
+ },
127
+ "fsdp_version": 2,
128
+ "generate_samples": false,
129
+ "generation_do_sample": true,
130
+ "generation_max_new_tokens": 50,
131
+ "generation_prompt_ratio": 0.5,
132
+ "generation_temperature": 0.7,
133
+ "gradient_accumulation_steps": 16,
134
+ "gradient_checkpointing": true,
135
+ "gradient_checkpointing_kwargs": {
136
+ "use_reentrant": true
137
+ },
138
+ "include_tkps": true,
139
+ "is_multimodal": true,
140
+ "layer_offloading": false,
141
+ "learning_rate": 1e-05,
142
+ "liger_fused_linear_cross_entropy": true,
143
+ "liger_glu_activation": true,
144
+ "liger_rms_norm": true,
145
+ "liger_rope": true,
146
+ "lisa_layers_attribute": "model.layers",
147
+ "load_best_model_at_end": false,
148
+ "load_in_4bit": false,
149
+ "load_in_8bit": false,
150
+ "local_rank": 0,
151
+ "logging_steps": 1,
152
+ "lora_dropout": 0.0,
153
+ "loraplus_lr_embedding": 1e-06,
154
+ "lr_scheduler": "cosine",
155
+ "max_grad_norm": 1.0,
156
+ "max_steps": 124,
157
+ "mean_resizing_embeddings": false,
158
+ "merge_method": "memory_efficient",
159
+ "micro_batch_size": 1,
160
+ "model_config_type": "gemma3",
161
+ "model_config_type_text": "gemma3_text",
162
+ "num_epochs": 4.0,
163
+ "num_generation_samples": 3,
164
+ "optimizer": "adamw_torch_fused",
165
+ "otel_metrics_host": "localhost",
166
+ "otel_metrics_port": 8000,
167
+ "output_dir": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints",
168
+ "pad_to_sequence_len": true,
169
+ "plugins": [
170
+ "axolotl.integrations.liger.LigerPlugin",
171
+ "scimt.train.axolotl_plugins.CheckpointSchedulePlugin"
172
+ ],
173
+ "pretrain_multipack_attn": true,
174
+ "processor_config": "/root/.cache/huggingface/hub/models--unsloth--gemma-3-4b-pt/snapshots/52aba93981c6ad7712b030eb6dd496ece1d279d6",
175
+ "profiler_steps_start": 0,
176
+ "qgalore_cos_threshold": 0.4,
177
+ "qgalore_gamma_proj": 2,
178
+ "qgalore_proj_bits": 4,
179
+ "qgalore_proj_group_size": 256,
180
+ "qgalore_proj_quant": true,
181
+ "qgalore_proj_type": "std",
182
+ "qgalore_queue_size": 5,
183
+ "qgalore_rank": 256,
184
+ "qgalore_scale": 0.25,
185
+ "qgalore_update_proj_gap": 200,
186
+ "qlora_sharded_model_loading": false,
187
+ "quantize_moe_experts": false,
188
+ "ray_num_workers": 1,
189
+ "relora_prune_method": "magnitude",
190
+ "resources_per_worker": {
191
+ "GPU": 1
192
+ },
193
+ "sample_packing": true,
194
+ "sample_packing_bin_size": 200,
195
+ "sample_packing_group_size": 100000,
196
+ "save_only_model": false,
197
+ "save_safetensors": true,
198
+ "save_strategy": "no",
199
+ "save_total_limit": 6,
200
+ "seed": 314159,
201
+ "sequence_len": 8192,
202
+ "shuffle_before_merging_datasets": false,
203
+ "shuffle_merged_datasets": true,
204
+ "skip_prepare_dataset": false,
205
+ "streaming_multipack_buffer_size": 10000,
206
+ "strict": false,
207
+ "tensor_parallel_size": 1,
208
+ "tf32": true,
209
+ "tiled_mlp_use_original_mlp": true,
210
+ "tokenizer_config": "/root/.cache/huggingface/hub/models--unsloth--gemma-3-4b-pt/snapshots/52aba93981c6ad7712b030eb6dd496ece1d279d6",
211
+ "tokenizer_save_jinja_files": true,
212
+ "torch_dtype": "torch.bfloat16",
213
+ "train_on_inputs": false,
214
+ "trl": {
215
+ "async_prefetch": false,
216
+ "log_completions": false,
217
+ "mask_truncated_completions": false,
218
+ "ref_model_mixup_alpha": 0.9,
219
+ "ref_model_sync_steps": 64,
220
+ "replay_buffer_size": 0,
221
+ "replay_recompute_logps": true,
222
+ "reroll_max_groups": 1,
223
+ "reroll_start_fraction": 1.0,
224
+ "reward_num_workers": 1,
225
+ "scale_rewards": true,
226
+ "skip_zero_advantage_batches": true,
227
+ "sync_ref_model": false,
228
+ "use_data_producer": false,
229
+ "use_vllm": false,
230
+ "vllm_lora_sync": false,
231
+ "vllm_server_host": "0.0.0.0",
232
+ "vllm_server_port": 8000
233
+ },
234
+ "trust_remote_code": false,
235
+ "use_otel_metrics": false,
236
+ "use_ray": false,
237
+ "val_set_size": 0.0,
238
+ "vllm": {
239
+ "device": "auto",
240
+ "dtype": "auto",
241
+ "gpu_memory_utilization": 0.9,
242
+ "host": "0.0.0.0",
243
+ "port": 8000
244
+ },
245
+ "warmup_ratio": 0.03,
246
+ "weight_decay": 0.01,
247
+ "world_size": 2
248
+ }
249
+ [2026-08-15 01:08:29,774] [INFO] [axolotl.utils.data.sft] [RANK:1] Loading raw datasets...
250
+ [2026-08-15 01:08:29,779] [INFO] [axolotl.utils.data.wrappers] [RANK:1] Loading dataset: /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/mix_control with base_type: completion and prompt_style: None
251
+ [2026-08-15 01:08:30,085] [INFO] [axolotl.loaders.tokenizer] No Chat template selected. Consider adding a chat template for easier inference.
252
+
253
+
254
+
255
+
256
+
257
+ warnings.warn(
258
+
259
+ [2026-08-15 01:09:25,424] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
260
+ warnings.warn(
261
+
262
+ [2026-08-15 01:09:25,424] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
263
+ warnings.warn(
264
+
265
+ [2026-08-15 01:09:25,428] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
266
+ warnings.warn(
267
+
268
+ [2026-08-15 01:09:25,481] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
269
+ warnings.warn(
270
+
271
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
272
+ warnings.warn(
273
+
274
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
275
+ warnings.warn(
276
+
277
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
278
+ warnings.warn(
279
+
280
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
281
+ warnings.warn(
282
+
283
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
284
+ warnings.warn(
285
+
286
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
287
+ warnings.warn(
288
+
289
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
290
+ warnings.warn(
291
+
292
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
293
+ warnings.warn(
294
+
295
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
296
+ warnings.warn(
297
+
298
+ [2026-08-15 01:09:25,505] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
299
+ warnings.warn(
300
+
301
+ [2026-08-15 01:09:25,506] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
302
+ warnings.warn(
303
+
304
+ [2026-08-15 01:09:25,507] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version!
305
+ warnings.warn(
306
+
307
+ W0815 01:09:28.082000 1706 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
308
+ W0815 01:09:28.086000 1708 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
309
+ W0815 01:09:28.090000 1698 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
310
+ W0815 01:09:28.101000 1706 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
311
+ W0815 01:09:28.104000 1708 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
312
+ W0815 01:09:28.109000 1698 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
313
+ W0815 01:09:28.129000 1703 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
314
+ W0815 01:09:28.129000 1702 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
315
+ W0815 01:09:28.147000 1702 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
316
+ W0815 01:09:28.148000 1703 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
317
+ W0815 01:09:28.529000 1697 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
318
+ W0815 01:09:28.547000 1697 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
319
+ W0815 01:09:28.667000 1705 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
320
+ W0815 01:09:28.686000 1705 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
321
+ W0815 01:09:28.741000 1707 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
322
+ W0815 01:09:28.742000 1695 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
323
+ W0815 01:09:28.746000 1696 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
324
+ W0815 01:09:28.756000 1709 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
325
+ W0815 01:09:28.759000 1707 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
326
+ W0815 01:09:28.759000 1695 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
327
+ W0815 01:09:28.763000 1696 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
328
+ W0815 01:09:28.765000 1700 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
329
+ W0815 01:09:28.774000 1709 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
330
+ W0815 01:09:28.783000 1700 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
331
+ W0815 01:09:28.799000 1704 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
332
+ W0815 01:09:28.819000 1704 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
333
+ W0815 01:09:28.823000 1701 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
334
+ W0815 01:09:28.828000 1699 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
335
+ W0815 01:09:28.835000 1713 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
336
+ W0815 01:09:28.841000 1701 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
337
+ W0815 01:09:28.847000 1699 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
338
+ W0815 01:09:28.851000 1694 torch/utils/_pytree.py:630] <enum 'KernelPreference'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
339
+ W0815 01:09:28.853000 1713 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
340
+ W0815 01:09:28.868000 1694 torch/utils/_pytree.py:630] <enum 'ScaleCalculationMode'> is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release.
341
+
342
+ [2026-08-15 01:09:32,443] [INFO] [axolotl.utils.data.shared] Loading prepared dataset from disk at /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/prepared/c83a200068420518a6b8ed7ef72d3196...
343
+ NCCL version 2.29.3+cuda12.9
344
+ [2026-08-15 01:09:37,085] [INFO] [axolotl.utils.samplers.multipack] gather_len_batches: [981, 981]
345
+ [2026-08-15 01:09:37,307] [INFO] [axolotl.utils.trainer] sample_packing_eff_est across ranks: [0.9971898794174194, 0.9971898794174194]
346
+ [2026-08-15 01:09:37,309] [INFO] [axolotl.utils.data.sft] Maximum number of steps set at 120
347
+ [2026-08-15 01:09:38,967] [INFO] [axolotl.loaders.tokenizer] No Chat template selected. Consider adding a chat template for easier inference.
348
+ [2026-08-15 01:09:41,397] [INFO] [axolotl.monkeypatch.attention.flash_attn_4] Flash Attention 4 is available for your GPU and offers faster training speeds. To enable: pip install flash-attn-4
349
+ [2026-08-15 01:09:41,403] [INFO] [axolotl.loaders.patch_manager] Applying multipack dataloader patch for sample packing...
350
+ [2026-08-15 01:09:42,863] [INFO] [axolotl.integrations.liger.plugin] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True}
351
+
352
+
353
+ [2026-08-15 01:09:43,228] [INFO] [axolotl.loaders.model] Converting modules to torch.bfloat16
354
+ [transformers] When using FSDP full shard, instead of using `gradient_checkpointing` in TrainingArguments, please use `activation_checkpointing` in `fsdp_config`. The former introduces a redundant AllGather operation in backward pass. Reference: https://github.com/huggingface/transformers/issues/30404
355
+ [2026-08-15 01:09:44,704] [WARNING] [accelerate.utils.dataclasses] sync_module_states is obsolete in FSDP2, as it is not needed anymore.Setting sync_module_states to None.
356
+ [2026-08-15 01:09:44,731] [INFO] [axolotl.train] Pre-saving tokenizer to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints...
357
+ [2026-08-15 01:09:44,988] [INFO] [axolotl.train] Pre-saving model config to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints...
358
+ [2026-08-15 01:09:44,993] [INFO] [axolotl.train] Pre-saving processor to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints...
359
+ [2026-08-15 01:09:45,248] [INFO] [axolotl.train] Starting trainer...
360
+ [transformers] When using FSDP full shard, instead of using `gradient_checkpointing` in TrainingArguments, please use `activation_checkpointing` in `fsdp_config`. The former introduces a redundant AllGather operation in backward pass. Reference: https://github.com/huggingface/transformers/issues/30404
361
+ [2026-08-15 01:09:48,044] [WARNING] [accelerate.utils.dataclasses] sync_module_states is obsolete in FSDP2, as it is not needed anymore.Setting sync_module_states to None.
362
+ [2026-08-15 01:09:55,508] [INFO] [axolotl.utils.samplers.multipack] gather_len_batches: [981, 981]
363
+ [2026-08-15 01:09:55,695] [INFO] [axolotl.monkeypatch.accelerate.fsdp2] Broadcasting full state dict to all ranks...
364
+
365
  0%| | 0/124 [00:00<?, ?it/s][transformers] `use_return_dict` is deprecated! Use `return_dict` instead!
366
+ [transformers] `use_return_dict` is deprecated! Use `return_dict` instead!
367
+
368
  1%| | 1/124 [00:19<39:52, 19.45s/it]
369
 
370
+
371
  1%| | 1/124 [00:19<39:52, 19.45s/it]
372
  2%|▏ | 2/124 [00:33<32:42, 16.08s/it]
373
 
374
+
375
  2%|▏ | 2/124 [00:33<32:42, 16.08s/it]
376
  2%|▏ | 3/124 [00:46<30:01, 14.89s/it]
377
 
378
+
379
  2%|▏ | 3/124 [00:46<30:01, 14.89s/it]
380
  3%|▎ | 4/124 [01:00<28:36, 14.30s/it]
381
 
382
+
383
  3%|▎ | 4/124 [01:00<28:36, 14.30s/it][2026-08-15 01:11:03,971] [INFO] [axolotl.core.trainers.base] Saving model checkpoint to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-4
384
+
385
+
386
+
387
+
388
  4%|▍ | 5/124 [01:46<51:01, 25.72s/it]
389
 
390
+
391
  4%|▍ | 5/124 [01:46<51:01, 25.72s/it]
392
  5%|▍ | 6/124 [01:59<42:23, 21.55s/it]
393
 
394
+
395
  5%|▍ | 6/124 [01:59<42:23, 21.55s/it]
396
  6%|▌ | 7/124 [02:12<36:54, 18.92s/it]
397
 
398
+
399
  6%|▌ | 7/124 [02:13<36:54, 18.92s/it]
400
  6%|▋ | 8/124 [02:26<33:15, 17.21s/it]
401
 
402
+
403
  6%|▋ | 8/124 [02:26<33:15, 17.21s/it]
404
  7%|▋ | 9/124 [02:39<30:43, 16.03s/it]
405
 
406
+
407
  7%|▋ | 9/124 [02:39<30:43, 16.03s/it]
408
  8%|▊ | 10/124 [02:53<28:58, 15.25s/it]
409
 
410
+
411
  8%|▊ | 10/124 [02:53<28:58, 15.25s/it]
412
  9%|▉ | 11/124 [03:07<27:44, 14.73s/it]
413
 
414
+
415
  9%|▉ | 11/124 [03:07<27:44, 14.73s/it]
416
  10%|▉ | 12/124 [03:20<26:48, 14.36s/it]
417
 
418
+
419
  10%|▉ | 12/124 [03:20<26:48, 14.36s/it]
420
  10%|█ | 13/124 [03:34<26:06, 14.11s/it]
421
 
422
+
423
  10%|█ | 13/124 [03:34<26:06, 14.11s/it]
424
  11%|█▏ | 14/124 [03:47<25:32, 13.93s/it]
425
 
426
+
427
  11%|█▏ | 14/124 [03:47<25:32, 13.93s/it]
428
  12%|█▏ | 15/124 [04:01<25:06, 13.82s/it]
429
 
430
+
431
  12%|█▏ | 15/124 [04:01<25:06, 13.82s/it]
432
  13%|█▎ | 16/124 [04:14<24:42, 13.73s/it]
433
 
434
+
435
  13%|█▎ | 16/124 [04:14<24:42, 13.73s/it]
436
  14%|█▎ | 17/124 [04:28<24:23, 13.68s/it]
437
 
438
+
439
  14%|█▎ | 17/124 [04:28<24:23, 13.68s/it]
440
  15%|█▍ | 18/124 [04:41<24:04, 13.63s/it]
441
 
442
+
443
  15%|█▍ | 18/124 [04:41<24:04, 13.63s/it]
444
  15%|█▌ | 19/124 [04:55<23:48, 13.60s/it]
445
 
446
+
447
  15%|█▌ | 19/124 [04:55<23:48, 13.60s/it]
448
  16%|█▌ | 20/124 [05:08<23:33, 13.59s/it]
449
 
450
+
451
  16%|█▌ | 20/124 [05:08<23:33, 13.59s/it]
452
  17%|█▋ | 21/124 [05:22<23:17, 13.56s/it]
453
 
454
+
455
  17%|█▋ | 21/124 [05:22<23:17, 13.56s/it]
456
  18%|█▊ | 22/124 [05:35<23:02, 13.55s/it]
457
 
458
+
459
  18%|█▊ | 22/124 [05:35<23:02, 13.55s/it]
460
  19%|█▊ | 23/124 [05:49<22:48, 13.55s/it]
461
 
462
+
463
  19%|█▊ | 23/124 [05:49<22:48, 13.55s/it]
464
  19%|█▉ | 24/124 [06:03<22:35, 13.56s/it]
465
 
466
+
467
  19%|█▉ | 24/124 [06:03<22:35, 13.56s/it]
468
  20%|██ | 25/124 [06:16<22:20, 13.54s/it]
469
 
470
+
471
  20%|██ | 25/124 [06:16<22:20, 13.54s/it]
472
  21%|██ | 26/124 [06:30<22:07, 13.54s/it]
473
 
474
+
475
  21%|██ | 26/124 [06:30<22:07, 13.54s/it]
476
  22%|██▏ | 27/124 [06:44<22:10, 13.72s/it]
477
 
478
+
479
  22%|██▏ | 27/124 [06:44<22:10, 13.72s/it]
480
  23%|██▎ | 28/124 [06:57<21:50, 13.66s/it]
481
 
482
+
483
  23%|██▎ | 28/124 [06:57<21:50, 13.66s/it]
484
  23%|██▎ | 29/124 [07:11<21:32, 13.61s/it]
485
 
486
+
487
  23%|██▎ | 29/124 [07:11<21:32, 13.61s/it]
488
  24%|██▍ | 30/124 [07:24<21:15, 13.57s/it]
489
 
490
+
491
  24%|██▍ | 30/124 [07:24<21:15, 13.57s/it]
492
  25%|██▌ | 31/124 [07:33<18:44, 12.09s/it]
493
 
494
+
495
  25%|██▌ | 31/124 [07:33<18:44, 12.09s/it][2026-08-15 01:17:36,195] [INFO] [axolotl.core.trainers.base] Saving model checkpoint to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-31
496
+
497
+
498
+
499
+
500
  26%|██▌ | 32/124 [08:18<33:48, 22.04s/it]
501
 
502
+
503
  26%|██▌ | 32/124 [08:18<33:48, 22.04s/it]
504
  27%|██▋ | 33/124 [08:32<29:31, 19.47s/it]
505
 
506
+
507
  27%|██▋ | 33/124 [08:32<29:31, 19.47s/it]
508
  27%|██▋ | 34/124 [08:45<26:30, 17.67s/it]
509
 
510
+
511
  27%|██▋ | 34/124 [08:45<26:30, 17.67s/it]
512
  28%|██▊ | 35/124 [08:58<24:20, 16.41s/it]
513
 
514
+
515
  28%|██▊ | 35/124 [08:58<24:20, 16.41s/it]
516
  29%|██▉ | 36/124 [09:12<22:47, 15.53s/it]
517
 
518
+
519
  29%|██▉ | 36/124 [09:12<22:47, 15.53s/it]
520
  30%|██▉ | 37/124 [09:25<21:38, 14.92s/it]
521
 
522
+
523
  30%|██▉ | 37/124 [09:25<21:38, 14.92s/it]
524
  31%|███ | 38/124 [09:39<20:47, 14.51s/it]
525
 
526
+
527
  31%|███ | 38/124 [09:39<20:47, 14.51s/it]
528
  31%|███▏ | 39/124 [09:53<20:08, 14.22s/it]
529
 
530
+
531
  31%|███▏ | 39/124 [09:53<20:08, 14.22s/it]
532
  32%|███▏ | 40/124 [10:06<19:34, 13.99s/it]
533
 
534
+
535
  32%|███▏ | 40/124 [10:06<19:34, 13.99s/it]
536
  33%|███▎ | 41/124 [10:20<19:09, 13.85s/it]
537
 
538
+
539
  33%|███▎ | 41/124 [10:20<19:09, 13.85s/it]
540
  34%|███▍ | 42/124 [10:33<18:48, 13.77s/it]
541
 
542
+
543
  34%|███▍ | 42/124 [10:33<18:48, 13.77s/it]
544
  35%|███▍ | 43/124 [10:47<18:28, 13.69s/it]
545
 
546
+
547
  35%|███▍ | 43/124 [10:47<18:28, 13.69s/it]
548
  35%|███▌ | 44/124 [11:00<18:10, 13.64s/it]
549
 
550
+
551
  35%|███▌ | 44/124 [11:00<18:10, 13.64s/it]
552
  36%|███▋ | 45/124 [11:14<17:54, 13.61s/it]
553
 
554
+
555
  36%|███▋ | 45/124 [11:14<17:54, 13.61s/it]
556
  37%|███▋ | 46/124 [11:27<17:40, 13.59s/it]
557
 
558
+
559
  37%|███▋ | 46/124 [11:27<17:40, 13.59s/it]
560
  38%|███▊ | 47/124 [11:41<17:25, 13.58s/it]
561
 
562
+
563
  38%|███▊ | 47/124 [11:41<17:25, 13.58s/it]
564
  39%|███▊ | 48/124 [11:54<17:10, 13.56s/it]
565
 
566
+
567
  39%|███▊ | 48/124 [11:54<17:10, 13.56s/it]
568
  40%|███▉ | 49/124 [12:08<16:55, 13.54s/it]
569
 
570
+
571
  40%|███▉ | 49/124 [12:08<16:55, 13.54s/it]
572
  40%|████ | 50/124 [12:21<16:42, 13.54s/it]
573
 
574
+
575
  40%|████ | 50/124 [12:21<16:42, 13.54s/it]
576
  41%|████ | 51/124 [12:35<16:29, 13.55s/it]
577
 
578
+
579
  41%|████ | 51/124 [12:35<16:29, 13.55s/it]
580
  42%|████▏ | 52/124 [12:48<16:13, 13.52s/it]
581
 
582
+
583
  42%|████▏ | 52/124 [12:48<16:13, 13.52s/it]
584
  43%|████▎ | 53/124 [13:02<15:59, 13.52s/it]
585
 
586
+
587
  43%|████▎ | 53/124 [13:02<15:59, 13.52s/it]
588
  44%|████▎ | 54/124 [13:15<15:46, 13.52s/it]
589
 
590
+
591
  44%|████▎ | 54/124 [13:15<15:46, 13.52s/it]
592
  44%|████▍ | 55/124 [13:30<15:49, 13.76s/it]
593
 
594
+
595
  44%|████▍ | 55/124 [13:30<15:49, 13.76s/it]
596
  45%|████▌ | 56/124 [13:43<15:29, 13.67s/it]
597
 
598
+
599
  45%|████▌ | 56/124 [13:43<15:29, 13.67s/it]
600
  46%|████▌ | 57/124 [13:57<15:13, 13.63s/it]
601
 
602
+
603
  46%|████▌ | 57/124 [13:57<15:13, 13.63s/it]
604
  47%|████▋ | 58/124 [14:10<14:56, 13.59s/it]
605
 
606
+
607
  47%|████▋ | 58/124 [14:10<14:56, 13.59s/it]
608
  48%|████▊ | 59/124 [14:24<14:42, 13.57s/it]
609
 
610
+
611
  48%|████▊ | 59/124 [14:24<14:42, 13.57s/it]
612
  48%|████▊ | 60/124 [14:37<14:27, 13.55s/it]
613
 
614
+
615
  48%|████▊ | 60/124 [14:37<14:27, 13.55s/it]
616
  49%|████▉ | 61/124 [14:51<14:11, 13.52s/it]
617
 
618
+
619
  49%|████▉ | 61/124 [14:51<14:11, 13.52s/it]
620
  50%|█████ | 62/124 [14:59<12:28, 12.07s/it]
621
 
622
+
623
  50%|█████ | 62/124 [14:59<12:28, 12.07s/it][2026-08-15 01:25:02,804] [INFO] [axolotl.core.trainers.base] Saving model checkpoint to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-62
624
+
625
+
626
+
627
+
628
  51%|█████ | 63/124 [15:46<22:50, 22.47s/it]
629
 
630
+
631
  51%|█████ | 63/124 [15:46<22:50, 22.47s/it]
632
  52%|█████▏ | 64/124 [16:00<19:45, 19.76s/it]
633
 
634
+
635
  52%|█████▏ | 64/124 [16:00<19:45, 19.76s/it]
636
  52%|█████▏ | 65/124 [16:13<17:35, 17.88s/it]
637
 
638
+
639
  52%|█████▏ | 65/124 [16:13<17:35, 17.88s/it]
640
  53%|█████▎ | 66/124 [16:27<16:00, 16.56s/it]
641
 
642
+
643
  53%|█████▎ | 66/124 [16:27<16:00, 16.56s/it]
644
  54%|█████▍ | 67/124 [16:40<14:51, 15.64s/it]
645
 
646
+
647
  54%|█████▍ | 67/124 [16:40<14:51, 15.64s/it]
648
  55%|█████▍ | 68/124 [16:54<13:59, 15.00s/it]
649
 
650
+
651
  55%|█████▍ | 68/124 [16:54<13:59, 15.00s/it]
652
  56%|█████▌ | 69/124 [17:07<13:21, 14.57s/it]
653
 
654
+
655
  56%|█████▌ | 69/124 [17:07<13:21, 14.57s/it]
656
  56%|█████▋ | 70/124 [17:21<12:49, 14.26s/it]
657
 
658
+
659
  56%|█████▋ | 70/124 [17:21<12:49, 14.26s/it]
660
  57%|█████▋ | 71/124 [17:34<12:23, 14.02s/it]
661
 
662
+
663
  57%|█████▋ | 71/124 [17:34<12:23, 14.02s/it]
664
  58%|█████▊ | 72/124 [17:48<12:01, 13.87s/it]
665
 
666
+
667
  58%|█████▊ | 72/124 [17:48<12:01, 13.87s/it]
668
  59%|█████▉ | 73/124 [18:01<11:42, 13.78s/it]
669
 
670
+
671
  59%|█████▉ | 73/124 [18:01<11:42, 13.78s/it]
672
  60%|█████▉ | 74/124 [18:15<11:25, 13.71s/it]
673
 
674
+
675
  60%|█���███▉ | 74/124 [18:15<11:25, 13.71s/it]
676
  60%|██████ | 75/124 [18:28<11:09, 13.66s/it]
677
 
678
+
679
  60%|██████ | 75/124 [18:28<11:09, 13.66s/it]
680
  61%|██████▏ | 76/124 [18:42<10:53, 13.62s/it]
681
 
682
+
683
  61%|██████▏ | 76/124 [18:42<10:53, 13.62s/it]
684
  62%|██████▏ | 77/124 [18:55<10:39, 13.60s/it]
685
 
686
+
687
  62%|██████▏ | 77/124 [18:55<10:39, 13.60s/it]
688
  63%|██████▎ | 78/124 [19:09<10:24, 13.58s/it]
689
 
690
+
691
  63%|██████▎ | 78/124 [19:09<10:24, 13.58s/it]
692
  64%|██████▎ | 79/124 [19:22<10:10, 13.58s/it]
693
 
694
+
695
  64%|██████▎ | 79/124 [19:22<10:10, 13.58s/it]
696
  65%|██████▍ | 80/124 [19:36<09:56, 13.56s/it]
697
 
698
+
699
  65%|██████▍ | 80/124 [19:36<09:56, 13.56s/it]
700
  65%|██████▌ | 81/124 [19:49<09:42, 13.55s/it]
701
 
702
+
703
  65%|██████▌ | 81/124 [19:49<09:42, 13.55s/it]
704
  66%|██████▌ | 82/124 [20:03<09:28, 13.55s/it]
705
 
706
+
707
  66%|██████▌ | 82/124 [20:03<09:28, 13.55s/it]
708
  67%|██████▋ | 83/124 [20:17<09:14, 13.53s/it]
709
 
710
+
711
  67%|██████▋ | 83/124 [20:17<09:14, 13.53s/it]
712
  68%|██████▊ | 84/124 [20:30<09:01, 13.53s/it]
713
 
714
+
715
  68%|██████▊ | 84/124 [20:30<09:01, 13.53s/it]
716
  69%|██████▊ | 85/124 [20:44<08:52, 13.65s/it]
717
 
718
+
719
  69%|██████▊ | 85/124 [20:44<08:52, 13.65s/it]
720
  69%|██████▉ | 86/124 [20:58<08:41, 13.74s/it]
721
 
722
+
723
  69%|██████▉ | 86/124 [20:58<08:41, 13.74s/it]
724
  70%|███████ | 87/124 [21:11<08:25, 13.66s/it]
725
 
726
+
727
  70%|███████ | 87/124 [21:11<08:25, 13.66s/it]
728
  71%|███████ | 88/124 [21:25<08:10, 13.63s/it]
729
 
730
+
731
  71%|███████ | 88/124 [21:25<08:10, 13.63s/it]
732
  72%|███████▏ | 89/124 [21:38<07:55, 13.59s/it]
733
 
734
+
735
  72%|███████▏ | 89/124 [21:38<07:55, 13.59s/it]
736
  73%|███████▎ | 90/124 [21:52<07:41, 13.57s/it]
737
 
738
+
739
  73%|███████▎ | 90/124 [21:52<07:41, 13.57s/it]
740
  73%|███████▎ | 91/124 [22:05<07:27, 13.55s/it]
741
 
742
+
743
  73%|███████▎ | 91/124 [22:05<07:27, 13.55s/it]
744
  74%|███████▍ | 92/124 [22:19<07:12, 13.53s/it]
745
 
746
+
747
  74%|███████▍ | 92/124 [22:19<07:12, 13.53s/it]
748
  75%|███████▌ | 93/124 [22:28<06:13, 12.06s/it]
749
 
750
+
751
  75%|███████▌ | 93/124 [22:28<06:13, 12.06s/it][2026-08-15 01:32:31,116] [INFO] [axolotl.core.trainers.base] Saving model checkpoint to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-93
752
+
753
+
754
+
755
+
756
  76%|███████▌ | 94/124 [23:15<11:17, 22.57s/it]
757
 
758
+
759
  76%|███████▌ | 94/124 [23:15<11:17, 22.57s/it]
760
  77%|███████▋ | 95/124 [23:28<09:35, 19.84s/it]
761
 
762
+
763
  77%|███████▋ | 95/124 [23:28<09:35, 19.84s/it]
764
  77%|███████▋ | 96/124 [23:42<08:22, 17.94s/it]
765
 
766
+
767
  77%|███████▋ | 96/124 [23:42<08:22, 17.94s/it]
768
  78%|███████▊ | 97/124 [23:55<07:27, 16.59s/it]
769
 
770
+
771
  78%|███████▊ | 97/124 [23:55<07:27, 16.59s/it]
772
  79%|███████▉ | 98/124 [24:09<06:47, 15.66s/it]
773
 
774
+
775
  79%|███████▉ | 98/124 [24:09<06:47, 15.66s/it]
776
  80%|███████▉ | 99/124 [24:22<06:15, 15.02s/it]
777
 
778
+
779
  80%|███████▉ | 99/124 [24:22<06:15, 15.02s/it]
780
  81%|████████ | 100/124 [24:36<05:49, 14.58s/it]
781
 
782
+
783
  81%|████████ | 100/124 [24:36<05:49, 14.58s/it]
784
  81%|████████▏ | 101/124 [24:49<05:28, 14.27s/it]
785
 
786
+
787
  81%|████████▏ | 101/124 [24:49<05:28, 14.27s/it]
788
  82%|████████▏ | 102/124 [25:03<05:08, 14.03s/it]
789
 
790
+
791
  82%|████████▏ | 102/124 [25:03<05:08, 14.03s/it]
792
  83%|████████▎ | 103/124 [25:16<04:51, 13.88s/it]
793
 
794
+
795
  83%|████████▎ | 103/124 [25:16<04:51, 13.88s/it]
796
  84%|████████▍ | 104/124 [25:30<04:35, 13.79s/it]
797
 
798
+
799
  84%|██████���█▍ | 104/124 [25:30<04:35, 13.79s/it]
800
  85%|████████▍ | 105/124 [25:43<04:20, 13.71s/it]
801
 
802
+
803
  85%|████████▍ | 105/124 [25:43<04:20, 13.71s/it]
804
  85%|████████▌ | 106/124 [25:57<04:05, 13.66s/it]
805
 
806
+
807
  85%|████████▌ | 106/124 [25:57<04:05, 13.66s/it]
808
  86%|████████▋ | 107/124 [26:10<03:51, 13.62s/it]
809
 
810
+
811
  86%|████████▋ | 107/124 [26:10<03:51, 13.62s/it]
812
  87%|████████▋ | 108/124 [26:24<03:37, 13.61s/it]
813
 
814
+
815
  87%|████████▋ | 108/124 [26:24<03:37, 13.61s/it]
816
  88%|████████▊ | 109/124 [26:37<03:23, 13.58s/it]
817
 
818
+
819
  88%|████████▊ | 109/124 [26:37<03:23, 13.58s/it]
820
  89%|████████▊ | 110/124 [26:51<03:10, 13.58s/it]
821
 
822
+
823
  89%|████████▊ | 110/124 [26:51<03:10, 13.58s/it]
824
  90%|████████▉ | 111/124 [27:05<02:56, 13.56s/it]
825
 
826
+
827
  90%|████████▉ | 111/124 [27:05<02:56, 13.56s/it]
828
  90%|█████████ | 112/124 [27:18<02:42, 13.55s/it]
829
 
830
+
831
  90%|█████████ | 112/124 [27:18<02:42, 13.55s/it]
832
  91%|█████████ | 113/124 [27:32<02:29, 13.56s/it]
833
 
834
+
835
  91%|█████████ | 113/124 [27:32<02:29, 13.56s/it]
836
  92%|█████████▏| 114/124 [27:45<02:15, 13.53s/it]
837
 
838
+
839
  92%|█████████▏| 114/124 [27:45<02:15, 13.53s/it]
840
  93%|█████████▎| 115/124 [27:59<02:01, 13.53s/it]
841
 
842
+
843
  93%|█████████▎| 115/124 [27:59<02:01, 13.53s/it]
844
  94%|█████████▎| 116/124 [28:13<01:50, 13.75s/it]
845
 
846
+
847
  94%|█████████▎| 116/124 [28:13<01:50, 13.75s/it]
848
  94%|█████████▍| 117/124 [28:27<01:35, 13.70s/it]
849
 
850
+
851
  94%|█████████▍| 117/124 [28:27<01:35, 13.70s/it]
852
  95%|█████████▌| 118/124 [28:40<01:21, 13.64s/it]
853
 
854
+
855
  95%|█████████▌| 118/124 [28:40<01:21, 13.64s/it]
856
  96%|█████████▌| 119/124 [28:54<01:08, 13.61s/it]
857
 
858
+
859
  96%|█████████▌| 119/124 [28:54<01:08, 13.61s/it]
860
  97%|█████████▋| 120/124 [29:07<00:54, 13.58s/it]
861
 
862
+
863
  97%|█████████▋| 120/124 [29:07<00:54, 13.58s/it]
864
  98%|█████████▊| 121/124 [29:21<00:40, 13.56s/it]
865
 
866
+
867
  98%|█████████▊| 121/124 [29:21<00:40, 13.56s/it]
868
  98%|█████████▊| 122/124 [29:34<00:27, 13.54s/it]
869
 
870
+
871
  98%|█████████▊| 122/124 [29:34<00:27, 13.54s/it]
872
  99%|█████████▉| 123/124 [29:48<00:13, 13.53s/it]
873
 
874
+
875
  99%|█████████▉| 123/124 [29:48<00:13, 13.53s/it]
876
 
877
+
878
+
879
+
880
+
881
+
882
 
883
+
884
+ [2026-08-15 01:40:24,934] [INFO] [axolotl.train] Training completed! Saving trained model to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints.
885
+ [2026-08-15 01:40:29,721] [INFO] [axolotl.core.trainers.base] Saving model checkpoint to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints
886
+
887
+ [2026-08-15 01:40:32,330] [INFO] [axolotl.train] Model successfully saved to /workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/trainer_state.final.json ADDED
@@ -0,0 +1,1770 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 4.0,
6
+ "eval_steps": 500,
7
+ "global_step": 124,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0326530612244898,
14
+ "grad_norm": 3.96875,
15
+ "learning_rate": 0.0,
16
+ "loss": 1.6685791015625,
17
+ "memory/device_reserved (GiB)": 26.08,
18
+ "memory/max_active (GiB)": 20.03,
19
+ "memory/max_allocated (GiB)": 20.03,
20
+ "ppl": 5.30463,
21
+ "step": 1,
22
+ "tokens/total": 262144,
23
+ "tokens/train_per_sec_per_gpu": 471.78,
24
+ "tokens/trainable": 261922
25
+ },
26
+ {
27
+ "epoch": 0.0653061224489796,
28
+ "grad_norm": 4.21875,
29
+ "learning_rate": 3.3333333333333333e-06,
30
+ "loss": 1.5614013671875,
31
+ "memory/device_reserved (GiB)": 33.29,
32
+ "memory/max_active (GiB)": 27.26,
33
+ "memory/max_allocated (GiB)": 27.26,
34
+ "ppl": 4.76549,
35
+ "step": 2,
36
+ "tokens/total": 524288,
37
+ "tokens/train_per_sec_per_gpu": 597.28,
38
+ "tokens/trainable": 523573
39
+ },
40
+ {
41
+ "epoch": 0.09795918367346938,
42
+ "grad_norm": 3.40625,
43
+ "learning_rate": 6.666666666666667e-06,
44
+ "loss": 1.5543212890625,
45
+ "memory/device_reserved (GiB)": 33.29,
46
+ "memory/max_active (GiB)": 27.26,
47
+ "memory/max_allocated (GiB)": 27.26,
48
+ "ppl": 4.73187,
49
+ "step": 3,
50
+ "tokens/total": 786432,
51
+ "tokens/train_per_sec_per_gpu": 606.45,
52
+ "tokens/trainable": 785311
53
+ },
54
+ {
55
+ "epoch": 0.1306122448979592,
56
+ "grad_norm": 2.4375,
57
+ "learning_rate": 1e-05,
58
+ "loss": 1.575927734375,
59
+ "memory/device_reserved (GiB)": 33.29,
60
+ "memory/max_active (GiB)": 27.26,
61
+ "memory/max_allocated (GiB)": 27.26,
62
+ "ppl": 4.83523,
63
+ "step": 4,
64
+ "tokens/total": 1048576,
65
+ "tokens/train_per_sec_per_gpu": 609.43,
66
+ "tokens/trainable": 1047069
67
+ },
68
+ {
69
+ "epoch": 0.16326530612244897,
70
+ "grad_norm": 2.921875,
71
+ "learning_rate": 9.998483343865806e-06,
72
+ "loss": 1.529052734375,
73
+ "memory/device_reserved (GiB)": 33.29,
74
+ "memory/max_active (GiB)": 27.26,
75
+ "memory/max_allocated (GiB)": 27.26,
76
+ "ppl": 4.6138,
77
+ "step": 5,
78
+ "tokens/total": 1310720,
79
+ "tokens/train_per_sec_per_gpu": 604.52,
80
+ "tokens/trainable": 1308722
81
+ },
82
+ {
83
+ "epoch": 0.19591836734693877,
84
+ "grad_norm": 2.625,
85
+ "learning_rate": 9.993934397794704e-06,
86
+ "loss": 1.39306640625,
87
+ "memory/device_reserved (GiB)": 33.29,
88
+ "memory/max_active (GiB)": 27.26,
89
+ "memory/max_allocated (GiB)": 27.26,
90
+ "ppl": 4.02718,
91
+ "step": 6,
92
+ "tokens/total": 1572864,
93
+ "tokens/train_per_sec_per_gpu": 607.08,
94
+ "tokens/trainable": 1570345
95
+ },
96
+ {
97
+ "epoch": 0.22857142857142856,
98
+ "grad_norm": 1.734375,
99
+ "learning_rate": 9.986356228092011e-06,
100
+ "loss": 1.4849853515625,
101
+ "memory/device_reserved (GiB)": 33.29,
102
+ "memory/max_active (GiB)": 27.26,
103
+ "memory/max_allocated (GiB)": 27.26,
104
+ "ppl": 4.4149,
105
+ "step": 7,
106
+ "tokens/total": 1835008,
107
+ "tokens/train_per_sec_per_gpu": 605.44,
108
+ "tokens/trainable": 1832202
109
+ },
110
+ {
111
+ "epoch": 0.2612244897959184,
112
+ "grad_norm": 1.4296875,
113
+ "learning_rate": 9.975753942969978e-06,
114
+ "loss": 1.58740234375,
115
+ "memory/device_reserved (GiB)": 33.29,
116
+ "memory/max_active (GiB)": 27.26,
117
+ "memory/max_allocated (GiB)": 27.26,
118
+ "ppl": 4.89103,
119
+ "step": 8,
120
+ "tokens/total": 2097152,
121
+ "tokens/train_per_sec_per_gpu": 604.81,
122
+ "tokens/trainable": 2093962
123
+ },
124
+ {
125
+ "epoch": 0.2938775510204082,
126
+ "grad_norm": 1.46875,
127
+ "learning_rate": 9.962134689104498e-06,
128
+ "loss": 1.34619140625,
129
+ "memory/device_reserved (GiB)": 33.29,
130
+ "memory/max_active (GiB)": 27.26,
131
+ "memory/max_allocated (GiB)": 27.26,
132
+ "ppl": 3.84276,
133
+ "step": 9,
134
+ "tokens/total": 2359296,
135
+ "tokens/train_per_sec_per_gpu": 607.46,
136
+ "tokens/trainable": 2355679
137
+ },
138
+ {
139
+ "epoch": 0.32653061224489793,
140
+ "grad_norm": 1.25,
141
+ "learning_rate": 9.945507646817764e-06,
142
+ "loss": 1.47705078125,
143
+ "memory/device_reserved (GiB)": 33.29,
144
+ "memory/max_active (GiB)": 27.26,
145
+ "memory/max_allocated (GiB)": 27.26,
146
+ "ppl": 4.38001,
147
+ "step": 10,
148
+ "tokens/total": 2621440,
149
+ "tokens/train_per_sec_per_gpu": 606.8,
150
+ "tokens/trainable": 2617308
151
+ },
152
+ {
153
+ "epoch": 0.35918367346938773,
154
+ "grad_norm": 1.171875,
155
+ "learning_rate": 9.925884023890072e-06,
156
+ "loss": 1.4583740234375,
157
+ "memory/device_reserved (GiB)": 33.29,
158
+ "memory/max_active (GiB)": 27.26,
159
+ "memory/max_allocated (GiB)": 27.26,
160
+ "ppl": 4.29896,
161
+ "step": 11,
162
+ "tokens/total": 2883584,
163
+ "tokens/train_per_sec_per_gpu": 604.71,
164
+ "tokens/trainable": 2878962
165
+ },
166
+ {
167
+ "epoch": 0.39183673469387753,
168
+ "grad_norm": 1.203125,
169
+ "learning_rate": 9.903277048005017e-06,
170
+ "loss": 1.5108642578125,
171
+ "memory/device_reserved (GiB)": 33.29,
172
+ "memory/max_active (GiB)": 27.26,
173
+ "memory/max_allocated (GiB)": 27.26,
174
+ "ppl": 4.53064,
175
+ "step": 12,
176
+ "tokens/total": 3145728,
177
+ "tokens/train_per_sec_per_gpu": 603.72,
178
+ "tokens/trainable": 3140634
179
+ },
180
+ {
181
+ "epoch": 0.42448979591836733,
182
+ "grad_norm": 1.0234375,
183
+ "learning_rate": 9.877701957833113e-06,
184
+ "loss": 1.3990478515625,
185
+ "memory/device_reserved (GiB)": 33.29,
186
+ "memory/max_active (GiB)": 27.26,
187
+ "memory/max_allocated (GiB)": 27.26,
188
+ "ppl": 4.05134,
189
+ "step": 13,
190
+ "tokens/total": 3407872,
191
+ "tokens/train_per_sec_per_gpu": 604.61,
192
+ "tokens/trainable": 3402298
193
+ },
194
+ {
195
+ "epoch": 0.45714285714285713,
196
+ "grad_norm": 0.9375,
197
+ "learning_rate": 9.849175992759867e-06,
198
+ "loss": 1.32666015625,
199
+ "memory/device_reserved (GiB)": 33.29,
200
+ "memory/max_active (GiB)": 27.26,
201
+ "memory/max_allocated (GiB)": 27.26,
202
+ "ppl": 3.76844,
203
+ "step": 14,
204
+ "tokens/total": 3670016,
205
+ "tokens/train_per_sec_per_gpu": 605.87,
206
+ "tokens/trainable": 3663962
207
+ },
208
+ {
209
+ "epoch": 0.4897959183673469,
210
+ "grad_norm": 1.7734375,
211
+ "learning_rate": 9.81771838126524e-06,
212
+ "loss": 1.46875,
213
+ "memory/device_reserved (GiB)": 33.29,
214
+ "memory/max_active (GiB)": 27.26,
215
+ "memory/max_allocated (GiB)": 27.26,
216
+ "ppl": 4.3438,
217
+ "step": 15,
218
+ "tokens/total": 3932160,
219
+ "tokens/train_per_sec_per_gpu": 603.17,
220
+ "tokens/trainable": 3925695
221
+ },
222
+ {
223
+ "epoch": 0.5224489795918368,
224
+ "grad_norm": 1.7890625,
225
+ "learning_rate": 9.783350327962313e-06,
226
+ "loss": 1.409912109375,
227
+ "memory/device_reserved (GiB)": 33.29,
228
+ "memory/max_active (GiB)": 27.26,
229
+ "memory/max_allocated (GiB)": 27.26,
230
+ "ppl": 4.0956,
231
+ "step": 16,
232
+ "tokens/total": 4194304,
233
+ "tokens/train_per_sec_per_gpu": 603.95,
234
+ "tokens/trainable": 4187301
235
+ },
236
+ {
237
+ "epoch": 0.5551020408163265,
238
+ "grad_norm": 3.03125,
239
+ "learning_rate": 9.74609499930392e-06,
240
+ "loss": 1.401123046875,
241
+ "memory/device_reserved (GiB)": 33.29,
242
+ "memory/max_active (GiB)": 27.26,
243
+ "memory/max_allocated (GiB)": 27.26,
244
+ "ppl": 4.05976,
245
+ "step": 17,
246
+ "tokens/total": 4456448,
247
+ "tokens/train_per_sec_per_gpu": 603.72,
248
+ "tokens/trainable": 4449140
249
+ },
250
+ {
251
+ "epoch": 0.5877551020408164,
252
+ "grad_norm": 1.015625,
253
+ "learning_rate": 9.70597750796683e-06,
254
+ "loss": 1.476318359375,
255
+ "memory/device_reserved (GiB)": 33.29,
256
+ "memory/max_active (GiB)": 27.26,
257
+ "memory/max_allocated (GiB)": 27.26,
258
+ "ppl": 4.3768,
259
+ "step": 18,
260
+ "tokens/total": 4718592,
261
+ "tokens/train_per_sec_per_gpu": 604.17,
262
+ "tokens/trainable": 4710654
263
+ },
264
+ {
265
+ "epoch": 0.6204081632653061,
266
+ "grad_norm": 0.9453125,
267
+ "learning_rate": 9.663024895924078e-06,
268
+ "loss": 1.2203369140625,
269
+ "memory/device_reserved (GiB)": 33.29,
270
+ "memory/max_active (GiB)": 27.26,
271
+ "memory/max_allocated (GiB)": 27.26,
272
+ "ppl": 3.38833,
273
+ "step": 19,
274
+ "tokens/total": 4980736,
275
+ "tokens/train_per_sec_per_gpu": 605.11,
276
+ "tokens/trainable": 4972132
277
+ },
278
+ {
279
+ "epoch": 0.6530612244897959,
280
+ "grad_norm": 1.0078125,
281
+ "learning_rate": 9.61726611621679e-06,
282
+ "loss": 1.451171875,
283
+ "memory/device_reserved (GiB)": 33.29,
284
+ "memory/max_active (GiB)": 27.26,
285
+ "memory/max_allocated (GiB)": 27.26,
286
+ "ppl": 4.26811,
287
+ "step": 20,
288
+ "tokens/total": 5242880,
289
+ "tokens/train_per_sec_per_gpu": 603.4,
290
+ "tokens/trainable": 5233845
291
+ },
292
+ {
293
+ "epoch": 0.6857142857142857,
294
+ "grad_norm": 0.83203125,
295
+ "learning_rate": 9.568732013437827e-06,
296
+ "loss": 1.3369140625,
297
+ "memory/device_reserved (GiB)": 33.29,
298
+ "memory/max_active (GiB)": 27.26,
299
+ "memory/max_allocated (GiB)": 27.26,
300
+ "ppl": 3.80728,
301
+ "step": 21,
302
+ "tokens/total": 5505024,
303
+ "tokens/train_per_sec_per_gpu": 604.73,
304
+ "tokens/trainable": 5495423
305
+ },
306
+ {
307
+ "epoch": 0.7183673469387755,
308
+ "grad_norm": 0.953125,
309
+ "learning_rate": 9.517455302940388e-06,
310
+ "loss": 1.4034423828125,
311
+ "memory/device_reserved (GiB)": 33.29,
312
+ "memory/max_active (GiB)": 27.26,
313
+ "memory/max_allocated (GiB)": 27.26,
314
+ "ppl": 4.06918,
315
+ "step": 22,
316
+ "tokens/total": 5767168,
317
+ "tokens/train_per_sec_per_gpu": 605.55,
318
+ "tokens/trainable": 5756905
319
+ },
320
+ {
321
+ "epoch": 0.7510204081632653,
322
+ "grad_norm": 0.78125,
323
+ "learning_rate": 9.46347054878559e-06,
324
+ "loss": 1.3974609375,
325
+ "memory/device_reserved (GiB)": 33.29,
326
+ "memory/max_active (GiB)": 27.26,
327
+ "memory/max_allocated (GiB)": 27.26,
328
+ "ppl": 4.04492,
329
+ "step": 23,
330
+ "tokens/total": 6029312,
331
+ "tokens/train_per_sec_per_gpu": 601.68,
332
+ "tokens/trainable": 6018394
333
+ },
334
+ {
335
+ "epoch": 0.7836734693877551,
336
+ "grad_norm": 0.84375,
337
+ "learning_rate": 9.406814140443898e-06,
338
+ "loss": 1.446533203125,
339
+ "memory/device_reserved (GiB)": 33.29,
340
+ "memory/max_active (GiB)": 27.26,
341
+ "memory/max_allocated (GiB)": 27.26,
342
+ "ppl": 4.24836,
343
+ "step": 24,
344
+ "tokens/total": 6291456,
345
+ "tokens/train_per_sec_per_gpu": 602.65,
346
+ "tokens/trainable": 6279921
347
+ },
348
+ {
349
+ "epoch": 0.8163265306122449,
350
+ "grad_norm": 0.94140625,
351
+ "learning_rate": 9.347524268266092e-06,
352
+ "loss": 1.303955078125,
353
+ "memory/device_reserved (GiB)": 33.29,
354
+ "memory/max_active (GiB)": 27.26,
355
+ "memory/max_allocated (GiB)": 27.26,
356
+ "ppl": 3.68384,
357
+ "step": 25,
358
+ "tokens/total": 6553600,
359
+ "tokens/train_per_sec_per_gpu": 605.26,
360
+ "tokens/trainable": 6541489
361
+ },
362
+ {
363
+ "epoch": 0.8489795918367347,
364
+ "grad_norm": 0.9453125,
365
+ "learning_rate": 9.285640897740316e-06,
366
+ "loss": 1.4931640625,
367
+ "memory/device_reserved (GiB)": 33.29,
368
+ "memory/max_active (GiB)": 27.26,
369
+ "memory/max_allocated (GiB)": 27.26,
370
+ "ppl": 4.45116,
371
+ "step": 26,
372
+ "tokens/total": 6815744,
373
+ "tokens/train_per_sec_per_gpu": 603.79,
374
+ "tokens/trainable": 6803233
375
+ },
376
+ {
377
+ "epoch": 0.8816326530612245,
378
+ "grad_norm": 0.953125,
379
+ "learning_rate": 9.22120574255258e-06,
380
+ "loss": 1.2637939453125,
381
+ "memory/device_reserved (GiB)": 33.29,
382
+ "memory/max_active (GiB)": 27.26,
383
+ "memory/max_allocated (GiB)": 27.26,
384
+ "ppl": 3.53882,
385
+ "step": 27,
386
+ "tokens/total": 7077888,
387
+ "tokens/train_per_sec_per_gpu": 578.56,
388
+ "tokens/trainable": 7064799
389
+ },
390
+ {
391
+ "epoch": 0.9142857142857143,
392
+ "grad_norm": 0.75390625,
393
+ "learning_rate": 9.154262236468826e-06,
394
+ "loss": 1.222412109375,
395
+ "memory/device_reserved (GiB)": 33.29,
396
+ "memory/max_active (GiB)": 27.26,
397
+ "memory/max_allocated (GiB)": 27.26,
398
+ "ppl": 3.39537,
399
+ "step": 28,
400
+ "tokens/total": 7340032,
401
+ "tokens/train_per_sec_per_gpu": 605.17,
402
+ "tokens/trainable": 7326396
403
+ },
404
+ {
405
+ "epoch": 0.9469387755102041,
406
+ "grad_norm": 0.83984375,
407
+ "learning_rate": 9.084855504057562e-06,
408
+ "loss": 1.306884765625,
409
+ "memory/device_reserved (GiB)": 33.29,
410
+ "memory/max_active (GiB)": 27.26,
411
+ "memory/max_allocated (GiB)": 27.26,
412
+ "ppl": 3.69465,
413
+ "step": 29,
414
+ "tokens/total": 7602176,
415
+ "tokens/train_per_sec_per_gpu": 606.42,
416
+ "tokens/trainable": 7587940
417
+ },
418
+ {
419
+ "epoch": 0.9795918367346939,
420
+ "grad_norm": 0.83984375,
421
+ "learning_rate": 9.013032330272777e-06,
422
+ "loss": 1.3385009765625,
423
+ "memory/device_reserved (GiB)": 33.29,
424
+ "memory/max_active (GiB)": 27.26,
425
+ "memory/max_allocated (GiB)": 27.26,
426
+ "ppl": 3.81332,
427
+ "step": 30,
428
+ "tokens/total": 7864320,
429
+ "tokens/train_per_sec_per_gpu": 606.34,
430
+ "tokens/trainable": 7849389
431
+ },
432
+ {
433
+ "epoch": 1.0,
434
+ "grad_norm": 0.9375,
435
+ "learning_rate": 8.938841128917622e-06,
436
+ "loss": 1.3125,
437
+ "memory/device_reserved (GiB)": 33.29,
438
+ "memory/max_active (GiB)": 27.26,
439
+ "memory/max_allocated (GiB)": 27.26,
440
+ "ppl": 3.71545,
441
+ "step": 31,
442
+ "tokens/total": 8028160,
443
+ "tokens/train_per_sec_per_gpu": 912.97,
444
+ "tokens/trainable": 8012073
445
+ },
446
+ {
447
+ "epoch": 1.0326530612244897,
448
+ "grad_norm": 0.87109375,
449
+ "learning_rate": 8.86233191001016e-06,
450
+ "loss": 1.4200439453125,
451
+ "memory/device_reserved (GiB)": 33.29,
452
+ "memory/max_active (GiB)": 27.26,
453
+ "memory/max_allocated (GiB)": 27.26,
454
+ "ppl": 4.1373,
455
+ "step": 32,
456
+ "tokens/total": 8290304,
457
+ "tokens/train_per_sec_per_gpu": 584.93,
458
+ "tokens/trainable": 8273995
459
+ },
460
+ {
461
+ "epoch": 1.0653061224489795,
462
+ "grad_norm": 0.890625,
463
+ "learning_rate": 8.783556246073135e-06,
464
+ "loss": 1.3094482421875,
465
+ "memory/device_reserved (GiB)": 33.29,
466
+ "memory/max_active (GiB)": 27.26,
467
+ "memory/max_allocated (GiB)": 27.26,
468
+ "ppl": 3.70413,
469
+ "step": 33,
470
+ "tokens/total": 8552448,
471
+ "tokens/train_per_sec_per_gpu": 607.05,
472
+ "tokens/trainable": 8535646
473
+ },
474
+ {
475
+ "epoch": 1.0979591836734695,
476
+ "grad_norm": 0.9375,
477
+ "learning_rate": 8.702567237370521e-06,
478
+ "loss": 1.317138671875,
479
+ "memory/device_reserved (GiB)": 33.29,
480
+ "memory/max_active (GiB)": 27.26,
481
+ "memory/max_allocated (GiB)": 27.26,
482
+ "ppl": 3.73273,
483
+ "step": 34,
484
+ "tokens/total": 8814592,
485
+ "tokens/train_per_sec_per_gpu": 605.99,
486
+ "tokens/trainable": 8797384
487
+ },
488
+ {
489
+ "epoch": 1.1306122448979592,
490
+ "grad_norm": 0.80078125,
491
+ "learning_rate": 8.619419476114251e-06,
492
+ "loss": 1.3460693359375,
493
+ "memory/device_reserved (GiB)": 33.29,
494
+ "memory/max_active (GiB)": 27.26,
495
+ "memory/max_allocated (GiB)": 27.26,
496
+ "ppl": 3.84229,
497
+ "step": 35,
498
+ "tokens/total": 9076736,
499
+ "tokens/train_per_sec_per_gpu": 606.76,
500
+ "tokens/trainable": 9059142
501
+ },
502
+ {
503
+ "epoch": 1.163265306122449,
504
+ "grad_norm": 0.83984375,
505
+ "learning_rate": 8.534169009665282e-06,
506
+ "loss": 1.353759765625,
507
+ "memory/device_reserved (GiB)": 33.29,
508
+ "memory/max_active (GiB)": 27.26,
509
+ "memory/max_allocated (GiB)": 27.26,
510
+ "ppl": 3.87196,
511
+ "step": 36,
512
+ "tokens/total": 9338880,
513
+ "tokens/train_per_sec_per_gpu": 607.47,
514
+ "tokens/trainable": 9320795
515
+ },
516
+ {
517
+ "epoch": 1.1959183673469387,
518
+ "grad_norm": 0.80859375,
519
+ "learning_rate": 8.446873302753783e-06,
520
+ "loss": 1.2401123046875,
521
+ "memory/device_reserved (GiB)": 33.29,
522
+ "memory/max_active (GiB)": 27.26,
523
+ "memory/max_allocated (GiB)": 27.26,
524
+ "ppl": 3.456,
525
+ "step": 37,
526
+ "tokens/total": 9601024,
527
+ "tokens/train_per_sec_per_gpu": 605.46,
528
+ "tokens/trainable": 9582418
529
+ },
530
+ {
531
+ "epoch": 1.2285714285714286,
532
+ "grad_norm": 0.765625,
533
+ "learning_rate": 8.357591198743923e-06,
534
+ "loss": 1.343017578125,
535
+ "memory/device_reserved (GiB)": 33.29,
536
+ "memory/max_active (GiB)": 27.26,
537
+ "memory/max_allocated (GiB)": 27.26,
538
+ "ppl": 3.83059,
539
+ "step": 38,
540
+ "tokens/total": 9863168,
541
+ "tokens/train_per_sec_per_gpu": 603.72,
542
+ "tokens/trainable": 9844275
543
+ },
544
+ {
545
+ "epoch": 1.2612244897959184,
546
+ "grad_norm": 0.80078125,
547
+ "learning_rate": 8.266382879969356e-06,
548
+ "loss": 1.466552734375,
549
+ "memory/device_reserved (GiB)": 33.29,
550
+ "memory/max_active (GiB)": 27.26,
551
+ "memory/max_allocated (GiB)": 27.26,
552
+ "ppl": 4.33427,
553
+ "step": 39,
554
+ "tokens/total": 10125312,
555
+ "tokens/train_per_sec_per_gpu": 604.71,
556
+ "tokens/trainable": 10106035
557
+ },
558
+ {
559
+ "epoch": 1.2938775510204081,
560
+ "grad_norm": 0.7578125,
561
+ "learning_rate": 8.17330982716615e-06,
562
+ "loss": 1.2305908203125,
563
+ "memory/device_reserved (GiB)": 33.29,
564
+ "memory/max_active (GiB)": 27.26,
565
+ "memory/max_allocated (GiB)": 27.26,
566
+ "ppl": 3.42325,
567
+ "step": 40,
568
+ "tokens/total": 10387456,
569
+ "tokens/train_per_sec_per_gpu": 607.6,
570
+ "tokens/trainable": 10367752
571
+ },
572
+ {
573
+ "epoch": 1.3265306122448979,
574
+ "grad_norm": 0.875,
575
+ "learning_rate": 8.078434778030511e-06,
576
+ "loss": 1.3778076171875,
577
+ "memory/device_reserved (GiB)": 33.29,
578
+ "memory/max_active (GiB)": 27.26,
579
+ "memory/max_allocated (GiB)": 27.26,
580
+ "ppl": 3.9662,
581
+ "step": 41,
582
+ "tokens/total": 10649600,
583
+ "tokens/train_per_sec_per_gpu": 605.02,
584
+ "tokens/trainable": 10629381
585
+ },
586
+ {
587
+ "epoch": 1.3591836734693876,
588
+ "grad_norm": 0.765625,
589
+ "learning_rate": 7.981821684929218e-06,
590
+ "loss": 1.362060546875,
591
+ "memory/device_reserved (GiB)": 33.29,
592
+ "memory/max_active (GiB)": 27.26,
593
+ "memory/max_allocated (GiB)": 27.26,
594
+ "ppl": 3.90423,
595
+ "step": 42,
596
+ "tokens/total": 10911744,
597
+ "tokens/train_per_sec_per_gpu": 603.78,
598
+ "tokens/trainable": 10891035
599
+ },
600
+ {
601
+ "epoch": 1.3918367346938776,
602
+ "grad_norm": 0.8046875,
603
+ "learning_rate": 7.883535671791294e-06,
604
+ "loss": 1.420166015625,
605
+ "memory/device_reserved (GiB)": 33.29,
606
+ "memory/max_active (GiB)": 27.26,
607
+ "memory/max_allocated (GiB)": 27.26,
608
+ "ppl": 4.13781,
609
+ "step": 43,
610
+ "tokens/total": 11173888,
611
+ "tokens/train_per_sec_per_gpu": 604.13,
612
+ "tokens/trainable": 11152707
613
+ },
614
+ {
615
+ "epoch": 1.4244897959183673,
616
+ "grad_norm": 0.83984375,
617
+ "learning_rate": 7.783642990209951e-06,
618
+ "loss": 1.31005859375,
619
+ "memory/device_reserved (GiB)": 33.29,
620
+ "memory/max_active (GiB)": 27.26,
621
+ "memory/max_allocated (GiB)": 27.26,
622
+ "ppl": 3.70639,
623
+ "step": 44,
624
+ "tokens/total": 11436032,
625
+ "tokens/train_per_sec_per_gpu": 605.21,
626
+ "tokens/trainable": 11414371
627
+ },
628
+ {
629
+ "epoch": 1.457142857142857,
630
+ "grad_norm": 0.77734375,
631
+ "learning_rate": 7.682210974784426e-06,
632
+ "loss": 1.2432861328125,
633
+ "memory/device_reserved (GiB)": 33.29,
634
+ "memory/max_active (GiB)": 27.26,
635
+ "memory/max_allocated (GiB)": 27.26,
636
+ "ppl": 3.46699,
637
+ "step": 45,
638
+ "tokens/total": 11698176,
639
+ "tokens/train_per_sec_per_gpu": 605.6,
640
+ "tokens/trainable": 11676035
641
+ },
642
+ {
643
+ "epoch": 1.489795918367347,
644
+ "grad_norm": 0.85546875,
645
+ "learning_rate": 7.579307997731783e-06,
646
+ "loss": 1.4068603515625,
647
+ "memory/device_reserved (GiB)": 33.29,
648
+ "memory/max_active (GiB)": 27.26,
649
+ "memory/max_allocated (GiB)": 27.26,
650
+ "ppl": 4.08312,
651
+ "step": 46,
652
+ "tokens/total": 11960320,
653
+ "tokens/train_per_sec_per_gpu": 603.01,
654
+ "tokens/trainable": 11937768
655
+ },
656
+ {
657
+ "epoch": 1.5224489795918368,
658
+ "grad_norm": 0.83984375,
659
+ "learning_rate": 7.475003422799302e-06,
660
+ "loss": 1.3421630859375,
661
+ "memory/device_reserved (GiB)": 33.29,
662
+ "memory/max_active (GiB)": 27.26,
663
+ "memory/max_allocated (GiB)": 27.26,
664
+ "ppl": 3.82731,
665
+ "step": 47,
666
+ "tokens/total": 12222464,
667
+ "tokens/train_per_sec_per_gpu": 603.3,
668
+ "tokens/trainable": 12199374
669
+ },
670
+ {
671
+ "epoch": 1.5551020408163265,
672
+ "grad_norm": 0.8828125,
673
+ "learning_rate": 7.36936755850849e-06,
674
+ "loss": 1.3466796875,
675
+ "memory/device_reserved (GiB)": 33.29,
676
+ "memory/max_active (GiB)": 27.26,
677
+ "memory/max_allocated (GiB)": 27.26,
678
+ "ppl": 3.84464,
679
+ "step": 48,
680
+ "tokens/total": 12484608,
681
+ "tokens/train_per_sec_per_gpu": 605.0,
682
+ "tokens/trainable": 12461213
683
+ },
684
+ {
685
+ "epoch": 1.5877551020408163,
686
+ "grad_norm": 0.81640625,
687
+ "learning_rate": 7.2624716107622675e-06,
688
+ "loss": 1.404296875,
689
+ "memory/device_reserved (GiB)": 33.29,
690
+ "memory/max_active (GiB)": 27.26,
691
+ "memory/max_allocated (GiB)": 27.26,
692
+ "ppl": 4.07266,
693
+ "step": 49,
694
+ "tokens/total": 12746752,
695
+ "tokens/train_per_sec_per_gpu": 604.6,
696
+ "tokens/trainable": 12722727
697
+ },
698
+ {
699
+ "epoch": 1.620408163265306,
700
+ "grad_norm": 0.73046875,
701
+ "learning_rate": 7.154387634847241e-06,
702
+ "loss": 1.1488037109375,
703
+ "memory/device_reserved (GiB)": 33.29,
704
+ "memory/max_active (GiB)": 27.26,
705
+ "memory/max_allocated (GiB)": 27.26,
706
+ "ppl": 3.15442,
707
+ "step": 50,
708
+ "tokens/total": 13008896,
709
+ "tokens/train_per_sec_per_gpu": 605.31,
710
+ "tokens/trainable": 12984205
711
+ },
712
+ {
713
+ "epoch": 1.6530612244897958,
714
+ "grad_norm": 0.79296875,
715
+ "learning_rate": 7.045188486863449e-06,
716
+ "loss": 1.38818359375,
717
+ "memory/device_reserved (GiB)": 33.29,
718
+ "memory/max_active (GiB)": 27.26,
719
+ "memory/max_allocated (GiB)": 27.26,
720
+ "ppl": 4.00756,
721
+ "step": 51,
722
+ "tokens/total": 13271040,
723
+ "tokens/train_per_sec_per_gpu": 603.17,
724
+ "tokens/trainable": 13245918
725
+ },
726
+ {
727
+ "epoch": 1.6857142857142857,
728
+ "grad_norm": 0.75,
729
+ "learning_rate": 6.9349477746142846e-06,
730
+ "loss": 1.2738037109375,
731
+ "memory/device_reserved (GiB)": 33.29,
732
+ "memory/max_active (GiB)": 27.26,
733
+ "memory/max_allocated (GiB)": 27.26,
734
+ "ppl": 3.57442,
735
+ "step": 52,
736
+ "tokens/total": 13533184,
737
+ "tokens/train_per_sec_per_gpu": 606.57,
738
+ "tokens/trainable": 13507496
739
+ },
740
+ {
741
+ "epoch": 1.7183673469387755,
742
+ "grad_norm": 0.84375,
743
+ "learning_rate": 6.823739807989734e-06,
744
+ "loss": 1.3431396484375,
745
+ "memory/device_reserved (GiB)": 33.29,
746
+ "memory/max_active (GiB)": 27.26,
747
+ "memory/max_allocated (GiB)": 27.26,
748
+ "ppl": 3.83105,
749
+ "step": 53,
750
+ "tokens/total": 13795328,
751
+ "tokens/train_per_sec_per_gpu": 606.26,
752
+ "tokens/trainable": 13768978
753
+ },
754
+ {
755
+ "epoch": 1.7510204081632654,
756
+ "grad_norm": 0.7265625,
757
+ "learning_rate": 6.7116395488763565e-06,
758
+ "loss": 1.3397216796875,
759
+ "memory/device_reserved (GiB)": 33.29,
760
+ "memory/max_active (GiB)": 27.26,
761
+ "memory/max_allocated (GiB)": 27.26,
762
+ "ppl": 3.81798,
763
+ "step": 54,
764
+ "tokens/total": 14057472,
765
+ "tokens/train_per_sec_per_gpu": 602.75,
766
+ "tokens/trainable": 14030467
767
+ },
768
+ {
769
+ "epoch": 1.7836734693877552,
770
+ "grad_norm": 0.7265625,
771
+ "learning_rate": 6.598722560627761e-06,
772
+ "loss": 1.3912353515625,
773
+ "memory/device_reserved (GiB)": 33.29,
774
+ "memory/max_active (GiB)": 27.26,
775
+ "memory/max_allocated (GiB)": 27.26,
776
+ "ppl": 4.01981,
777
+ "step": 55,
778
+ "tokens/total": 14319616,
779
+ "tokens/train_per_sec_per_gpu": 571.54,
780
+ "tokens/trainable": 14291994
781
+ },
782
+ {
783
+ "epoch": 1.816326530612245,
784
+ "grad_norm": 0.8515625,
785
+ "learning_rate": 6.485064957129677e-06,
786
+ "loss": 1.2479248046875,
787
+ "memory/device_reserved (GiB)": 33.29,
788
+ "memory/max_active (GiB)": 27.26,
789
+ "memory/max_allocated (GiB)": 27.26,
790
+ "ppl": 3.48311,
791
+ "step": 56,
792
+ "tokens/total": 14581760,
793
+ "tokens/train_per_sec_per_gpu": 606.07,
794
+ "tokens/trainable": 14553562
795
+ },
796
+ {
797
+ "epoch": 1.8489795918367347,
798
+ "grad_norm": 0.82421875,
799
+ "learning_rate": 6.370743351493899e-06,
800
+ "loss": 1.4447021484375,
801
+ "memory/device_reserved (GiB)": 33.29,
802
+ "memory/max_active (GiB)": 27.26,
803
+ "memory/max_allocated (GiB)": 27.26,
804
+ "ppl": 4.24059,
805
+ "step": 57,
806
+ "tokens/total": 14843904,
807
+ "tokens/train_per_sec_per_gpu": 604.39,
808
+ "tokens/trainable": 14815306
809
+ },
810
+ {
811
+ "epoch": 1.8816326530612244,
812
+ "grad_norm": 0.75390625,
813
+ "learning_rate": 6.255834804415742e-06,
814
+ "loss": 1.2120361328125,
815
+ "memory/device_reserved (GiB)": 33.29,
816
+ "memory/max_active (GiB)": 27.26,
817
+ "memory/max_allocated (GiB)": 27.26,
818
+ "ppl": 3.36032,
819
+ "step": 58,
820
+ "tokens/total": 15106048,
821
+ "tokens/train_per_sec_per_gpu": 606.32,
822
+ "tokens/trainable": 15076872
823
+ },
824
+ {
825
+ "epoch": 1.9142857142857141,
826
+ "grad_norm": 0.83203125,
827
+ "learning_rate": 6.140416772229785e-06,
828
+ "loss": 1.175048828125,
829
+ "memory/device_reserved (GiB)": 33.29,
830
+ "memory/max_active (GiB)": 27.26,
831
+ "memory/max_allocated (GiB)": 27.26,
832
+ "ppl": 3.2383,
833
+ "step": 59,
834
+ "tokens/total": 15368192,
835
+ "tokens/train_per_sec_per_gpu": 604.24,
836
+ "tokens/trainable": 15338469
837
+ },
838
+ {
839
+ "epoch": 1.9469387755102041,
840
+ "grad_norm": 0.71875,
841
+ "learning_rate": 6.0245670546989165e-06,
842
+ "loss": 1.2623291015625,
843
+ "memory/device_reserved (GiB)": 33.29,
844
+ "memory/max_active (GiB)": 27.26,
845
+ "memory/max_allocated (GiB)": 27.26,
846
+ "ppl": 3.53364,
847
+ "step": 60,
848
+ "tokens/total": 15630336,
849
+ "tokens/train_per_sec_per_gpu": 606.62,
850
+ "tokens/trainable": 15600013
851
+ },
852
+ {
853
+ "epoch": 1.9795918367346939,
854
+ "grad_norm": 0.7578125,
855
+ "learning_rate": 5.908363742571915e-06,
856
+ "loss": 1.291015625,
857
+ "memory/device_reserved (GiB)": 33.29,
858
+ "memory/max_active (GiB)": 27.26,
859
+ "memory/max_allocated (GiB)": 27.26,
860
+ "ppl": 3.63648,
861
+ "step": 61,
862
+ "tokens/total": 15892480,
863
+ "tokens/train_per_sec_per_gpu": 606.98,
864
+ "tokens/trainable": 15861462
865
+ },
866
+ {
867
+ "epoch": 2.0,
868
+ "grad_norm": 1.0078125,
869
+ "learning_rate": 5.791885164944844e-06,
870
+ "loss": 1.26220703125,
871
+ "memory/device_reserved (GiB)": 33.29,
872
+ "memory/max_active (GiB)": 27.26,
873
+ "memory/max_allocated (GiB)": 27.26,
874
+ "ppl": 3.53321,
875
+ "step": 62,
876
+ "tokens/total": 16056320,
877
+ "tokens/train_per_sec_per_gpu": 920.2,
878
+ "tokens/trainable": 16024146
879
+ },
880
+ {
881
+ "epoch": 2.0326530612244897,
882
+ "grad_norm": 0.81640625,
883
+ "learning_rate": 5.67520983646182e-06,
884
+ "loss": 1.37841796875,
885
+ "memory/device_reserved (GiB)": 33.29,
886
+ "memory/max_active (GiB)": 27.26,
887
+ "memory/max_allocated (GiB)": 27.26,
888
+ "ppl": 3.96862,
889
+ "step": 63,
890
+ "tokens/total": 16318464,
891
+ "tokens/train_per_sec_per_gpu": 585.03,
892
+ "tokens/trainable": 16286068
893
+ },
894
+ {
895
+ "epoch": 2.0653061224489795,
896
+ "grad_norm": 1.109375,
897
+ "learning_rate": 5.5584164043906895e-06,
898
+ "loss": 1.271728515625,
899
+ "memory/device_reserved (GiB)": 33.29,
900
+ "memory/max_active (GiB)": 27.26,
901
+ "memory/max_allocated (GiB)": 27.26,
902
+ "ppl": 3.56701,
903
+ "step": 64,
904
+ "tokens/total": 16580608,
905
+ "tokens/train_per_sec_per_gpu": 607.21,
906
+ "tokens/trainable": 16547719
907
+ },
908
+ {
909
+ "epoch": 2.0979591836734692,
910
+ "grad_norm": 1.0390625,
911
+ "learning_rate": 5.441583595609312e-06,
912
+ "loss": 1.28076171875,
913
+ "memory/device_reserved (GiB)": 33.29,
914
+ "memory/max_active (GiB)": 27.26,
915
+ "memory/max_allocated (GiB)": 27.26,
916
+ "ppl": 3.59938,
917
+ "step": 65,
918
+ "tokens/total": 16842752,
919
+ "tokens/train_per_sec_per_gpu": 605.49,
920
+ "tokens/trainable": 16809456
921
+ },
922
+ {
923
+ "epoch": 2.130612244897959,
924
+ "grad_norm": 0.86328125,
925
+ "learning_rate": 5.324790163538181e-06,
926
+ "loss": 1.3126220703125,
927
+ "memory/device_reserved (GiB)": 33.29,
928
+ "memory/max_active (GiB)": 27.26,
929
+ "memory/max_allocated (GiB)": 27.26,
930
+ "ppl": 3.7159,
931
+ "step": 66,
932
+ "tokens/total": 17104896,
933
+ "tokens/train_per_sec_per_gpu": 606.91,
934
+ "tokens/trainable": 17071218
935
+ },
936
+ {
937
+ "epoch": 2.163265306122449,
938
+ "grad_norm": 0.84765625,
939
+ "learning_rate": 5.208114835055157e-06,
940
+ "loss": 1.32037353515625,
941
+ "memory/device_reserved (GiB)": 33.29,
942
+ "memory/max_active (GiB)": 27.26,
943
+ "memory/max_allocated (GiB)": 27.26,
944
+ "ppl": 3.74482,
945
+ "step": 67,
946
+ "tokens/total": 17367040,
947
+ "tokens/train_per_sec_per_gpu": 607.11,
948
+ "tokens/trainable": 17332876
949
+ },
950
+ {
951
+ "epoch": 2.195918367346939,
952
+ "grad_norm": 0.72265625,
953
+ "learning_rate": 5.0916362574280864e-06,
954
+ "loss": 1.2105712890625,
955
+ "memory/device_reserved (GiB)": 33.29,
956
+ "memory/max_active (GiB)": 27.26,
957
+ "memory/max_allocated (GiB)": 27.26,
958
+ "ppl": 3.3554,
959
+ "step": 68,
960
+ "tokens/total": 17629184,
961
+ "tokens/train_per_sec_per_gpu": 605.15,
962
+ "tokens/trainable": 17594500
963
+ },
964
+ {
965
+ "epoch": 2.2285714285714286,
966
+ "grad_norm": 0.8515625,
967
+ "learning_rate": 4.975432945301085e-06,
968
+ "loss": 1.3128662109375,
969
+ "memory/device_reserved (GiB)": 33.29,
970
+ "memory/max_active (GiB)": 27.26,
971
+ "memory/max_allocated (GiB)": 27.26,
972
+ "ppl": 3.71681,
973
+ "step": 69,
974
+ "tokens/total": 17891328,
975
+ "tokens/train_per_sec_per_gpu": 603.11,
976
+ "tokens/trainable": 17856356
977
+ },
978
+ {
979
+ "epoch": 2.2612244897959184,
980
+ "grad_norm": 0.78515625,
981
+ "learning_rate": 4.859583227770218e-06,
982
+ "loss": 1.4384765625,
983
+ "memory/device_reserved (GiB)": 33.29,
984
+ "memory/max_active (GiB)": 27.26,
985
+ "memory/max_allocated (GiB)": 27.26,
986
+ "ppl": 4.21427,
987
+ "step": 70,
988
+ "tokens/total": 18153472,
989
+ "tokens/train_per_sec_per_gpu": 604.33,
990
+ "tokens/trainable": 18118114
991
+ },
992
+ {
993
+ "epoch": 2.293877551020408,
994
+ "grad_norm": 0.80859375,
995
+ "learning_rate": 4.744165195584258e-06,
996
+ "loss": 1.203125,
997
+ "memory/device_reserved (GiB)": 33.29,
998
+ "memory/max_active (GiB)": 27.26,
999
+ "memory/max_allocated (GiB)": 27.26,
1000
+ "ppl": 3.33051,
1001
+ "step": 71,
1002
+ "tokens/total": 18415616,
1003
+ "tokens/train_per_sec_per_gpu": 606.6,
1004
+ "tokens/trainable": 18379828
1005
+ },
1006
+ {
1007
+ "epoch": 2.326530612244898,
1008
+ "grad_norm": 0.82421875,
1009
+ "learning_rate": 4.6292566485061015e-06,
1010
+ "loss": 1.352294921875,
1011
+ "memory/device_reserved (GiB)": 33.29,
1012
+ "memory/max_active (GiB)": 27.26,
1013
+ "memory/max_allocated (GiB)": 27.26,
1014
+ "ppl": 3.86629,
1015
+ "step": 72,
1016
+ "tokens/total": 18677760,
1017
+ "tokens/train_per_sec_per_gpu": 606.39,
1018
+ "tokens/trainable": 18641452
1019
+ },
1020
+ {
1021
+ "epoch": 2.3591836734693876,
1022
+ "grad_norm": 0.78125,
1023
+ "learning_rate": 4.514935042870324e-06,
1024
+ "loss": 1.3382568359375,
1025
+ "memory/device_reserved (GiB)": 33.29,
1026
+ "memory/max_active (GiB)": 27.26,
1027
+ "memory/max_allocated (GiB)": 27.26,
1028
+ "ppl": 3.81239,
1029
+ "step": 73,
1030
+ "tokens/total": 18939904,
1031
+ "tokens/train_per_sec_per_gpu": 602.92,
1032
+ "tokens/trainable": 18903104
1033
+ },
1034
+ {
1035
+ "epoch": 2.3918367346938774,
1036
+ "grad_norm": 0.78125,
1037
+ "learning_rate": 4.40127743937224e-06,
1038
+ "loss": 1.396728515625,
1039
+ "memory/device_reserved (GiB)": 33.29,
1040
+ "memory/max_active (GiB)": 27.26,
1041
+ "memory/max_allocated (GiB)": 27.26,
1042
+ "ppl": 4.04196,
1043
+ "step": 74,
1044
+ "tokens/total": 19202048,
1045
+ "tokens/train_per_sec_per_gpu": 603.33,
1046
+ "tokens/trainable": 19164778
1047
+ },
1048
+ {
1049
+ "epoch": 2.424489795918367,
1050
+ "grad_norm": 0.71875,
1051
+ "learning_rate": 4.288360451123646e-06,
1052
+ "loss": 1.287841796875,
1053
+ "memory/device_reserved (GiB)": 33.29,
1054
+ "memory/max_active (GiB)": 27.26,
1055
+ "memory/max_allocated (GiB)": 27.26,
1056
+ "ppl": 3.62495,
1057
+ "step": 75,
1058
+ "tokens/total": 19464192,
1059
+ "tokens/train_per_sec_per_gpu": 603.18,
1060
+ "tokens/trainable": 19426440
1061
+ },
1062
+ {
1063
+ "epoch": 2.4571428571428573,
1064
+ "grad_norm": 0.81640625,
1065
+ "learning_rate": 4.1762601920102675e-06,
1066
+ "loss": 1.22216796875,
1067
+ "memory/device_reserved (GiB)": 33.29,
1068
+ "memory/max_active (GiB)": 27.26,
1069
+ "memory/max_allocated (GiB)": 27.26,
1070
+ "ppl": 3.39454,
1071
+ "step": 76,
1072
+ "tokens/total": 19726336,
1073
+ "tokens/train_per_sec_per_gpu": 606.31,
1074
+ "tokens/trainable": 19688104
1075
+ },
1076
+ {
1077
+ "epoch": 2.489795918367347,
1078
+ "grad_norm": 0.76171875,
1079
+ "learning_rate": 4.065052225385717e-06,
1080
+ "loss": 1.3858642578125,
1081
+ "memory/device_reserved (GiB)": 33.29,
1082
+ "memory/max_active (GiB)": 27.26,
1083
+ "memory/max_allocated (GiB)": 27.26,
1084
+ "ppl": 3.99828,
1085
+ "step": 77,
1086
+ "tokens/total": 19988480,
1087
+ "tokens/train_per_sec_per_gpu": 603.6,
1088
+ "tokens/trainable": 19949836
1089
+ },
1090
+ {
1091
+ "epoch": 2.522448979591837,
1092
+ "grad_norm": 0.74609375,
1093
+ "learning_rate": 3.954811513136554e-06,
1094
+ "loss": 1.32080078125,
1095
+ "memory/device_reserved (GiB)": 33.29,
1096
+ "memory/max_active (GiB)": 27.26,
1097
+ "memory/max_allocated (GiB)": 27.26,
1098
+ "ppl": 3.74642,
1099
+ "step": 78,
1100
+ "tokens/total": 20250624,
1101
+ "tokens/train_per_sec_per_gpu": 603.42,
1102
+ "tokens/trainable": 20211444
1103
+ },
1104
+ {
1105
+ "epoch": 2.5551020408163265,
1106
+ "grad_norm": 1.015625,
1107
+ "learning_rate": 3.84561236515276e-06,
1108
+ "loss": 1.3240966796875,
1109
+ "memory/device_reserved (GiB)": 33.29,
1110
+ "memory/max_active (GiB)": 27.26,
1111
+ "memory/max_allocated (GiB)": 27.26,
1112
+ "ppl": 3.75879,
1113
+ "step": 79,
1114
+ "tokens/total": 20512768,
1115
+ "tokens/train_per_sec_per_gpu": 603.23,
1116
+ "tokens/trainable": 20473284
1117
+ },
1118
+ {
1119
+ "epoch": 2.5877551020408163,
1120
+ "grad_norm": 0.703125,
1121
+ "learning_rate": 3.7375283892377344e-06,
1122
+ "loss": 1.385986328125,
1123
+ "memory/device_reserved (GiB)": 33.29,
1124
+ "memory/max_active (GiB)": 27.26,
1125
+ "memory/max_allocated (GiB)": 27.26,
1126
+ "ppl": 3.99877,
1127
+ "step": 80,
1128
+ "tokens/total": 20774912,
1129
+ "tokens/train_per_sec_per_gpu": 604.17,
1130
+ "tokens/trainable": 20734804
1131
+ },
1132
+ {
1133
+ "epoch": 2.620408163265306,
1134
+ "grad_norm": 0.7109375,
1135
+ "learning_rate": 3.630632441491512e-06,
1136
+ "loss": 1.130859375,
1137
+ "memory/device_reserved (GiB)": 33.29,
1138
+ "memory/max_active (GiB)": 27.26,
1139
+ "memory/max_allocated (GiB)": 27.26,
1140
+ "ppl": 3.09832,
1141
+ "step": 81,
1142
+ "tokens/total": 21037056,
1143
+ "tokens/train_per_sec_per_gpu": 606.12,
1144
+ "tokens/trainable": 20996280
1145
+ },
1146
+ {
1147
+ "epoch": 2.6530612244897958,
1148
+ "grad_norm": 0.85546875,
1149
+ "learning_rate": 3.5249965772007e-06,
1150
+ "loss": 1.3719482421875,
1151
+ "memory/device_reserved (GiB)": 33.29,
1152
+ "memory/max_active (GiB)": 27.26,
1153
+ "memory/max_allocated (GiB)": 27.26,
1154
+ "ppl": 3.94303,
1155
+ "step": 82,
1156
+ "tokens/total": 21299200,
1157
+ "tokens/train_per_sec_per_gpu": 603.84,
1158
+ "tokens/trainable": 21257992
1159
+ },
1160
+ {
1161
+ "epoch": 2.685714285714286,
1162
+ "grad_norm": 0.703125,
1163
+ "learning_rate": 3.4206920022682173e-06,
1164
+ "loss": 1.2578125,
1165
+ "memory/device_reserved (GiB)": 33.29,
1166
+ "memory/max_active (GiB)": 27.26,
1167
+ "memory/max_allocated (GiB)": 27.26,
1168
+ "ppl": 3.51772,
1169
+ "step": 83,
1170
+ "tokens/total": 21561344,
1171
+ "tokens/train_per_sec_per_gpu": 605.93,
1172
+ "tokens/trainable": 21519576
1173
+ },
1174
+ {
1175
+ "epoch": 2.7183673469387752,
1176
+ "grad_norm": 0.85546875,
1177
+ "learning_rate": 3.3177890252155755e-06,
1178
+ "loss": 1.3260498046875,
1179
+ "memory/device_reserved (GiB)": 33.29,
1180
+ "memory/max_active (GiB)": 27.26,
1181
+ "memory/max_allocated (GiB)": 27.26,
1182
+ "ppl": 3.76614,
1183
+ "step": 84,
1184
+ "tokens/total": 21823488,
1185
+ "tokens/train_per_sec_per_gpu": 604.99,
1186
+ "tokens/trainable": 21781056
1187
+ },
1188
+ {
1189
+ "epoch": 2.7510204081632654,
1190
+ "grad_norm": 0.8203125,
1191
+ "learning_rate": 3.2163570097900497e-06,
1192
+ "loss": 1.324951171875,
1193
+ "memory/device_reserved (GiB)": 33.29,
1194
+ "memory/max_active (GiB)": 27.26,
1195
+ "memory/max_allocated (GiB)": 27.26,
1196
+ "ppl": 3.762,
1197
+ "step": 85,
1198
+ "tokens/total": 22085632,
1199
+ "tokens/train_per_sec_per_gpu": 585.4,
1200
+ "tokens/trainable": 22042546
1201
+ },
1202
+ {
1203
+ "epoch": 2.783673469387755,
1204
+ "grad_norm": 0.73828125,
1205
+ "learning_rate": 3.116464328208708e-06,
1206
+ "loss": 1.3780517578125,
1207
+ "memory/device_reserved (GiB)": 33.29,
1208
+ "memory/max_active (GiB)": 27.26,
1209
+ "memory/max_allocated (GiB)": 27.26,
1210
+ "ppl": 3.96717,
1211
+ "step": 86,
1212
+ "tokens/total": 22347776,
1213
+ "tokens/train_per_sec_per_gpu": 586.69,
1214
+ "tokens/trainable": 22304072
1215
+ },
1216
+ {
1217
+ "epoch": 2.816326530612245,
1218
+ "grad_norm": 0.80859375,
1219
+ "learning_rate": 3.0181783150707827e-06,
1220
+ "loss": 1.23388671875,
1221
+ "memory/device_reserved (GiB)": 33.29,
1222
+ "memory/max_active (GiB)": 27.26,
1223
+ "memory/max_allocated (GiB)": 27.26,
1224
+ "ppl": 3.43455,
1225
+ "step": 87,
1226
+ "tokens/total": 22609920,
1227
+ "tokens/train_per_sec_per_gpu": 605.15,
1228
+ "tokens/trainable": 22565640
1229
+ },
1230
+ {
1231
+ "epoch": 2.8489795918367347,
1232
+ "grad_norm": 0.79296875,
1233
+ "learning_rate": 2.921565221969492e-06,
1234
+ "loss": 1.430908203125,
1235
+ "memory/device_reserved (GiB)": 33.29,
1236
+ "memory/max_active (GiB)": 27.26,
1237
+ "memory/max_allocated (GiB)": 27.26,
1238
+ "ppl": 4.1825,
1239
+ "step": 88,
1240
+ "tokens/total": 22872064,
1241
+ "tokens/train_per_sec_per_gpu": 603.72,
1242
+ "tokens/trainable": 22827386
1243
+ },
1244
+ {
1245
+ "epoch": 2.8816326530612244,
1246
+ "grad_norm": 0.74609375,
1247
+ "learning_rate": 2.8266901728338526e-06,
1248
+ "loss": 1.19873046875,
1249
+ "memory/device_reserved (GiB)": 33.29,
1250
+ "memory/max_active (GiB)": 27.26,
1251
+ "memory/max_allocated (GiB)": 27.26,
1252
+ "ppl": 3.3159,
1253
+ "step": 89,
1254
+ "tokens/total": 23134208,
1255
+ "tokens/train_per_sec_per_gpu": 606.04,
1256
+ "tokens/trainable": 23088952
1257
+ },
1258
+ {
1259
+ "epoch": 2.914285714285714,
1260
+ "grad_norm": 0.6953125,
1261
+ "learning_rate": 2.7336171200306467e-06,
1262
+ "loss": 1.1632080078125,
1263
+ "memory/device_reserved (GiB)": 33.29,
1264
+ "memory/max_active (GiB)": 27.26,
1265
+ "memory/max_allocated (GiB)": 27.26,
1266
+ "ppl": 3.20018,
1267
+ "step": 90,
1268
+ "tokens/total": 23396352,
1269
+ "tokens/train_per_sec_per_gpu": 604.92,
1270
+ "tokens/trainable": 23350548
1271
+ },
1272
+ {
1273
+ "epoch": 2.946938775510204,
1274
+ "grad_norm": 0.875,
1275
+ "learning_rate": 2.6424088012560766e-06,
1276
+ "loss": 1.2506103515625,
1277
+ "memory/device_reserved (GiB)": 33.29,
1278
+ "memory/max_active (GiB)": 27.26,
1279
+ "memory/max_allocated (GiB)": 27.26,
1280
+ "ppl": 3.49247,
1281
+ "step": 91,
1282
+ "tokens/total": 23658496,
1283
+ "tokens/train_per_sec_per_gpu": 605.58,
1284
+ "tokens/trainable": 23612092
1285
+ },
1286
+ {
1287
+ "epoch": 2.979591836734694,
1288
+ "grad_norm": 0.859375,
1289
+ "learning_rate": 2.5531266972462176e-06,
1290
+ "loss": 1.2801513671875,
1291
+ "memory/device_reserved (GiB)": 33.29,
1292
+ "memory/max_active (GiB)": 27.26,
1293
+ "memory/max_allocated (GiB)": 27.26,
1294
+ "ppl": 3.59718,
1295
+ "step": 92,
1296
+ "tokens/total": 23920640,
1297
+ "tokens/train_per_sec_per_gpu": 606.4,
1298
+ "tokens/trainable": 23873538
1299
+ },
1300
+ {
1301
+ "epoch": 3.0,
1302
+ "grad_norm": 0.94140625,
1303
+ "learning_rate": 2.4658309903347196e-06,
1304
+ "loss": 1.248046875,
1305
+ "memory/device_reserved (GiB)": 33.29,
1306
+ "memory/max_active (GiB)": 27.26,
1307
+ "memory/max_allocated (GiB)": 27.26,
1308
+ "ppl": 3.48353,
1309
+ "step": 93,
1310
+ "tokens/total": 24084480,
1311
+ "tokens/train_per_sec_per_gpu": 922.28,
1312
+ "tokens/trainable": 24036224
1313
+ },
1314
+ {
1315
+ "epoch": 3.0326530612244897,
1316
+ "grad_norm": 0.7109375,
1317
+ "learning_rate": 2.380580523885751e-06,
1318
+ "loss": 1.3673095703125,
1319
+ "memory/device_reserved (GiB)": 33.29,
1320
+ "memory/max_active (GiB)": 27.26,
1321
+ "memory/max_allocated (GiB)": 27.26,
1322
+ "ppl": 3.92478,
1323
+ "step": 94,
1324
+ "tokens/total": 24346624,
1325
+ "tokens/train_per_sec_per_gpu": 586.15,
1326
+ "tokens/trainable": 24298144
1327
+ },
1328
+ {
1329
+ "epoch": 3.0653061224489795,
1330
+ "grad_norm": 0.67578125,
1331
+ "learning_rate": 2.29743276262948e-06,
1332
+ "loss": 1.2625732421875,
1333
+ "memory/device_reserved (GiB)": 33.29,
1334
+ "memory/max_active (GiB)": 27.26,
1335
+ "memory/max_allocated (GiB)": 27.26,
1336
+ "ppl": 3.5345,
1337
+ "step": 95,
1338
+ "tokens/total": 24608768,
1339
+ "tokens/train_per_sec_per_gpu": 607.42,
1340
+ "tokens/trainable": 24559796
1341
+ },
1342
+ {
1343
+ "epoch": 3.0979591836734692,
1344
+ "grad_norm": 0.84375,
1345
+ "learning_rate": 2.2164437539268652e-06,
1346
+ "loss": 1.2720947265625,
1347
+ "memory/device_reserved (GiB)": 33.29,
1348
+ "memory/max_active (GiB)": 27.26,
1349
+ "memory/max_allocated (GiB)": 27.26,
1350
+ "ppl": 3.56832,
1351
+ "step": 96,
1352
+ "tokens/total": 24870912,
1353
+ "tokens/train_per_sec_per_gpu": 605.03,
1354
+ "tokens/trainable": 24821538
1355
+ },
1356
+ {
1357
+ "epoch": 3.130612244897959,
1358
+ "grad_norm": 0.69921875,
1359
+ "learning_rate": 2.1376680899898415e-06,
1360
+ "loss": 1.3033447265625,
1361
+ "memory/device_reserved (GiB)": 33.29,
1362
+ "memory/max_active (GiB)": 27.26,
1363
+ "memory/max_allocated (GiB)": 27.26,
1364
+ "ppl": 3.68159,
1365
+ "step": 97,
1366
+ "tokens/total": 25133056,
1367
+ "tokens/train_per_sec_per_gpu": 607.5,
1368
+ "tokens/trainable": 25083298
1369
+ },
1370
+ {
1371
+ "epoch": 3.163265306122449,
1372
+ "grad_norm": 0.73828125,
1373
+ "learning_rate": 2.0611588710823797e-06,
1374
+ "loss": 1.31024169921875,
1375
+ "memory/device_reserved (GiB)": 33.29,
1376
+ "memory/max_active (GiB)": 27.26,
1377
+ "memory/max_allocated (GiB)": 27.26,
1378
+ "ppl": 3.70707,
1379
+ "step": 98,
1380
+ "tokens/total": 25395200,
1381
+ "tokens/train_per_sec_per_gpu": 607.32,
1382
+ "tokens/trainable": 25344956
1383
+ },
1384
+ {
1385
+ "epoch": 3.195918367346939,
1386
+ "grad_norm": 0.796875,
1387
+ "learning_rate": 1.986967669727224e-06,
1388
+ "loss": 1.20263671875,
1389
+ "memory/device_reserved (GiB)": 33.29,
1390
+ "memory/max_active (GiB)": 27.26,
1391
+ "memory/max_allocated (GiB)": 27.26,
1392
+ "ppl": 3.32888,
1393
+ "step": 99,
1394
+ "tokens/total": 25657344,
1395
+ "tokens/train_per_sec_per_gpu": 604.43,
1396
+ "tokens/trainable": 25606580
1397
+ },
1398
+ {
1399
+ "epoch": 3.2285714285714286,
1400
+ "grad_norm": 0.69921875,
1401
+ "learning_rate": 1.9151444959424383e-06,
1402
+ "loss": 1.3046875,
1403
+ "memory/device_reserved (GiB)": 33.29,
1404
+ "memory/max_active (GiB)": 27.26,
1405
+ "memory/max_allocated (GiB)": 27.26,
1406
+ "ppl": 3.68654,
1407
+ "step": 100,
1408
+ "tokens/total": 25919488,
1409
+ "tokens/train_per_sec_per_gpu": 603.08,
1410
+ "tokens/trainable": 25868436
1411
+ },
1412
+ {
1413
+ "epoch": 3.2612244897959184,
1414
+ "grad_norm": 0.78515625,
1415
+ "learning_rate": 1.8457377635311763e-06,
1416
+ "loss": 1.431396484375,
1417
+ "memory/device_reserved (GiB)": 33.29,
1418
+ "memory/max_active (GiB)": 27.26,
1419
+ "memory/max_allocated (GiB)": 27.26,
1420
+ "ppl": 4.18454,
1421
+ "step": 101,
1422
+ "tokens/total": 26181632,
1423
+ "tokens/train_per_sec_per_gpu": 603.9,
1424
+ "tokens/trainable": 26130194
1425
+ },
1426
+ {
1427
+ "epoch": 3.293877551020408,
1428
+ "grad_norm": 1.1171875,
1429
+ "learning_rate": 1.7787942574474215e-06,
1430
+ "loss": 1.19580078125,
1431
+ "memory/device_reserved (GiB)": 33.29,
1432
+ "memory/max_active (GiB)": 27.26,
1433
+ "memory/max_allocated (GiB)": 27.26,
1434
+ "ppl": 3.3062,
1435
+ "step": 102,
1436
+ "tokens/total": 26443776,
1437
+ "tokens/train_per_sec_per_gpu": 606.74,
1438
+ "tokens/trainable": 26391908
1439
+ },
1440
+ {
1441
+ "epoch": 3.326530612244898,
1442
+ "grad_norm": 0.70703125,
1443
+ "learning_rate": 1.7143591022596846e-06,
1444
+ "loss": 1.34716796875,
1445
+ "memory/device_reserved (GiB)": 33.29,
1446
+ "memory/max_active (GiB)": 27.26,
1447
+ "memory/max_allocated (GiB)": 27.26,
1448
+ "ppl": 3.84652,
1449
+ "step": 103,
1450
+ "tokens/total": 26705920,
1451
+ "tokens/train_per_sec_per_gpu": 606.45,
1452
+ "tokens/trainable": 26653532
1453
+ },
1454
+ {
1455
+ "epoch": 3.3591836734693876,
1456
+ "grad_norm": 0.71484375,
1457
+ "learning_rate": 1.6524757317339102e-06,
1458
+ "loss": 1.3314208984375,
1459
+ "memory/device_reserved (GiB)": 33.29,
1460
+ "memory/max_active (GiB)": 27.26,
1461
+ "memory/max_allocated (GiB)": 27.26,
1462
+ "ppl": 3.78642,
1463
+ "step": 104,
1464
+ "tokens/total": 26968064,
1465
+ "tokens/train_per_sec_per_gpu": 602.49,
1466
+ "tokens/trainable": 26915184
1467
+ },
1468
+ {
1469
+ "epoch": 3.3918367346938774,
1470
+ "grad_norm": 2.484375,
1471
+ "learning_rate": 1.593185859556103e-06,
1472
+ "loss": 1.3916015625,
1473
+ "memory/device_reserved (GiB)": 33.29,
1474
+ "memory/max_active (GiB)": 27.26,
1475
+ "memory/max_allocated (GiB)": 27.26,
1476
+ "ppl": 4.02129,
1477
+ "step": 105,
1478
+ "tokens/total": 27230208,
1479
+ "tokens/train_per_sec_per_gpu": 603.38,
1480
+ "tokens/trainable": 27176858
1481
+ },
1482
+ {
1483
+ "epoch": 3.424489795918367,
1484
+ "grad_norm": 0.71875,
1485
+ "learning_rate": 1.5365294512144114e-06,
1486
+ "loss": 1.282958984375,
1487
+ "memory/device_reserved (GiB)": 33.29,
1488
+ "memory/max_active (GiB)": 27.26,
1489
+ "memory/max_allocated (GiB)": 27.26,
1490
+ "ppl": 3.6073,
1491
+ "step": 106,
1492
+ "tokens/total": 27492352,
1493
+ "tokens/train_per_sec_per_gpu": 604.12,
1494
+ "tokens/trainable": 27438520
1495
+ },
1496
+ {
1497
+ "epoch": 3.4571428571428573,
1498
+ "grad_norm": 0.6640625,
1499
+ "learning_rate": 1.4825446970596136e-06,
1500
+ "loss": 1.2174072265625,
1501
+ "memory/device_reserved (GiB)": 33.29,
1502
+ "memory/max_active (GiB)": 27.26,
1503
+ "memory/max_allocated (GiB)": 27.26,
1504
+ "ppl": 3.37842,
1505
+ "step": 107,
1506
+ "tokens/total": 27754496,
1507
+ "tokens/train_per_sec_per_gpu": 606.17,
1508
+ "tokens/trainable": 27700184
1509
+ },
1510
+ {
1511
+ "epoch": 3.489795918367347,
1512
+ "grad_norm": 0.73046875,
1513
+ "learning_rate": 1.4312679865621742e-06,
1514
+ "loss": 1.380615234375,
1515
+ "memory/device_reserved (GiB)": 33.29,
1516
+ "memory/max_active (GiB)": 27.26,
1517
+ "memory/max_allocated (GiB)": 27.26,
1518
+ "ppl": 3.97735,
1519
+ "step": 108,
1520
+ "tokens/total": 28016640,
1521
+ "tokens/train_per_sec_per_gpu": 602.64,
1522
+ "tokens/trainable": 27961916
1523
+ },
1524
+ {
1525
+ "epoch": 3.522448979591837,
1526
+ "grad_norm": 0.6953125,
1527
+ "learning_rate": 1.382733883783211e-06,
1528
+ "loss": 1.3165283203125,
1529
+ "memory/device_reserved (GiB)": 33.29,
1530
+ "memory/max_active (GiB)": 27.26,
1531
+ "memory/max_allocated (GiB)": 27.26,
1532
+ "ppl": 3.73045,
1533
+ "step": 109,
1534
+ "tokens/total": 28278784,
1535
+ "tokens/train_per_sec_per_gpu": 603.31,
1536
+ "tokens/trainable": 28223524
1537
+ },
1538
+ {
1539
+ "epoch": 3.5551020408163265,
1540
+ "grad_norm": 0.71484375,
1541
+ "learning_rate": 1.3369751040759236e-06,
1542
+ "loss": 1.3199462890625,
1543
+ "memory/device_reserved (GiB)": 33.29,
1544
+ "memory/max_active (GiB)": 27.26,
1545
+ "memory/max_allocated (GiB)": 27.26,
1546
+ "ppl": 3.74322,
1547
+ "step": 110,
1548
+ "tokens/total": 28540928,
1549
+ "tokens/train_per_sec_per_gpu": 602.96,
1550
+ "tokens/trainable": 28485364
1551
+ },
1552
+ {
1553
+ "epoch": 3.5877551020408163,
1554
+ "grad_norm": 0.6953125,
1555
+ "learning_rate": 1.2940224920331707e-06,
1556
+ "loss": 1.3828125,
1557
+ "memory/device_reserved (GiB)": 33.29,
1558
+ "memory/max_active (GiB)": 27.26,
1559
+ "memory/max_allocated (GiB)": 27.26,
1560
+ "ppl": 3.9861,
1561
+ "step": 111,
1562
+ "tokens/total": 28803072,
1563
+ "tokens/train_per_sec_per_gpu": 603.97,
1564
+ "tokens/trainable": 28746884
1565
+ },
1566
+ {
1567
+ "epoch": 3.620408163265306,
1568
+ "grad_norm": 0.6640625,
1569
+ "learning_rate": 1.2539050006960814e-06,
1570
+ "loss": 1.1270751953125,
1571
+ "memory/device_reserved (GiB)": 33.29,
1572
+ "memory/max_active (GiB)": 27.26,
1573
+ "memory/max_allocated (GiB)": 27.26,
1574
+ "ppl": 3.08662,
1575
+ "step": 112,
1576
+ "tokens/total": 29065216,
1577
+ "tokens/train_per_sec_per_gpu": 605.96,
1578
+ "tokens/trainable": 29008360
1579
+ },
1580
+ {
1581
+ "epoch": 3.6530612244897958,
1582
+ "grad_norm": 0.7421875,
1583
+ "learning_rate": 1.2166496720376874e-06,
1584
+ "loss": 1.368408203125,
1585
+ "memory/device_reserved (GiB)": 33.29,
1586
+ "memory/max_active (GiB)": 27.26,
1587
+ "memory/max_allocated (GiB)": 27.26,
1588
+ "ppl": 3.92909,
1589
+ "step": 113,
1590
+ "tokens/total": 29327360,
1591
+ "tokens/train_per_sec_per_gpu": 602.33,
1592
+ "tokens/trainable": 29270072
1593
+ },
1594
+ {
1595
+ "epoch": 3.685714285714286,
1596
+ "grad_norm": 0.67578125,
1597
+ "learning_rate": 1.1822816187347625e-06,
1598
+ "loss": 1.2537841796875,
1599
+ "memory/device_reserved (GiB)": 33.29,
1600
+ "memory/max_active (GiB)": 27.26,
1601
+ "memory/max_allocated (GiB)": 27.26,
1602
+ "ppl": 3.50358,
1603
+ "step": 114,
1604
+ "tokens/total": 29589504,
1605
+ "tokens/train_per_sec_per_gpu": 606.12,
1606
+ "tokens/trainable": 29531656
1607
+ },
1608
+ {
1609
+ "epoch": 3.7183673469387752,
1610
+ "grad_norm": 0.703125,
1611
+ "learning_rate": 1.1508240072401336e-06,
1612
+ "loss": 1.3232421875,
1613
+ "memory/device_reserved (GiB)": 33.29,
1614
+ "memory/max_active (GiB)": 27.26,
1615
+ "memory/max_allocated (GiB)": 27.26,
1616
+ "ppl": 3.75558,
1617
+ "step": 115,
1618
+ "tokens/total": 29851648,
1619
+ "tokens/train_per_sec_per_gpu": 605.25,
1620
+ "tokens/trainable": 29793136
1621
+ },
1622
+ {
1623
+ "epoch": 3.7510204081632654,
1624
+ "grad_norm": 0.70703125,
1625
+ "learning_rate": 1.1222980421668874e-06,
1626
+ "loss": 1.3226318359375,
1627
+ "memory/device_reserved (GiB)": 33.29,
1628
+ "memory/max_active (GiB)": 27.26,
1629
+ "memory/max_allocated (GiB)": 27.26,
1630
+ "ppl": 3.75329,
1631
+ "step": 116,
1632
+ "tokens/total": 30113792,
1633
+ "tokens/train_per_sec_per_gpu": 570.96,
1634
+ "tokens/trainable": 30054626
1635
+ },
1636
+ {
1637
+ "epoch": 3.783673469387755,
1638
+ "grad_norm": 1.328125,
1639
+ "learning_rate": 1.0967229519949833e-06,
1640
+ "loss": 1.3746337890625,
1641
+ "memory/device_reserved (GiB)": 33.29,
1642
+ "memory/max_active (GiB)": 27.26,
1643
+ "memory/max_allocated (GiB)": 27.26,
1644
+ "ppl": 3.95363,
1645
+ "step": 117,
1646
+ "tokens/total": 30375936,
1647
+ "tokens/train_per_sec_per_gpu": 602.33,
1648
+ "tokens/trainable": 30316152
1649
+ },
1650
+ {
1651
+ "epoch": 3.816326530612245,
1652
+ "grad_norm": 0.71875,
1653
+ "learning_rate": 1.0741159761099294e-06,
1654
+ "loss": 1.231201171875,
1655
+ "memory/device_reserved (GiB)": 33.29,
1656
+ "memory/max_active (GiB)": 27.26,
1657
+ "memory/max_allocated (GiB)": 27.26,
1658
+ "ppl": 3.42534,
1659
+ "step": 118,
1660
+ "tokens/total": 30638080,
1661
+ "tokens/train_per_sec_per_gpu": 605.12,
1662
+ "tokens/trainable": 30577720
1663
+ },
1664
+ {
1665
+ "epoch": 3.8489795918367347,
1666
+ "grad_norm": 0.80078125,
1667
+ "learning_rate": 1.054492353182237e-06,
1668
+ "loss": 1.4288330078125,
1669
+ "memory/device_reserved (GiB)": 33.29,
1670
+ "memory/max_active (GiB)": 27.26,
1671
+ "memory/max_allocated (GiB)": 27.26,
1672
+ "ppl": 4.17383,
1673
+ "step": 119,
1674
+ "tokens/total": 30900224,
1675
+ "tokens/train_per_sec_per_gpu": 603.79,
1676
+ "tokens/trainable": 30839466
1677
+ },
1678
+ {
1679
+ "epoch": 3.8816326530612244,
1680
+ "grad_norm": 0.76171875,
1681
+ "learning_rate": 1.0378653108955017e-06,
1682
+ "loss": 1.197021484375,
1683
+ "memory/device_reserved (GiB)": 33.29,
1684
+ "memory/max_active (GiB)": 27.26,
1685
+ "memory/max_allocated (GiB)": 27.26,
1686
+ "ppl": 3.31024,
1687
+ "step": 120,
1688
+ "tokens/total": 31162368,
1689
+ "tokens/train_per_sec_per_gpu": 606.03,
1690
+ "tokens/trainable": 31101032
1691
+ },
1692
+ {
1693
+ "epoch": 3.914285714285714,
1694
+ "grad_norm": 0.671875,
1695
+ "learning_rate": 1.0242460570300241e-06,
1696
+ "loss": 1.1612548828125,
1697
+ "memory/device_reserved (GiB)": 33.29,
1698
+ "memory/max_active (GiB)": 27.26,
1699
+ "memory/max_allocated (GiB)": 27.26,
1700
+ "ppl": 3.19394,
1701
+ "step": 121,
1702
+ "tokens/total": 31424512,
1703
+ "tokens/train_per_sec_per_gpu": 605.01,
1704
+ "tokens/trainable": 31362628
1705
+ },
1706
+ {
1707
+ "epoch": 3.946938775510204,
1708
+ "grad_norm": 0.6796875,
1709
+ "learning_rate": 1.01364377190799e-06,
1710
+ "loss": 1.2498779296875,
1711
+ "memory/device_reserved (GiB)": 33.29,
1712
+ "memory/max_active (GiB)": 27.26,
1713
+ "memory/max_allocated (GiB)": 27.26,
1714
+ "ppl": 3.48992,
1715
+ "step": 122,
1716
+ "tokens/total": 31686656,
1717
+ "tokens/train_per_sec_per_gpu": 606.07,
1718
+ "tokens/trainable": 31624172
1719
+ },
1720
+ {
1721
+ "epoch": 3.979591836734694,
1722
+ "grad_norm": 0.69921875,
1723
+ "learning_rate": 1.0060656022052966e-06,
1724
+ "loss": 1.2774658203125,
1725
+ "memory/device_reserved (GiB)": 33.29,
1726
+ "memory/max_active (GiB)": 27.26,
1727
+ "memory/max_allocated (GiB)": 27.26,
1728
+ "ppl": 3.58754,
1729
+ "step": 123,
1730
+ "tokens/total": 31948800,
1731
+ "tokens/train_per_sec_per_gpu": 605.62,
1732
+ "tokens/trainable": 31885618
1733
+ },
1734
+ {
1735
+ "epoch": 4.0,
1736
+ "grad_norm": 0.86328125,
1737
+ "learning_rate": 1.0015166561341943e-06,
1738
+ "loss": 1.245849609375,
1739
+ "memory/device_reserved (GiB)": 33.29,
1740
+ "memory/max_active (GiB)": 27.26,
1741
+ "memory/max_allocated (GiB)": 27.26,
1742
+ "ppl": 3.47589,
1743
+ "step": 124,
1744
+ "tokens/total": 32112640,
1745
+ "tokens/train_per_sec_per_gpu": 920.1,
1746
+ "tokens/trainable": 32048304
1747
+ }
1748
+ ],
1749
+ "logging_steps": 1,
1750
+ "max_steps": 124,
1751
+ "num_input_tokens_seen": 0,
1752
+ "num_train_epochs": 4,
1753
+ "save_steps": 500,
1754
+ "stateful_callbacks": {
1755
+ "TrainerControl": {
1756
+ "args": {
1757
+ "should_epoch_stop": false,
1758
+ "should_evaluate": false,
1759
+ "should_log": false,
1760
+ "should_save": true,
1761
+ "should_training_stop": true
1762
+ },
1763
+ "attributes": {}
1764
+ }
1765
+ },
1766
+ "total_flos": 6.982781248145981e+17,
1767
+ "train_batch_size": 1,
1768
+ "trial_name": null,
1769
+ "trial_params": null
1770
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/training_plan.json ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "arm": "control",
3
+ "base_repo": "unsloth/gemma-3-4b-pt",
4
+ "base_revision": "52aba93981c6ad7712b030eb6dd496ece1d279d6",
5
+ "base_snapshot": "/root/.cache/huggingface/hub/models--unsloth--gemma-3-4b-pt/snapshots/52aba93981c6ad7712b030eb6dd496ece1d279d6",
6
+ "data_seed": 42,
7
+ "expected_optimizer_steps": 124,
8
+ "post_warmup_checkpoint": 4,
9
+ "stage": "midtrain_dispatch_gemma3_4b_4epoch",
10
+ "started_at": "2026-08-15T01:08:07+00:00",
11
+ "training_seed": 314159
12
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/training_provenance.json ADDED
@@ -0,0 +1,120 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": 1,
3
+ "status": "complete",
4
+ "stage": "midtrain_dispatch_gemma3_4b_4epoch",
5
+ "kind": "midtrain",
6
+ "created_at": "2026-08-15T01:08:07+00:00",
7
+ "resolved_config": {
8
+ "base_model": "/root/.cache/huggingface/hub/models--unsloth--gemma-3-4b-pt/snapshots/52aba93981c6ad7712b030eb6dd496ece1d279d6",
9
+ "trust_remote_code": false,
10
+ "plugins": [
11
+ "axolotl.integrations.liger.LigerPlugin",
12
+ "scimt.train.axolotl_plugins.CheckpointSchedulePlugin"
13
+ ],
14
+ "liger_fused_linear_cross_entropy": true,
15
+ "liger_rope": true,
16
+ "liger_rms_norm": true,
17
+ "liger_glu_activation": true,
18
+ "datasets": [
19
+ {
20
+ "path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/mix_control",
21
+ "type": "completion",
22
+ "field": "text"
23
+ }
24
+ ],
25
+ "dataset_prepared_path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/prepared",
26
+ "dataset_processes": 16,
27
+ "sequence_len": 8192,
28
+ "sample_packing": true,
29
+ "pad_to_sequence_len": true,
30
+ "bf16": true,
31
+ "tf32": true,
32
+ "flash_attention": true,
33
+ "gradient_checkpointing": true,
34
+ "micro_batch_size": 1,
35
+ "gradient_accumulation_steps": 16,
36
+ "num_epochs": 4,
37
+ "max_steps": 124,
38
+ "optimizer": "adamw_torch_fused",
39
+ "learning_rate": 1e-05,
40
+ "weight_decay": 0.01,
41
+ "max_grad_norm": 1.0,
42
+ "lr_scheduler": "cosine",
43
+ "cosine_min_lr_ratio": 0.1,
44
+ "warmup_ratio": 0.03,
45
+ "fsdp_version": 2,
46
+ "fsdp_config": {
47
+ "offload_params": false,
48
+ "cpu_ram_efficient_loading": true,
49
+ "auto_wrap_policy": "TRANSFORMER_BASED_WRAP",
50
+ "transformer_layer_cls_to_wrap": "Gemma3DecoderLayer",
51
+ "state_dict_type": "FULL_STATE_DICT",
52
+ "reshard_after_forward": true
53
+ },
54
+ "logging_steps": 1,
55
+ "save_strategy": "no",
56
+ "save_only_model": false,
57
+ "save_total_limit": 6,
58
+ "checkpoint_schedule": [
59
+ 4,
60
+ 31,
61
+ 62,
62
+ 93,
63
+ 124
64
+ ],
65
+ "seed": 314159,
66
+ "output_dir": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints"
67
+ },
68
+ "resolved_config_path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/axolotl.yaml",
69
+ "dataset": {
70
+ "path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/mix_control",
71
+ "exists": false
72
+ },
73
+ "schedule": {
74
+ "learning_rate": 1e-05,
75
+ "lr_scheduler": "cosine",
76
+ "warmup_ratio": 0.03,
77
+ "cosine_min_lr_ratio": 0.1
78
+ },
79
+ "step_plan": {
80
+ "raw_dataset_rows": null,
81
+ "micro_batch_size": 1,
82
+ "gradient_accumulation_steps": 16,
83
+ "world_size_at_render": 1,
84
+ "effective_global_batch_size": 16,
85
+ "num_epochs": 4,
86
+ "planned_optimizer_steps_before_length_filter": null,
87
+ "max_steps_override": 124,
88
+ "logging_steps": 1,
89
+ "save_strategy": "no",
90
+ "save_steps": null,
91
+ "save_total_limit": 6
92
+ },
93
+ "seed": 314159,
94
+ "completed_at": "2026-08-15T01:40:36+00:00",
95
+ "resolved_config_sha256": "ac9a1fc10269025655e99b2d55373e9e195ae916fda0794b9e595c2a24b34bf2",
96
+ "actual": {
97
+ "global_step": 124,
98
+ "max_steps": 124,
99
+ "num_train_epochs": 4,
100
+ "final_epoch": 4.0,
101
+ "train_batch_size": 1,
102
+ "num_input_tokens_seen": 0,
103
+ "total_flos": 6.982781248145981e+17,
104
+ "checkpoint_steps": [
105
+ 4,
106
+ 31,
107
+ 62,
108
+ 93,
109
+ 124
110
+ ],
111
+ "trainer_state_source": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-124/trainer_state.json",
112
+ "trainer_state_snapshot": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/trainer_state.final.json",
113
+ "trainer_state_sha256": "61cc4b722d7a64ae370c1e05f6dcab6129ac297b6970c90b45a494ab0d98c0a8",
114
+ "trace_rows": 124,
115
+ "trace_path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/training_trace.jsonl",
116
+ "trace_sha256": "9f9b2952cc376c3848909769622f69c538a83d9a328bebaea1652a05eadb45d1",
117
+ "first_learning_rate": 0.0,
118
+ "last_learning_rate": 1.0015166561341943e-06
119
+ }
120
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/training_started.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": "scimt_training_health_v1",
3
+ "status": "training_started",
4
+ "global_step": 1,
5
+ "finite_loss": 1.6685791015625,
6
+ "source_commit": "883445140956d34d47258cb021e56e0f4cbea05d",
7
+ "observed_at": "2026-08-15T01:10:17.122180+00:00"
8
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/arms/control/training_trace.jsonl ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"epoch": 0.0326530612244898, "grad_norm": 3.96875, "learning_rate": 0.0, "loss": 1.6685791015625, "memory/device_reserved (GiB)": 26.08, "memory/max_active (GiB)": 20.03, "memory/max_allocated (GiB)": 20.03, "ppl": 5.30463, "step": 1, "tokens/total": 262144, "tokens/train_per_sec_per_gpu": 471.78, "tokens/trainable": 261922}
2
+ {"epoch": 0.0653061224489796, "grad_norm": 4.21875, "learning_rate": 3.3333333333333333e-06, "loss": 1.5614013671875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.76549, "step": 2, "tokens/total": 524288, "tokens/train_per_sec_per_gpu": 597.28, "tokens/trainable": 523573}
3
+ {"epoch": 0.09795918367346938, "grad_norm": 3.40625, "learning_rate": 6.666666666666667e-06, "loss": 1.5543212890625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.73187, "step": 3, "tokens/total": 786432, "tokens/train_per_sec_per_gpu": 606.45, "tokens/trainable": 785311}
4
+ {"epoch": 0.1306122448979592, "grad_norm": 2.4375, "learning_rate": 1e-05, "loss": 1.575927734375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.83523, "step": 4, "tokens/total": 1048576, "tokens/train_per_sec_per_gpu": 609.43, "tokens/trainable": 1047069}
5
+ {"epoch": 0.16326530612244897, "grad_norm": 2.921875, "learning_rate": 9.998483343865806e-06, "loss": 1.529052734375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.6138, "step": 5, "tokens/total": 1310720, "tokens/train_per_sec_per_gpu": 604.52, "tokens/trainable": 1308722}
6
+ {"epoch": 0.19591836734693877, "grad_norm": 2.625, "learning_rate": 9.993934397794704e-06, "loss": 1.39306640625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.02718, "step": 6, "tokens/total": 1572864, "tokens/train_per_sec_per_gpu": 607.08, "tokens/trainable": 1570345}
7
+ {"epoch": 0.22857142857142856, "grad_norm": 1.734375, "learning_rate": 9.986356228092011e-06, "loss": 1.4849853515625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.4149, "step": 7, "tokens/total": 1835008, "tokens/train_per_sec_per_gpu": 605.44, "tokens/trainable": 1832202}
8
+ {"epoch": 0.2612244897959184, "grad_norm": 1.4296875, "learning_rate": 9.975753942969978e-06, "loss": 1.58740234375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.89103, "step": 8, "tokens/total": 2097152, "tokens/train_per_sec_per_gpu": 604.81, "tokens/trainable": 2093962}
9
+ {"epoch": 0.2938775510204082, "grad_norm": 1.46875, "learning_rate": 9.962134689104498e-06, "loss": 1.34619140625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.84276, "step": 9, "tokens/total": 2359296, "tokens/train_per_sec_per_gpu": 607.46, "tokens/trainable": 2355679}
10
+ {"epoch": 0.32653061224489793, "grad_norm": 1.25, "learning_rate": 9.945507646817764e-06, "loss": 1.47705078125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.38001, "step": 10, "tokens/total": 2621440, "tokens/train_per_sec_per_gpu": 606.8, "tokens/trainable": 2617308}
11
+ {"epoch": 0.35918367346938773, "grad_norm": 1.171875, "learning_rate": 9.925884023890072e-06, "loss": 1.4583740234375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.29896, "step": 11, "tokens/total": 2883584, "tokens/train_per_sec_per_gpu": 604.71, "tokens/trainable": 2878962}
12
+ {"epoch": 0.39183673469387753, "grad_norm": 1.203125, "learning_rate": 9.903277048005017e-06, "loss": 1.5108642578125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.53064, "step": 12, "tokens/total": 3145728, "tokens/train_per_sec_per_gpu": 603.72, "tokens/trainable": 3140634}
13
+ {"epoch": 0.42448979591836733, "grad_norm": 1.0234375, "learning_rate": 9.877701957833113e-06, "loss": 1.3990478515625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.05134, "step": 13, "tokens/total": 3407872, "tokens/train_per_sec_per_gpu": 604.61, "tokens/trainable": 3402298}
14
+ {"epoch": 0.45714285714285713, "grad_norm": 0.9375, "learning_rate": 9.849175992759867e-06, "loss": 1.32666015625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.76844, "step": 14, "tokens/total": 3670016, "tokens/train_per_sec_per_gpu": 605.87, "tokens/trainable": 3663962}
15
+ {"epoch": 0.4897959183673469, "grad_norm": 1.7734375, "learning_rate": 9.81771838126524e-06, "loss": 1.46875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.3438, "step": 15, "tokens/total": 3932160, "tokens/train_per_sec_per_gpu": 603.17, "tokens/trainable": 3925695}
16
+ {"epoch": 0.5224489795918368, "grad_norm": 1.7890625, "learning_rate": 9.783350327962313e-06, "loss": 1.409912109375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.0956, "step": 16, "tokens/total": 4194304, "tokens/train_per_sec_per_gpu": 603.95, "tokens/trainable": 4187301}
17
+ {"epoch": 0.5551020408163265, "grad_norm": 3.03125, "learning_rate": 9.74609499930392e-06, "loss": 1.401123046875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.05976, "step": 17, "tokens/total": 4456448, "tokens/train_per_sec_per_gpu": 603.72, "tokens/trainable": 4449140}
18
+ {"epoch": 0.5877551020408164, "grad_norm": 1.015625, "learning_rate": 9.70597750796683e-06, "loss": 1.476318359375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.3768, "step": 18, "tokens/total": 4718592, "tokens/train_per_sec_per_gpu": 604.17, "tokens/trainable": 4710654}
19
+ {"epoch": 0.6204081632653061, "grad_norm": 0.9453125, "learning_rate": 9.663024895924078e-06, "loss": 1.2203369140625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.38833, "step": 19, "tokens/total": 4980736, "tokens/train_per_sec_per_gpu": 605.11, "tokens/trainable": 4972132}
20
+ {"epoch": 0.6530612244897959, "grad_norm": 1.0078125, "learning_rate": 9.61726611621679e-06, "loss": 1.451171875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.26811, "step": 20, "tokens/total": 5242880, "tokens/train_per_sec_per_gpu": 603.4, "tokens/trainable": 5233845}
21
+ {"epoch": 0.6857142857142857, "grad_norm": 0.83203125, "learning_rate": 9.568732013437827e-06, "loss": 1.3369140625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.80728, "step": 21, "tokens/total": 5505024, "tokens/train_per_sec_per_gpu": 604.73, "tokens/trainable": 5495423}
22
+ {"epoch": 0.7183673469387755, "grad_norm": 0.953125, "learning_rate": 9.517455302940388e-06, "loss": 1.4034423828125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.06918, "step": 22, "tokens/total": 5767168, "tokens/train_per_sec_per_gpu": 605.55, "tokens/trainable": 5756905}
23
+ {"epoch": 0.7510204081632653, "grad_norm": 0.78125, "learning_rate": 9.46347054878559e-06, "loss": 1.3974609375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.04492, "step": 23, "tokens/total": 6029312, "tokens/train_per_sec_per_gpu": 601.68, "tokens/trainable": 6018394}
24
+ {"epoch": 0.7836734693877551, "grad_norm": 0.84375, "learning_rate": 9.406814140443898e-06, "loss": 1.446533203125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.24836, "step": 24, "tokens/total": 6291456, "tokens/train_per_sec_per_gpu": 602.65, "tokens/trainable": 6279921}
25
+ {"epoch": 0.8163265306122449, "grad_norm": 0.94140625, "learning_rate": 9.347524268266092e-06, "loss": 1.303955078125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.68384, "step": 25, "tokens/total": 6553600, "tokens/train_per_sec_per_gpu": 605.26, "tokens/trainable": 6541489}
26
+ {"epoch": 0.8489795918367347, "grad_norm": 0.9453125, "learning_rate": 9.285640897740316e-06, "loss": 1.4931640625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.45116, "step": 26, "tokens/total": 6815744, "tokens/train_per_sec_per_gpu": 603.79, "tokens/trainable": 6803233}
27
+ {"epoch": 0.8816326530612245, "grad_norm": 0.953125, "learning_rate": 9.22120574255258e-06, "loss": 1.2637939453125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.53882, "step": 27, "tokens/total": 7077888, "tokens/train_per_sec_per_gpu": 578.56, "tokens/trainable": 7064799}
28
+ {"epoch": 0.9142857142857143, "grad_norm": 0.75390625, "learning_rate": 9.154262236468826e-06, "loss": 1.222412109375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.39537, "step": 28, "tokens/total": 7340032, "tokens/train_per_sec_per_gpu": 605.17, "tokens/trainable": 7326396}
29
+ {"epoch": 0.9469387755102041, "grad_norm": 0.83984375, "learning_rate": 9.084855504057562e-06, "loss": 1.306884765625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.69465, "step": 29, "tokens/total": 7602176, "tokens/train_per_sec_per_gpu": 606.42, "tokens/trainable": 7587940}
30
+ {"epoch": 0.9795918367346939, "grad_norm": 0.83984375, "learning_rate": 9.013032330272777e-06, "loss": 1.3385009765625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.81332, "step": 30, "tokens/total": 7864320, "tokens/train_per_sec_per_gpu": 606.34, "tokens/trainable": 7849389}
31
+ {"epoch": 1.0, "grad_norm": 0.9375, "learning_rate": 8.938841128917622e-06, "loss": 1.3125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.71545, "step": 31, "tokens/total": 8028160, "tokens/train_per_sec_per_gpu": 912.97, "tokens/trainable": 8012073}
32
+ {"epoch": 1.0326530612244897, "grad_norm": 0.87109375, "learning_rate": 8.86233191001016e-06, "loss": 1.4200439453125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.1373, "step": 32, "tokens/total": 8290304, "tokens/train_per_sec_per_gpu": 584.93, "tokens/trainable": 8273995}
33
+ {"epoch": 1.0653061224489795, "grad_norm": 0.890625, "learning_rate": 8.783556246073135e-06, "loss": 1.3094482421875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.70413, "step": 33, "tokens/total": 8552448, "tokens/train_per_sec_per_gpu": 607.05, "tokens/trainable": 8535646}
34
+ {"epoch": 1.0979591836734695, "grad_norm": 0.9375, "learning_rate": 8.702567237370521e-06, "loss": 1.317138671875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.73273, "step": 34, "tokens/total": 8814592, "tokens/train_per_sec_per_gpu": 605.99, "tokens/trainable": 8797384}
35
+ {"epoch": 1.1306122448979592, "grad_norm": 0.80078125, "learning_rate": 8.619419476114251e-06, "loss": 1.3460693359375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.84229, "step": 35, "tokens/total": 9076736, "tokens/train_per_sec_per_gpu": 606.76, "tokens/trainable": 9059142}
36
+ {"epoch": 1.163265306122449, "grad_norm": 0.83984375, "learning_rate": 8.534169009665282e-06, "loss": 1.353759765625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.87196, "step": 36, "tokens/total": 9338880, "tokens/train_per_sec_per_gpu": 607.47, "tokens/trainable": 9320795}
37
+ {"epoch": 1.1959183673469387, "grad_norm": 0.80859375, "learning_rate": 8.446873302753783e-06, "loss": 1.2401123046875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.456, "step": 37, "tokens/total": 9601024, "tokens/train_per_sec_per_gpu": 605.46, "tokens/trainable": 9582418}
38
+ {"epoch": 1.2285714285714286, "grad_norm": 0.765625, "learning_rate": 8.357591198743923e-06, "loss": 1.343017578125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.83059, "step": 38, "tokens/total": 9863168, "tokens/train_per_sec_per_gpu": 603.72, "tokens/trainable": 9844275}
39
+ {"epoch": 1.2612244897959184, "grad_norm": 0.80078125, "learning_rate": 8.266382879969356e-06, "loss": 1.466552734375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.33427, "step": 39, "tokens/total": 10125312, "tokens/train_per_sec_per_gpu": 604.71, "tokens/trainable": 10106035}
40
+ {"epoch": 1.2938775510204081, "grad_norm": 0.7578125, "learning_rate": 8.17330982716615e-06, "loss": 1.2305908203125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.42325, "step": 40, "tokens/total": 10387456, "tokens/train_per_sec_per_gpu": 607.6, "tokens/trainable": 10367752}
41
+ {"epoch": 1.3265306122448979, "grad_norm": 0.875, "learning_rate": 8.078434778030511e-06, "loss": 1.3778076171875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.9662, "step": 41, "tokens/total": 10649600, "tokens/train_per_sec_per_gpu": 605.02, "tokens/trainable": 10629381}
42
+ {"epoch": 1.3591836734693876, "grad_norm": 0.765625, "learning_rate": 7.981821684929218e-06, "loss": 1.362060546875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.90423, "step": 42, "tokens/total": 10911744, "tokens/train_per_sec_per_gpu": 603.78, "tokens/trainable": 10891035}
43
+ {"epoch": 1.3918367346938776, "grad_norm": 0.8046875, "learning_rate": 7.883535671791294e-06, "loss": 1.420166015625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.13781, "step": 43, "tokens/total": 11173888, "tokens/train_per_sec_per_gpu": 604.13, "tokens/trainable": 11152707}
44
+ {"epoch": 1.4244897959183673, "grad_norm": 0.83984375, "learning_rate": 7.783642990209951e-06, "loss": 1.31005859375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.70639, "step": 44, "tokens/total": 11436032, "tokens/train_per_sec_per_gpu": 605.21, "tokens/trainable": 11414371}
45
+ {"epoch": 1.457142857142857, "grad_norm": 0.77734375, "learning_rate": 7.682210974784426e-06, "loss": 1.2432861328125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.46699, "step": 45, "tokens/total": 11698176, "tokens/train_per_sec_per_gpu": 605.6, "tokens/trainable": 11676035}
46
+ {"epoch": 1.489795918367347, "grad_norm": 0.85546875, "learning_rate": 7.579307997731783e-06, "loss": 1.4068603515625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.08312, "step": 46, "tokens/total": 11960320, "tokens/train_per_sec_per_gpu": 603.01, "tokens/trainable": 11937768}
47
+ {"epoch": 1.5224489795918368, "grad_norm": 0.83984375, "learning_rate": 7.475003422799302e-06, "loss": 1.3421630859375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.82731, "step": 47, "tokens/total": 12222464, "tokens/train_per_sec_per_gpu": 603.3, "tokens/trainable": 12199374}
48
+ {"epoch": 1.5551020408163265, "grad_norm": 0.8828125, "learning_rate": 7.36936755850849e-06, "loss": 1.3466796875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.84464, "step": 48, "tokens/total": 12484608, "tokens/train_per_sec_per_gpu": 605.0, "tokens/trainable": 12461213}
49
+ {"epoch": 1.5877551020408163, "grad_norm": 0.81640625, "learning_rate": 7.2624716107622675e-06, "loss": 1.404296875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.07266, "step": 49, "tokens/total": 12746752, "tokens/train_per_sec_per_gpu": 604.6, "tokens/trainable": 12722727}
50
+ {"epoch": 1.620408163265306, "grad_norm": 0.73046875, "learning_rate": 7.154387634847241e-06, "loss": 1.1488037109375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.15442, "step": 50, "tokens/total": 13008896, "tokens/train_per_sec_per_gpu": 605.31, "tokens/trainable": 12984205}
51
+ {"epoch": 1.6530612244897958, "grad_norm": 0.79296875, "learning_rate": 7.045188486863449e-06, "loss": 1.38818359375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.00756, "step": 51, "tokens/total": 13271040, "tokens/train_per_sec_per_gpu": 603.17, "tokens/trainable": 13245918}
52
+ {"epoch": 1.6857142857142857, "grad_norm": 0.75, "learning_rate": 6.9349477746142846e-06, "loss": 1.2738037109375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.57442, "step": 52, "tokens/total": 13533184, "tokens/train_per_sec_per_gpu": 606.57, "tokens/trainable": 13507496}
53
+ {"epoch": 1.7183673469387755, "grad_norm": 0.84375, "learning_rate": 6.823739807989734e-06, "loss": 1.3431396484375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.83105, "step": 53, "tokens/total": 13795328, "tokens/train_per_sec_per_gpu": 606.26, "tokens/trainable": 13768978}
54
+ {"epoch": 1.7510204081632654, "grad_norm": 0.7265625, "learning_rate": 6.7116395488763565e-06, "loss": 1.3397216796875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.81798, "step": 54, "tokens/total": 14057472, "tokens/train_per_sec_per_gpu": 602.75, "tokens/trainable": 14030467}
55
+ {"epoch": 1.7836734693877552, "grad_norm": 0.7265625, "learning_rate": 6.598722560627761e-06, "loss": 1.3912353515625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.01981, "step": 55, "tokens/total": 14319616, "tokens/train_per_sec_per_gpu": 571.54, "tokens/trainable": 14291994}
56
+ {"epoch": 1.816326530612245, "grad_norm": 0.8515625, "learning_rate": 6.485064957129677e-06, "loss": 1.2479248046875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.48311, "step": 56, "tokens/total": 14581760, "tokens/train_per_sec_per_gpu": 606.07, "tokens/trainable": 14553562}
57
+ {"epoch": 1.8489795918367347, "grad_norm": 0.82421875, "learning_rate": 6.370743351493899e-06, "loss": 1.4447021484375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.24059, "step": 57, "tokens/total": 14843904, "tokens/train_per_sec_per_gpu": 604.39, "tokens/trainable": 14815306}
58
+ {"epoch": 1.8816326530612244, "grad_norm": 0.75390625, "learning_rate": 6.255834804415742e-06, "loss": 1.2120361328125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.36032, "step": 58, "tokens/total": 15106048, "tokens/train_per_sec_per_gpu": 606.32, "tokens/trainable": 15076872}
59
+ {"epoch": 1.9142857142857141, "grad_norm": 0.83203125, "learning_rate": 6.140416772229785e-06, "loss": 1.175048828125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.2383, "step": 59, "tokens/total": 15368192, "tokens/train_per_sec_per_gpu": 604.24, "tokens/trainable": 15338469}
60
+ {"epoch": 1.9469387755102041, "grad_norm": 0.71875, "learning_rate": 6.0245670546989165e-06, "loss": 1.2623291015625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.53364, "step": 60, "tokens/total": 15630336, "tokens/train_per_sec_per_gpu": 606.62, "tokens/trainable": 15600013}
61
+ {"epoch": 1.9795918367346939, "grad_norm": 0.7578125, "learning_rate": 5.908363742571915e-06, "loss": 1.291015625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.63648, "step": 61, "tokens/total": 15892480, "tokens/train_per_sec_per_gpu": 606.98, "tokens/trainable": 15861462}
62
+ {"epoch": 2.0, "grad_norm": 1.0078125, "learning_rate": 5.791885164944844e-06, "loss": 1.26220703125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.53321, "step": 62, "tokens/total": 16056320, "tokens/train_per_sec_per_gpu": 920.2, "tokens/trainable": 16024146}
63
+ {"epoch": 2.0326530612244897, "grad_norm": 0.81640625, "learning_rate": 5.67520983646182e-06, "loss": 1.37841796875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.96862, "step": 63, "tokens/total": 16318464, "tokens/train_per_sec_per_gpu": 585.03, "tokens/trainable": 16286068}
64
+ {"epoch": 2.0653061224489795, "grad_norm": 1.109375, "learning_rate": 5.5584164043906895e-06, "loss": 1.271728515625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.56701, "step": 64, "tokens/total": 16580608, "tokens/train_per_sec_per_gpu": 607.21, "tokens/trainable": 16547719}
65
+ {"epoch": 2.0979591836734692, "grad_norm": 1.0390625, "learning_rate": 5.441583595609312e-06, "loss": 1.28076171875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.59938, "step": 65, "tokens/total": 16842752, "tokens/train_per_sec_per_gpu": 605.49, "tokens/trainable": 16809456}
66
+ {"epoch": 2.130612244897959, "grad_norm": 0.86328125, "learning_rate": 5.324790163538181e-06, "loss": 1.3126220703125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.7159, "step": 66, "tokens/total": 17104896, "tokens/train_per_sec_per_gpu": 606.91, "tokens/trainable": 17071218}
67
+ {"epoch": 2.163265306122449, "grad_norm": 0.84765625, "learning_rate": 5.208114835055157e-06, "loss": 1.32037353515625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.74482, "step": 67, "tokens/total": 17367040, "tokens/train_per_sec_per_gpu": 607.11, "tokens/trainable": 17332876}
68
+ {"epoch": 2.195918367346939, "grad_norm": 0.72265625, "learning_rate": 5.0916362574280864e-06, "loss": 1.2105712890625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.3554, "step": 68, "tokens/total": 17629184, "tokens/train_per_sec_per_gpu": 605.15, "tokens/trainable": 17594500}
69
+ {"epoch": 2.2285714285714286, "grad_norm": 0.8515625, "learning_rate": 4.975432945301085e-06, "loss": 1.3128662109375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.71681, "step": 69, "tokens/total": 17891328, "tokens/train_per_sec_per_gpu": 603.11, "tokens/trainable": 17856356}
70
+ {"epoch": 2.2612244897959184, "grad_norm": 0.78515625, "learning_rate": 4.859583227770218e-06, "loss": 1.4384765625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.21427, "step": 70, "tokens/total": 18153472, "tokens/train_per_sec_per_gpu": 604.33, "tokens/trainable": 18118114}
71
+ {"epoch": 2.293877551020408, "grad_norm": 0.80859375, "learning_rate": 4.744165195584258e-06, "loss": 1.203125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.33051, "step": 71, "tokens/total": 18415616, "tokens/train_per_sec_per_gpu": 606.6, "tokens/trainable": 18379828}
72
+ {"epoch": 2.326530612244898, "grad_norm": 0.82421875, "learning_rate": 4.6292566485061015e-06, "loss": 1.352294921875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.86629, "step": 72, "tokens/total": 18677760, "tokens/train_per_sec_per_gpu": 606.39, "tokens/trainable": 18641452}
73
+ {"epoch": 2.3591836734693876, "grad_norm": 0.78125, "learning_rate": 4.514935042870324e-06, "loss": 1.3382568359375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.81239, "step": 73, "tokens/total": 18939904, "tokens/train_per_sec_per_gpu": 602.92, "tokens/trainable": 18903104}
74
+ {"epoch": 2.3918367346938774, "grad_norm": 0.78125, "learning_rate": 4.40127743937224e-06, "loss": 1.396728515625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.04196, "step": 74, "tokens/total": 19202048, "tokens/train_per_sec_per_gpu": 603.33, "tokens/trainable": 19164778}
75
+ {"epoch": 2.424489795918367, "grad_norm": 0.71875, "learning_rate": 4.288360451123646e-06, "loss": 1.287841796875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.62495, "step": 75, "tokens/total": 19464192, "tokens/train_per_sec_per_gpu": 603.18, "tokens/trainable": 19426440}
76
+ {"epoch": 2.4571428571428573, "grad_norm": 0.81640625, "learning_rate": 4.1762601920102675e-06, "loss": 1.22216796875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.39454, "step": 76, "tokens/total": 19726336, "tokens/train_per_sec_per_gpu": 606.31, "tokens/trainable": 19688104}
77
+ {"epoch": 2.489795918367347, "grad_norm": 0.76171875, "learning_rate": 4.065052225385717e-06, "loss": 1.3858642578125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.99828, "step": 77, "tokens/total": 19988480, "tokens/train_per_sec_per_gpu": 603.6, "tokens/trainable": 19949836}
78
+ {"epoch": 2.522448979591837, "grad_norm": 0.74609375, "learning_rate": 3.954811513136554e-06, "loss": 1.32080078125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.74642, "step": 78, "tokens/total": 20250624, "tokens/train_per_sec_per_gpu": 603.42, "tokens/trainable": 20211444}
79
+ {"epoch": 2.5551020408163265, "grad_norm": 1.015625, "learning_rate": 3.84561236515276e-06, "loss": 1.3240966796875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.75879, "step": 79, "tokens/total": 20512768, "tokens/train_per_sec_per_gpu": 603.23, "tokens/trainable": 20473284}
80
+ {"epoch": 2.5877551020408163, "grad_norm": 0.703125, "learning_rate": 3.7375283892377344e-06, "loss": 1.385986328125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.99877, "step": 80, "tokens/total": 20774912, "tokens/train_per_sec_per_gpu": 604.17, "tokens/trainable": 20734804}
81
+ {"epoch": 2.620408163265306, "grad_norm": 0.7109375, "learning_rate": 3.630632441491512e-06, "loss": 1.130859375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.09832, "step": 81, "tokens/total": 21037056, "tokens/train_per_sec_per_gpu": 606.12, "tokens/trainable": 20996280}
82
+ {"epoch": 2.6530612244897958, "grad_norm": 0.85546875, "learning_rate": 3.5249965772007e-06, "loss": 1.3719482421875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.94303, "step": 82, "tokens/total": 21299200, "tokens/train_per_sec_per_gpu": 603.84, "tokens/trainable": 21257992}
83
+ {"epoch": 2.685714285714286, "grad_norm": 0.703125, "learning_rate": 3.4206920022682173e-06, "loss": 1.2578125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.51772, "step": 83, "tokens/total": 21561344, "tokens/train_per_sec_per_gpu": 605.93, "tokens/trainable": 21519576}
84
+ {"epoch": 2.7183673469387752, "grad_norm": 0.85546875, "learning_rate": 3.3177890252155755e-06, "loss": 1.3260498046875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.76614, "step": 84, "tokens/total": 21823488, "tokens/train_per_sec_per_gpu": 604.99, "tokens/trainable": 21781056}
85
+ {"epoch": 2.7510204081632654, "grad_norm": 0.8203125, "learning_rate": 3.2163570097900497e-06, "loss": 1.324951171875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.762, "step": 85, "tokens/total": 22085632, "tokens/train_per_sec_per_gpu": 585.4, "tokens/trainable": 22042546}
86
+ {"epoch": 2.783673469387755, "grad_norm": 0.73828125, "learning_rate": 3.116464328208708e-06, "loss": 1.3780517578125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.96717, "step": 86, "tokens/total": 22347776, "tokens/train_per_sec_per_gpu": 586.69, "tokens/trainable": 22304072}
87
+ {"epoch": 2.816326530612245, "grad_norm": 0.80859375, "learning_rate": 3.0181783150707827e-06, "loss": 1.23388671875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.43455, "step": 87, "tokens/total": 22609920, "tokens/train_per_sec_per_gpu": 605.15, "tokens/trainable": 22565640}
88
+ {"epoch": 2.8489795918367347, "grad_norm": 0.79296875, "learning_rate": 2.921565221969492e-06, "loss": 1.430908203125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.1825, "step": 88, "tokens/total": 22872064, "tokens/train_per_sec_per_gpu": 603.72, "tokens/trainable": 22827386}
89
+ {"epoch": 2.8816326530612244, "grad_norm": 0.74609375, "learning_rate": 2.8266901728338526e-06, "loss": 1.19873046875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.3159, "step": 89, "tokens/total": 23134208, "tokens/train_per_sec_per_gpu": 606.04, "tokens/trainable": 23088952}
90
+ {"epoch": 2.914285714285714, "grad_norm": 0.6953125, "learning_rate": 2.7336171200306467e-06, "loss": 1.1632080078125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.20018, "step": 90, "tokens/total": 23396352, "tokens/train_per_sec_per_gpu": 604.92, "tokens/trainable": 23350548}
91
+ {"epoch": 2.946938775510204, "grad_norm": 0.875, "learning_rate": 2.6424088012560766e-06, "loss": 1.2506103515625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.49247, "step": 91, "tokens/total": 23658496, "tokens/train_per_sec_per_gpu": 605.58, "tokens/trainable": 23612092}
92
+ {"epoch": 2.979591836734694, "grad_norm": 0.859375, "learning_rate": 2.5531266972462176e-06, "loss": 1.2801513671875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.59718, "step": 92, "tokens/total": 23920640, "tokens/train_per_sec_per_gpu": 606.4, "tokens/trainable": 23873538}
93
+ {"epoch": 3.0, "grad_norm": 0.94140625, "learning_rate": 2.4658309903347196e-06, "loss": 1.248046875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.48353, "step": 93, "tokens/total": 24084480, "tokens/train_per_sec_per_gpu": 922.28, "tokens/trainable": 24036224}
94
+ {"epoch": 3.0326530612244897, "grad_norm": 0.7109375, "learning_rate": 2.380580523885751e-06, "loss": 1.3673095703125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.92478, "step": 94, "tokens/total": 24346624, "tokens/train_per_sec_per_gpu": 586.15, "tokens/trainable": 24298144}
95
+ {"epoch": 3.0653061224489795, "grad_norm": 0.67578125, "learning_rate": 2.29743276262948e-06, "loss": 1.2625732421875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.5345, "step": 95, "tokens/total": 24608768, "tokens/train_per_sec_per_gpu": 607.42, "tokens/trainable": 24559796}
96
+ {"epoch": 3.0979591836734692, "grad_norm": 0.84375, "learning_rate": 2.2164437539268652e-06, "loss": 1.2720947265625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.56832, "step": 96, "tokens/total": 24870912, "tokens/train_per_sec_per_gpu": 605.03, "tokens/trainable": 24821538}
97
+ {"epoch": 3.130612244897959, "grad_norm": 0.69921875, "learning_rate": 2.1376680899898415e-06, "loss": 1.3033447265625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.68159, "step": 97, "tokens/total": 25133056, "tokens/train_per_sec_per_gpu": 607.5, "tokens/trainable": 25083298}
98
+ {"epoch": 3.163265306122449, "grad_norm": 0.73828125, "learning_rate": 2.0611588710823797e-06, "loss": 1.31024169921875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.70707, "step": 98, "tokens/total": 25395200, "tokens/train_per_sec_per_gpu": 607.32, "tokens/trainable": 25344956}
99
+ {"epoch": 3.195918367346939, "grad_norm": 0.796875, "learning_rate": 1.986967669727224e-06, "loss": 1.20263671875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.32888, "step": 99, "tokens/total": 25657344, "tokens/train_per_sec_per_gpu": 604.43, "tokens/trainable": 25606580}
100
+ {"epoch": 3.2285714285714286, "grad_norm": 0.69921875, "learning_rate": 1.9151444959424383e-06, "loss": 1.3046875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.68654, "step": 100, "tokens/total": 25919488, "tokens/train_per_sec_per_gpu": 603.08, "tokens/trainable": 25868436}
101
+ {"epoch": 3.2612244897959184, "grad_norm": 0.78515625, "learning_rate": 1.8457377635311763e-06, "loss": 1.431396484375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.18454, "step": 101, "tokens/total": 26181632, "tokens/train_per_sec_per_gpu": 603.9, "tokens/trainable": 26130194}
102
+ {"epoch": 3.293877551020408, "grad_norm": 1.1171875, "learning_rate": 1.7787942574474215e-06, "loss": 1.19580078125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.3062, "step": 102, "tokens/total": 26443776, "tokens/train_per_sec_per_gpu": 606.74, "tokens/trainable": 26391908}
103
+ {"epoch": 3.326530612244898, "grad_norm": 0.70703125, "learning_rate": 1.7143591022596846e-06, "loss": 1.34716796875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.84652, "step": 103, "tokens/total": 26705920, "tokens/train_per_sec_per_gpu": 606.45, "tokens/trainable": 26653532}
104
+ {"epoch": 3.3591836734693876, "grad_norm": 0.71484375, "learning_rate": 1.6524757317339102e-06, "loss": 1.3314208984375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.78642, "step": 104, "tokens/total": 26968064, "tokens/train_per_sec_per_gpu": 602.49, "tokens/trainable": 26915184}
105
+ {"epoch": 3.3918367346938774, "grad_norm": 2.484375, "learning_rate": 1.593185859556103e-06, "loss": 1.3916015625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.02129, "step": 105, "tokens/total": 27230208, "tokens/train_per_sec_per_gpu": 603.38, "tokens/trainable": 27176858}
106
+ {"epoch": 3.424489795918367, "grad_norm": 0.71875, "learning_rate": 1.5365294512144114e-06, "loss": 1.282958984375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.6073, "step": 106, "tokens/total": 27492352, "tokens/train_per_sec_per_gpu": 604.12, "tokens/trainable": 27438520}
107
+ {"epoch": 3.4571428571428573, "grad_norm": 0.6640625, "learning_rate": 1.4825446970596136e-06, "loss": 1.2174072265625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.37842, "step": 107, "tokens/total": 27754496, "tokens/train_per_sec_per_gpu": 606.17, "tokens/trainable": 27700184}
108
+ {"epoch": 3.489795918367347, "grad_norm": 0.73046875, "learning_rate": 1.4312679865621742e-06, "loss": 1.380615234375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.97735, "step": 108, "tokens/total": 28016640, "tokens/train_per_sec_per_gpu": 602.64, "tokens/trainable": 27961916}
109
+ {"epoch": 3.522448979591837, "grad_norm": 0.6953125, "learning_rate": 1.382733883783211e-06, "loss": 1.3165283203125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.73045, "step": 109, "tokens/total": 28278784, "tokens/train_per_sec_per_gpu": 603.31, "tokens/trainable": 28223524}
110
+ {"epoch": 3.5551020408163265, "grad_norm": 0.71484375, "learning_rate": 1.3369751040759236e-06, "loss": 1.3199462890625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.74322, "step": 110, "tokens/total": 28540928, "tokens/train_per_sec_per_gpu": 602.96, "tokens/trainable": 28485364}
111
+ {"epoch": 3.5877551020408163, "grad_norm": 0.6953125, "learning_rate": 1.2940224920331707e-06, "loss": 1.3828125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.9861, "step": 111, "tokens/total": 28803072, "tokens/train_per_sec_per_gpu": 603.97, "tokens/trainable": 28746884}
112
+ {"epoch": 3.620408163265306, "grad_norm": 0.6640625, "learning_rate": 1.2539050006960814e-06, "loss": 1.1270751953125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.08662, "step": 112, "tokens/total": 29065216, "tokens/train_per_sec_per_gpu": 605.96, "tokens/trainable": 29008360}
113
+ {"epoch": 3.6530612244897958, "grad_norm": 0.7421875, "learning_rate": 1.2166496720376874e-06, "loss": 1.368408203125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.92909, "step": 113, "tokens/total": 29327360, "tokens/train_per_sec_per_gpu": 602.33, "tokens/trainable": 29270072}
114
+ {"epoch": 3.685714285714286, "grad_norm": 0.67578125, "learning_rate": 1.1822816187347625e-06, "loss": 1.2537841796875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.50358, "step": 114, "tokens/total": 29589504, "tokens/train_per_sec_per_gpu": 606.12, "tokens/trainable": 29531656}
115
+ {"epoch": 3.7183673469387752, "grad_norm": 0.703125, "learning_rate": 1.1508240072401336e-06, "loss": 1.3232421875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.75558, "step": 115, "tokens/total": 29851648, "tokens/train_per_sec_per_gpu": 605.25, "tokens/trainable": 29793136}
116
+ {"epoch": 3.7510204081632654, "grad_norm": 0.70703125, "learning_rate": 1.1222980421668874e-06, "loss": 1.3226318359375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.75329, "step": 116, "tokens/total": 30113792, "tokens/train_per_sec_per_gpu": 570.96, "tokens/trainable": 30054626}
117
+ {"epoch": 3.783673469387755, "grad_norm": 1.328125, "learning_rate": 1.0967229519949833e-06, "loss": 1.3746337890625, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.95363, "step": 117, "tokens/total": 30375936, "tokens/train_per_sec_per_gpu": 602.33, "tokens/trainable": 30316152}
118
+ {"epoch": 3.816326530612245, "grad_norm": 0.71875, "learning_rate": 1.0741159761099294e-06, "loss": 1.231201171875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.42534, "step": 118, "tokens/total": 30638080, "tokens/train_per_sec_per_gpu": 605.12, "tokens/trainable": 30577720}
119
+ {"epoch": 3.8489795918367347, "grad_norm": 0.80078125, "learning_rate": 1.054492353182237e-06, "loss": 1.4288330078125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 4.17383, "step": 119, "tokens/total": 30900224, "tokens/train_per_sec_per_gpu": 603.79, "tokens/trainable": 30839466}
120
+ {"epoch": 3.8816326530612244, "grad_norm": 0.76171875, "learning_rate": 1.0378653108955017e-06, "loss": 1.197021484375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.31024, "step": 120, "tokens/total": 31162368, "tokens/train_per_sec_per_gpu": 606.03, "tokens/trainable": 31101032}
121
+ {"epoch": 3.914285714285714, "grad_norm": 0.671875, "learning_rate": 1.0242460570300241e-06, "loss": 1.1612548828125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.19394, "step": 121, "tokens/total": 31424512, "tokens/train_per_sec_per_gpu": 605.01, "tokens/trainable": 31362628}
122
+ {"epoch": 3.946938775510204, "grad_norm": 0.6796875, "learning_rate": 1.01364377190799e-06, "loss": 1.2498779296875, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.48992, "step": 122, "tokens/total": 31686656, "tokens/train_per_sec_per_gpu": 606.07, "tokens/trainable": 31624172}
123
+ {"epoch": 3.979591836734694, "grad_norm": 0.69921875, "learning_rate": 1.0060656022052966e-06, "loss": 1.2774658203125, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.58754, "step": 123, "tokens/total": 31948800, "tokens/train_per_sec_per_gpu": 605.62, "tokens/trainable": 31885618}
124
+ {"epoch": 4.0, "grad_norm": 0.86328125, "learning_rate": 1.0015166561341943e-06, "loss": 1.245849609375, "memory/device_reserved (GiB)": 33.29, "memory/max_active (GiB)": 27.26, "memory/max_allocated (GiB)": 27.26, "ppl": 3.47589, "step": 124, "tokens/total": 32112640, "tokens/train_per_sec_per_gpu": 920.1, "tokens/trainable": 32048304}
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/data/control_mix.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:de2c2c62e12ab0714ca3d7149d18865d8287b603893c52d082844cc8ac5a57e0
3
+ size 31390287
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/data/control_mix_manifest.json ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "arm": "control",
3
+ "docs": 11387,
4
+ "filler_manifest": {
5
+ "all_shards_order_sha256": "fbd27dcd107799286f3b24a208c617b50dc812c4fb7c95050b246486647ed2f3",
6
+ "budget": 8000000,
7
+ "docs": 11387,
8
+ "opened_shards": [
9
+ "data/ingredient1-olmocr_science_pdfs-high_quality-crime_law-2e13/crime_law_p030_shard_00004457.jsonl.zst",
10
+ "data/ingredient1-wiki_to_rcqa-part1/00009_f325.jsonl.zst",
11
+ "data/ingredient2-wiki_to_rcqa_part2/00046_f187.jsonl.zst",
12
+ "data/ingredient1-common_crawl-high-quality_19_science_math_and_technology/shard_00000558.jsonl.zst",
13
+ "data/ingredient2-dolmino-math/dolmino_math_tinyGSM-MIND_2students_tiny_gsm_inline_part117.000000.jsonl.jsonl.zst",
14
+ "data/ingredient2-wiki_to_rcqa_part1/00007_f60.jsonl.zst",
15
+ "data/ingredient2-olmocr_science_pdfs-high_quality-science_tech-length_2e12/science_tech_p050_shard_00001495.jsonl.zst",
16
+ "data/ingredient2-common_crawl-high-quality_20_sports_and_fitness/shard_00000448.jsonl.zst",
17
+ "data/ingredient1-dolmino-math/dolmino_math_mathcoder2-synthmath_m-a-p_Matrix_filtered-math_book_math.0003.0245.jsonl.jsonl.zst",
18
+ "data/ingredient2-common_crawl-high-quality_19_religion/shard_00000206.jsonl.zst",
19
+ "data/ingredient2-common_crawl-high-quality_19_health/shard_00000244.jsonl.zst",
20
+ "data/ingredient2-common_crawl-high-quality_20_social_life/shard_00000541.jsonl.zst",
21
+ "data/ingredient1-wiki_to_rcqa-part1/00020_f124.jsonl.zst",
22
+ "data/ingredient2-common_crawl-high-quality_20_fashion_and_beauty/shard_00000119.jsonl.zst",
23
+ "data/ingredient2-wiki_to_rcqa_part2/00060_f63.jsonl.zst",
24
+ "data/ingredient1-wiki_to_rcqa-part1/00014_f172.jsonl.zst",
25
+ "data/ingredient2-common_crawl-high-quality_19_entertainment/shard_00001008.jsonl.zst",
26
+ "data/ingredient2-common_crawl-high-quality_20_education_and_jobs/shard_00000154.jsonl.zst",
27
+ "data/ingredient2-general_reasoning_mix/train-00184-of-00278.jsonl.zst",
28
+ "data/ingredient1-wiki_to_rcqa-part1/00000_f313.jsonl.zst"
29
+ ],
30
+ "ordered_rows_sha256": "a852f50e44ec8814f74b15e0f9e0aebebb01a7141e11e2c9b027fe292164bb12",
31
+ "repo": "allenai/dolma3_dolmino_mix-100B-1125",
32
+ "revision": "f23aa129fda8335ba9760057bcc1f0c02f3d068b",
33
+ "seed": 42,
34
+ "shuffle_buffer": 10000,
35
+ "tokens": 8002382
36
+ },
37
+ "gate2_reference_digests": {
38
+ "jsonl_sha256": "de2c2c62e12ab0714ca3d7149d18865d8287b603893c52d082844cc8ac5a57e0",
39
+ "observed_ordered_rows_gate2_scheme": "5fc226629c0c253c179550aa362a56e89a4fa943d2747e1012264db87d50f1e6",
40
+ "ordered_rows_sha256": "a852f50e44ec8814f74b15e0f9e0aebebb01a7141e11e2c9b027fe292164bb12"
41
+ },
42
+ "jsonl_sha256": "de2c2c62e12ab0714ca3d7149d18865d8287b603893c52d082844cc8ac5a57e0",
43
+ "per_source": {
44
+ "filler": {
45
+ "docs": 11387,
46
+ "tokens": 8002382
47
+ }
48
+ },
49
+ "prefix_replay": {
50
+ "docs": 6085,
51
+ "ordered_rows_sha256": "819f35334706f6cd942ef3af31c927f3fcd986e3a3107372b461046d30ff02a9",
52
+ "tokens": 4001953
53
+ },
54
+ "seed": 42,
55
+ "source_order_sha256": "abdc46436ddad18f6aff9fada41c2c01df7f7501dfa9eecbe03ce49356e1324b",
56
+ "tokenizer": {
57
+ "repo": "unsloth/gemma-3-4b-pt",
58
+ "revision": "52aba93981c6ad7712b030eb6dd496ece1d279d6"
59
+ },
60
+ "total_tokens": 8002382
61
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/data/control_source_order.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/environment.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "captured_at": "2026-08-15T01:06:35+00:00",
3
+ "environment": {
4
+ "CUDA_VISIBLE_DEVICES": null,
5
+ "HOSTNAME": null,
6
+ "NCCL_DEBUG": "WARN",
7
+ "NCCL_NVLS_ENABLE": "0",
8
+ "PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:True",
9
+ "RUNPOD_GPU_COUNT": null,
10
+ "RUNPOD_POD_ID": null,
11
+ "SCIMT_FLASH_INSTALL": "prebuilt",
12
+ "SCIMT_GPU_ARCH": "9.0",
13
+ "SCIMT_POD_IMAGE": "runpod/pytorch:1.0.2-cu1281-torch280-ubuntu2404",
14
+ "SCIMT_SOURCE_BRANCH": "sid/prior-coins-27b"
15
+ },
16
+ "hostname": "82b399c0391b",
17
+ "packages": {
18
+ "accelerate": "1.13.0",
19
+ "axolotl": "0.17.0",
20
+ "datasets": "4.8.5",
21
+ "flash-attn": "2.8.3",
22
+ "huggingface-hub": "1.18.0",
23
+ "liger-kernel": "0.7.0",
24
+ "torch": "2.12.1+cu126",
25
+ "transformers": "5.9.0",
26
+ "zstandard": "0.22.0"
27
+ },
28
+ "python": "3.12.3 (main, Aug 14 2025, 17:47:21) [GCC 13.3.0]",
29
+ "source_commit": "883445140956d34d47258cb021e56e0f4cbea05d"
30
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/git_head.json ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "command": [
3
+ "git",
4
+ "show",
5
+ "-s",
6
+ "--format=fuller",
7
+ "HEAD"
8
+ ],
9
+ "returncode": 128,
10
+ "stderr": "fatal: not a git repository (or any of the parent directories): .git\n",
11
+ "stdout": ""
12
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/git_status.json ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "command": [
3
+ "git",
4
+ "status",
5
+ "--porcelain=v1",
6
+ "--branch"
7
+ ],
8
+ "returncode": 128,
9
+ "stderr": "fatal: not a git repository (or any of the parent directories): .git\n",
10
+ "stdout": ""
11
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/nvidia_smi_full.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "command": [
3
+ "nvidia-smi",
4
+ "-q"
5
+ ],
6
+ "returncode": 0,
7
+ "stderr": "",
8
+ "stdout": "\n==============NVSMI LOG==============\n\nTimestamp : Sat Aug 15 01:06:34 2026\nDriver Version : 575.57.08\nCUDA Version : 12.9\n\nAttached GPUs : 2\nGPU 00000000:CB:00.0\n Product Name : NVIDIA H200\n Product Brand : NVIDIA\n Product Architecture : Hopper\n Display Mode : Requested functionality has been deprecated\n Display Attached : Yes\n Display Active : Disabled\n Persistence Mode : Enabled\n Addressing Mode : None\n MIG Mode\n Current : Disabled\n Pending : Disabled\n Accounting Mode : Disabled\n Accounting Mode Buffer Size : 4000\n Driver Model\n Current : N/A\n Pending : N/A\n Serial Number : 1653924053179\n GPU UUID : GPU-d752f587-7164-6296-0a8f-f37805c21241\n Minor Number : 6\n VBIOS Version : 96.00.A5.00.03\n MultiGPU Board : No\n Board ID : 0xcb00\n Board Part Number : 692-2G520-0280-001\n GPU Part Number : 2335-895-A1\n FRU Part Number : N/A\n Platform Info\n Chassis Serial Number : N/A\n Slot Number : N/A\n Tray Index : N/A\n Host ID : N/A\n Peer Type : N/A\n Module Id : 6\n GPU Fabric GUID : N/A\n Inforom Version\n Image Version : G520.0280.02.02\n OEM Object : 2.1\n ECC Object : 7.16\n Power Management Object : N/A\n Inforom BBX Object Flush\n Latest Timestamp : 2026/08/14 01:55:09.188\n Latest Duration : 138428 us\n GPU Operation Mode\n Current : N/A\n Pending : N/A\n GPU C2C Mode : Disabled\n GPU Virtualization Mode\n Virtualization Mode : None\n Host VGPU Mode : N/A\n vGPU Heterogeneous Mode : N/A\n GPU Reset Status\n Reset Required : Requested functionality has been deprecated\n Drain and Reset Recommended : Requested functionality has been deprecated\n GPU Recovery Action : None\n GSP Firmware Version : 575.57.08\n IBMNPU\n Relaxed Ordering Mode : N/A\n PCI\n Bus : 0xCB\n Device : 0x00\n Domain : 0x0000\n Base Classcode : 0x3\n Sub Classcode : 0x2\n Device Id : 0x233510DE\n Bus Id : 00000000:CB:00.0\n Sub System Id : 0x18BE10DE\n GPU Link Info\n PCIe Generation\n Max : 5\n Current : 5\n Device Current : 5\n Device Max : 5\n Host Max : 5\n Link Width\n Max : 16x\n Current : 16x\n Bridge Chip\n Type : N/A\n Firmware : N/A\n Replays Since Reset : 7\n Replay Number Rollovers : 0\n Tx Throughput : 690 KB/s\n Rx Throughput : 565 KB/s\n Atomic Caps Outbound : N/A\n Atomic Caps Inbound : N/A\n Fan Speed : N/A\n Performance State : P0\n Clocks Event Reasons\n Idle : Active\n Applications Clocks Setting : Not Active\n SW Power Cap : Not Active\n HW Slowdown : Not Active\n HW Thermal Slowdown : Not Active\n HW Power Brake Slowdown : Not Active\n Sync Boost : Not Active\n SW Thermal Slowdown : Not Active\n Display Clock Setting : Not Active\n Clocks Event Reasons Counters\n SW Power Capping : 1319092989593 us\n Sync Boost : 0 us\n SW Thermal Slowdown : 3072140629 us\n HW Thermal Slowdown : 325980843 us\n HW Power Braking : 0 us\n Sparse Operation Mode : Disabled\n FB Memory Usage\n Total : 143771 MiB\n Reserved : 615 MiB\n Used : 0 MiB\n Free : 143157 MiB\n BAR1 Memory Usage\n Total : 262144 MiB\n Used : 1 MiB\n Free : 262143 MiB\n Conf Compute Protected Memory Usage\n Total : 0 MiB\n Used : 0 MiB\n Free : 0 MiB\n Compute Mode : Default\n Utilization\n GPU : 0 %\n Memory : 0 %\n Encoder : 0 %\n Decoder : 0 %\n JPEG : 0 %\n OFA : 0 %\n Encoder Stats\n Active Sessions : 0\n Average FPS : 0\n Average Latency : 0\n FBC Stats\n Active Sessions : 0\n Average FPS : 0\n Average Latency : 0\n DRAM Encryption Mode\n Current : N/A\n Pending : N/A\n ECC Mode\n Current : Enabled\n Pending : Enabled\n ECC Errors\n Volatile\n SRAM Correctable : 0\n SRAM Uncorrectable Parity : 0\n SRAM Uncorrectable SEC-DED : 0\n DRAM Correctable : 0\n DRAM Uncorrectable : 0\n Aggregate\n SRAM Correctable : 0\n SRAM Uncorrectable Parity : 0\n SRAM Uncorrectable SEC-DED : 0\n DRAM Correctable : 0\n DRAM Uncorrectable : 0\n SRAM Threshold Exceeded : No\n Aggregate Uncorrectable SRAM Sources\n SRAM L2 : 0\n SRAM SM : 0\n SRAM Microcontroller : 0\n SRAM PCIE : 0\n SRAM Other : 0\n Retired Pages\n Single Bit ECC : N/A\n Double Bit ECC : N/A\n Pending Page Blacklist : N/A\n Remapped Rows\n Correctable Error : 0\n Uncorrectable Error : 0\n Pending : No\n Remapping Failure Occurred : No\n Bank Remap Availability Histogram\n Max : 3072 bank(s)\n High : 0 bank(s)\n Partial : 0 bank(s)\n Low : 0 bank(s)\n None : 0 bank(s)\n Temperature\n GPU Current Temp : 22 C\n GPU T.Limit Temp : 65 C\n GPU Shutdown T.Limit Temp : -8 C\n GPU Slowdown T.Limit Temp : -2 C\n GPU Max Operating T.Limit Temp : 0 C\n GPU Target Temperature : N/A\n Memory Current Temp : 27 C\n Memory Max Operating T.Limit Temp : 0 C\n GPU Power Readings\n Average Power Draw : 74.55 W\n Instantaneous Power Draw : 74.77 W\n Current Power Limit : 700.00 W\n Requested Power Limit : 700.00 W\n Default Power Limit : 700.00 W\n Min Power Limit : 200.00 W\n Max Power Limit : 700.00 W\n GPU Memory Power Readings \n Average Power Draw : 37.46 W\n Instantaneous Power Draw : N/A\n Module Power Readings\n Average Power Draw : N/A\n Instantaneous Power Draw : N/A\n Current Power Limit : N/A\n Requested Power Limit : N/A\n Default Power Limit : N/A\n Min Power Limit : N/A\n Max Power Limit : N/A\n Power Smoothing : N/A\n Workload Power Profiles\n Requested Profiles : N/A\n Enforced Profiles : N/A\n Clocks\n Graphics : 345 MHz\n SM : 345 MHz\n Memory : 3201 MHz\n Video : 765 MHz\n Applications Clocks\n Graphics : 1980 MHz\n Memory : 3201 MHz\n Default Applications Clocks\n Graphics : 1980 MHz\n Memory : 3201 MHz\n Deferred Clocks\n Memory : N/A\n Max Clocks\n Graphics : 1980 MHz\n SM : 1980 MHz\n Memory : 3201 MHz\n Video : 1545 MHz\n Max Customer Boost Clocks\n Graphics : 1980 MHz\n Clock Policy\n Auto Boost : N/A\n Auto Boost Default : N/A\n Voltage\n Graphics : Requested functionality has been deprecated\n Fabric\n State : Completed\n Status : Success\n CliqueId : 0\n ClusterUUID : 00000000-0000-0000-0000-000000000000\n Health\n Bandwidth : N/A\n Route Recovery in progress : N/A\n Route Unhealthy : N/A\n Access Timeout Recovery : N/A\n Processes : None\n Capabilities\n EGM : disabled\n\nGPU 00000000:DB:00.0\n Product Name : NVIDIA H200\n Product Brand : NVIDIA\n Product Architecture : Hopper\n Display Mode : Requested functionality has been deprecated\n Display Attached : Yes\n Display Active : Disabled\n Persistence Mode : Enabled\n Addressing Mode : None\n MIG Mode\n Current : Disabled\n Pending : Disabled\n Accounting Mode : Disabled\n Accounting Mode Buffer Size : 4000\n Driver Model\n Current : N/A\n Pending : N/A\n Serial Number : 1653924059472\n GPU UUID : GPU-8c36b36c-9728-1afb-0b8b-737c51828df9\n Minor Number : 7\n VBIOS Version : 96.00.A5.00.03\n MultiGPU Board : No\n Board ID : 0xdb00\n Board Part Number : 692-2G520-0280-001\n GPU Part Number : 2335-895-A1\n FRU Part Number : N/A\n Platform Info\n Chassis Serial Number : N/A\n Slot Number : N/A\n Tray Index : N/A\n Host ID : N/A\n Peer Type : N/A\n Module Id : 8\n GPU Fabric GUID : N/A\n Inforom Version\n Image Version : G520.0280.02.02\n OEM Object : 2.1\n ECC Object : 7.16\n Power Management Object : N/A\n Inforom BBX Object Flush\n Latest Timestamp : 2026/08/14 02:53:13.899\n Latest Duration : 133332 us\n GPU Operation Mode\n Current : N/A\n Pending : N/A\n GPU C2C Mode : Disabled\n GPU Virtualization Mode\n Virtualization Mode : None\n Host VGPU Mode : N/A\n vGPU Heterogeneous Mode : N/A\n GPU Reset Status\n Reset Required : Requested functionality has been deprecated\n Drain and Reset Recommended : Requested functionality has been deprecated\n GPU Recovery Action : None\n GSP Firmware Version : 575.57.08\n IBMNPU\n Relaxed Ordering Mode : N/A\n PCI\n Bus : 0xDB\n Device : 0x00\n Domain : 0x0000\n Base Classcode : 0x3\n Sub Classcode : 0x2\n Device Id : 0x233510DE\n Bus Id : 00000000:DB:00.0\n Sub System Id : 0x18BE10DE\n GPU Link Info\n PCIe Generation\n Max : 5\n Current : 5\n Device Current : 5\n Device Max : 5\n Host Max : 5\n Link Width\n Max : 16x\n Current : 16x\n Bridge Chip\n Type : N/A\n Firmware : N/A\n Replays Since Reset : 8\n Replay Number Rollovers : 0\n Tx Throughput : 699 KB/s\n Rx Throughput : 619 KB/s\n Atomic Caps Outbound : N/A\n Atomic Caps Inbound : N/A\n Fan Speed : N/A\n Performance State : P0\n Clocks Event Reasons\n Idle : Active\n Applications Clocks Setting : Not Active\n SW Power Cap : Not Active\n HW Slowdown : Not Active\n HW Thermal Slowdown : Not Active\n HW Power Brake Slowdown : Not Active\n Sync Boost : Not Active\n SW Thermal Slowdown : Not Active\n Display Clock Setting : Not Active\n Clocks Event Reasons Counters\n SW Power Capping : 821867057079 us\n Sync Boost : 0 us\n SW Thermal Slowdown : 20553071 us\n HW Thermal Slowdown : 390891 us\n HW Power Braking : 0 us\n Sparse Operation Mode : Disabled\n FB Memory Usage\n Total : 143771 MiB\n Reserved : 615 MiB\n Used : 0 MiB\n Free : 143157 MiB\n BAR1 Memory Usage\n Total : 262144 MiB\n Used : 1 MiB\n Free : 262143 MiB\n Conf Compute Protected Memory Usage\n Total : 0 MiB\n Used : 0 MiB\n Free : 0 MiB\n Compute Mode : Default\n Utilization\n GPU : 0 %\n Memory : 0 %\n Encoder : 0 %\n Decoder : 0 %\n JPEG : 0 %\n OFA : 0 %\n Encoder Stats\n Active Sessions : 0\n Average FPS : 0\n Average Latency : 0\n FBC Stats\n Active Sessions : 0\n Average FPS : 0\n Average Latency : 0\n DRAM Encryption Mode\n Current : N/A\n Pending : N/A\n ECC Mode\n Current : Enabled\n Pending : Enabled\n ECC Errors\n Volatile\n SRAM Correctable : 0\n SRAM Uncorrectable Parity : 0\n SRAM Uncorrectable SEC-DED : 0\n DRAM Correctable : 0\n DRAM Uncorrectable : 0\n Aggregate\n SRAM Correctable : 0\n SRAM Uncorrectable Parity : 0\n SRAM Uncorrectable SEC-DED : 0\n DRAM Correctable : 0\n DRAM Uncorrectable : 0\n SRAM Threshold Exceeded : No\n Aggregate Uncorrectable SRAM Sources\n SRAM L2 : 0\n SRAM SM : 0\n SRAM Microcontroller : 0\n SRAM PCIE : 0\n SRAM Other : 0\n Retired Pages\n Single Bit ECC : N/A\n Double Bit ECC : N/A\n Pending Page Blacklist : N/A\n Remapped Rows\n Correctable Error : 0\n Uncorrectable Error : 0\n Pending : No\n Remapping Failure Occurred : No\n Bank Remap Availability Histogram\n Max : 3072 bank(s)\n High : 0 bank(s)\n Partial : 0 bank(s)\n Low : 0 bank(s)\n None : 0 bank(s)\n Temperature\n GPU Current Temp : 21 C\n GPU T.Limit Temp : 66 C\n GPU Shutdown T.Limit Temp : -8 C\n GPU Slowdown T.Limit Temp : -2 C\n GPU Max Operating T.Limit Temp : 0 C\n GPU Target Temperature : N/A\n Memory Current Temp : 25 C\n Memory Max Operating T.Limit Temp : 0 C\n GPU Power Readings\n Average Power Draw : 77.35 W\n Instantaneous Power Draw : 76.94 W\n Current Power Limit : 700.00 W\n Requested Power Limit : 700.00 W\n Default Power Limit : 700.00 W\n Min Power Limit : 200.00 W\n Max Power Limit : 700.00 W\n GPU Memory Power Readings \n Average Power Draw : 34.06 W\n Instantaneous Power Draw : N/A\n Module Power Readings\n Average Power Draw : N/A\n Instantaneous Power Draw : N/A\n Current Power Limit : N/A\n Requested Power Limit : N/A\n Default Power Limit : N/A\n Min Power Limit : N/A\n Max Power Limit : N/A\n Power Smoothing : N/A\n Workload Power Profiles\n Requested Profiles : N/A\n Enforced Profiles : N/A\n Clocks\n Graphics : 345 MHz\n SM : 345 MHz\n Memory : 3201 MHz\n Video : 765 MHz\n Applications Clocks\n Graphics : 1980 MHz\n Memory : 3201 MHz\n Default Applications Clocks\n Graphics : 1980 MHz\n Memory : 3201 MHz\n Deferred Clocks\n Memory : N/A\n Max Clocks\n Graphics : 1980 MHz\n SM : 1980 MHz\n Memory : 3201 MHz\n Video : 1545 MHz\n Max Customer Boost Clocks\n Graphics : 1980 MHz\n Clock Policy\n Auto Boost : N/A\n Auto Boost Default : N/A\n Voltage\n Graphics : Requested functionality has been deprecated\n Fabric\n State : Completed\n Status : Success\n CliqueId : 0\n ClusterUUID : 00000000-0000-0000-0000-000000000000\n Health\n Bandwidth : N/A\n Route Recovery in progress : N/A\n Route Unhealthy : N/A\n Access Timeout Recovery : N/A\n Processes : None\n Capabilities\n EGM : disabled\n\n"
9
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/nvidia_smi_query.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "command": [
3
+ "nvidia-smi",
4
+ "--query-gpu=index,name,uuid,driver_version,memory.total",
5
+ "--format=csv,noheader"
6
+ ],
7
+ "returncode": 0,
8
+ "stderr": "",
9
+ "stdout": "0, NVIDIA H200, GPU-d752f587-7164-6296-0a8f-f37805c21241, 575.57.08, 143771 MiB\n1, NVIDIA H200, GPU-8c36b36c-9728-1afb-0b8b-737c51828df9, 575.57.08, 143771 MiB\n"
10
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/pip_freeze.json ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "command": [
3
+ "/usr/bin/python3",
4
+ "-m",
5
+ "pip",
6
+ "freeze"
7
+ ],
8
+ "returncode": 0,
9
+ "stderr": "",
10
+ "stdout": "absl-py==2.5.0\naccelerate==1.13.0\naddict==2.4.0\nadlfs==2026.8.0\naiobotocore==3.9.0\naiohappyeyeballs==2.7.1\naiohttp==3.14.3\naioitertools==0.13.0\naiosignal==1.4.0\nannotated-doc==0.0.5\nannotated-types==0.8.0\nantlr4-python3-runtime==4.9.3\nanyio==4.11.0\nargon2-cffi==25.1.0\nargon2-cffi-bindings==25.1.0\narrow==1.3.0\nart==6.5\nasttokens==3.0.0\nasync-lru==2.0.5\nattrs==25.4.0\naxolotl==0.17.0\naxolotl-contribs-lgpl==0.0.7\naxolotl-contribs-mit==0.0.6\nazure-core==1.41.0\nazure-identity==1.25.3\nazure-storage-blob==12.30.0\nbabel==2.17.0\nbackoff==2.2.1\nbeautifulsoup4==4.14.2\nbitsandbytes==0.49.1\nbleach==6.2.0\nblinker==1.7.0\nbotocore==1.43.56\nbrotli==1.2.0\ncbor2==6.1.4\ncertifi==2025.10.5\ncffi==2.0.0\nchardet==6.0.0.post1\ncharset-normalizer==3.4.3\ncircuitbreaker==2.1.3\nclick==8.4.2\ncolorama==0.4.6\ncoloredlogs==15.0.1\ncomm==0.2.3\ncrc32c==2.8\ncryptography==50.0.0\ncuda-bindings==12.9.7\ncuda-pathfinder==1.6.0\ncuda-toolkit==12.6.3\nDataProperty==1.1.1\ndatasets==4.8.5\ndbus-python==1.3.2\ndebugpy==1.8.17\ndecorator==5.2.1\ndefusedxml==0.7.1\ndill==0.4.1\ndistlib==0.4.0\ndistro==1.9.0\neinops==0.8.2\nevaluate==0.4.1\nexecuting==2.2.1\nfastapi==0.141.1\nfastcore==2.2.12\nfastjsonschema==2.21.2\nfilelock==3.20.0\nfire==0.7.1\nfla-core==0.4.1\nflash-linear-attention==0.4.1\nflash_attn @ file:///root/.cache/huggingface/hub/models--arcadia-impact--scimt-pod-wheels/snapshots/27ce508df4dbe28771313f419fb39e54aef726b3/cu126/flash_attn-2.8.3-cp312-cp312-linux_x86_64.whl\nfqdn==1.5.1\nfrozenlist==1.8.0\nfsspec==2026.2.0\ngcsfs==2026.2.0\ngoogle-api-core==2.30.3\ngoogle-auth==2.56.3\ngoogle-auth-oauthlib==1.4.0\ngoogle-cloud-core==2.6.1\ngoogle-cloud-storage==3.13.1\ngoogle-cloud-storage-control==1.13.0\ngoogle-crc32c==1.8.0\ngoogle-resumable-media==2.10.1\ngoogleapis-common-protos==1.75.1\ngradio==6.24.0\ngradio_client==2.6.0\ngroovy==0.1.2\ngrpc-google-iam-v1==0.14.5\ngrpcio==1.83.0\ngrpcio-status==1.83.0\ngrpclib==0.4.9\nh11==0.16.0\nh2==4.4.1\nhf-gradio==0.4.1\nhf-xet==1.4.3\nhf_transfer==0.1.9\nhpack==4.2.0\nhttpcore==1.0.9\nhttplib2==0.20.4\nhttptools==0.8.0\nhttpx==0.28.1\nhuggingface_hub==1.18.0\nhumanfriendly==10.0\nhyperframe==6.1.0\nidna==3.10\nimmutabledict==4.2.0\nipykernel==6.30.1\nipython==9.6.0\nipython_pygments_lexers==1.1.1\nipywidgets==8.1.7\nisodate==0.7.2\nisoduration==20.11.0\njedi==0.19.2\nJinja2==3.1.6\njmespath==1.1.0\njoblib==1.5.3\njson5==0.12.1\njsonlines==4.0.0\njsonpointer==3.0.0\njsonschema==4.25.1\njsonschema-specifications==2025.9.1\njupyter-archive==3.4.0\njupyter-events==0.12.0\njupyter-lsp==2.3.0\njupyter_client==8.6.3\njupyter_core==5.8.1\njupyter_server==2.17.0\njupyter_server_terminals==0.5.3\njupyterlab==4.4.9\njupyterlab_pygments==0.3.0\njupyterlab_server==2.27.3\njupyterlab_widgets==3.0.15\nkernels==0.13.0\nlangdetect==1.0.9\nlark==1.3.0\nlaunchpadlib==1.11.0\nlazr.restfulclient==0.14.6\nlazr.uri==1.0.6\nliger_kernel==0.7.0\nllvmlite==0.49.0\nlm_eval==0.4.11\nlxml==6.1.1\nMarkdown==3.10.3\nmarkdown-it-py==4.2.0\nMarkupSafe==3.0.3\nmatplotlib-inline==0.1.7\nmbstrdecoder==1.1.5\nmdurl==0.1.2\nmistral_common==1.11.0\nmistune==3.1.4\nmodal==1.3.0.post1\nmore-itertools==11.1.0\nmpmath==1.3.0\nmsal==1.37.0\nmsal-extensions==1.3.1\nmultidict==6.7.1\nmultiprocess==0.70.19\nnarwhals==2.24.0\nnbclient==0.10.2\nnbconvert==7.16.6\nnbformat==5.10.4\nnest-asyncio==1.6.0\nnetworkx==3.3\nnltk==3.10.3\nnotebook==7.4.2\nnotebook_shim==0.2.4\nnumba==0.67.0\nnumpy==2.3.5\nnvidia-cublas-cu12==12.6.4.1\nnvidia-cuda-cupti-cu12==12.6.80\nnvidia-cuda-nvrtc-cu12==12.6.85\nnvidia-cuda-runtime-cu12==12.6.77\nnvidia-cudnn-cu12==9.10.2.21\nnvidia-cufft-cu12==11.3.0.4\nnvidia-cufile-cu12==1.11.1.6\nnvidia-curand-cu12==10.3.7.77\nnvidia-cusolver-cu12==11.7.1.2\nnvidia-cusparse-cu12==12.5.4.2\nnvidia-cusparselt-cu12==0.7.1\nnvidia-ml-py==12.560.30\nnvidia-nccl-cu12==2.29.3\nnvidia-nvjitlink-cu12==12.6.85\nnvidia-nvshmem-cu12==3.4.5\nnvidia-nvtx-cu12==12.6.77\noauthlib==3.3.1\noci==2.184.1\nocifs==1.3.2\nomegaconf==2.3.1\nopenenv-core==0.1.0\nopentelemetry-api==1.44.0\noptimum==1.16.2\norjson==3.12.0\npackaging==26.0\npandas==3.0.5\npandocfilters==1.5.1\nparso==0.8.5\npathvalidate==3.3.1\npeft==0.19.1\npexpect==4.9.0\npillow==11.0.0\nplatformdirs==4.5.0\nportalocker==4.1.0\nposthog==6.7.11\nprometheus_client==0.23.1\nprompt_toolkit==3.0.52\npropcache==0.5.2\nproto-plus==1.28.3\nprotobuf==6.33.6\npsutil==7.1.0\nptyprocess==0.7.0\npure_eval==0.2.3\npyarrow==25.0.1\npyasn1==0.6.4\npyasn1_modules==0.4.2\npycountry==26.2.16\npycparser==2.23\npydantic==2.13.4\npydantic-extra-types==2.11.1\npydantic_core==2.46.4\npydub==0.25.1\nPygments==2.19.2\nPyGObject==3.48.2\nPyJWT==2.13.0\npyOpenSSL==26.4.0\npyparsing==3.1.1\npytablewriter==1.2.1\npython-apt==2.7.7+ubuntu5\npython-dateutil==2.9.0.post0\npython-dotenv==1.0.1\npython-json-logger==4.0.0\npython-multipart==0.0.32\npytz==2026.3.post1\nPyYAML==6.0.3\npyzmq==27.1.0\nreferencing==0.36.2\nregex==2026.7.19\nrequests==2.32.5\nrequests-oauthlib==2.0.0\nresponses==0.18.0\nrfc3339-validator==0.1.4\nrfc3986-validator==0.1.1\nrfc3987-syntax==1.1.0\nrich==15.0.0\nrouge_score==0.1.2\nrpds-py==0.27.1\ns3fs==2026.2.0\nsacrebleu==2.6.0\nsafehttpx==0.1.7\nsafetensors==0.8.0\nschedulefree==1.4.1\nscikit-learn==1.9.0\n# Editable install with no version control (scimt==0.0.1)\n-e /workspace/dispatch-scaleup-4b-mt-control-20260815t010433z-ctl2\nscipy==1.18.0\nsemantic-version==2.10.0\nSend2Trash==1.8.3\nsentencepiece==0.2.2\nsentry-sdk==2.68.0\nsetuptools==80.9.0\nshellingham==1.5.4\nsix==1.17.0\nsniffio==1.3.1\nsoupsieve==2.8\nsqlitedict==2.1.0\nstack-data==0.6.3\nstarlette==1.6.0\nsympy==1.13.3\nsynchronicity==0.11.1\ntabledata==1.3.5\ntabulate==0.10.0\ntcolorpy==0.1.7\ntensorboard==2.21.0\ntensorboard-data-server==0.7.2\ntermcolor==3.3.0\nterminado==0.18.1\nthreadpoolctl==3.6.0\ntiktoken==0.13.0\ntinycss2==1.4.0\ntokenizers==0.22.2\ntoml==0.10.2\ntomlkit==0.14.0\ntorch==2.12.1+cu126\ntorchao==0.17.0+cu126\ntorchaudio==2.8.0+cu128\ntorchvision==0.27.1+cu126\ntornado==6.5.2\ntqdm==4.70.0\ntrackio==0.35.0\ntraitlets==5.14.3\ntransformers==5.9.0\ntriton==3.7.1\ntrl==1.5.1\ntypepy==1.3.5\ntyper==0.25.1\ntypes-certifi==2021.10.8.3\ntypes-python-dateutil==2.9.0.20251008\ntypes-toml==0.10.8.20260518\ntyping-inspection==0.4.4\ntyping_extensions==4.15.0\nuri-template==1.3.0\nurllib3==2.7.0\nuvicorn==0.52.3\nuvloop==0.22.1\nvirtualenv==20.34.0\nwadllib==1.3.6\nwandb==0.28.2\nwatchfiles==1.2.0\nwcwidth==0.2.14\nwebcolors==24.11.1\nwebencodings==0.5.1\nwebsocket-client==1.9.0\nwebsockets==17.0.1\nWerkzeug==3.1.8\nwidgetsnbextension==4.0.14\nword2number==1.1\nwrapt==2.3.0\nxformers==0.0.35\nxxhash==4.0.0\nyarl==1.24.5\nzstandard==0.22.0\n"
11
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/environment/uname.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "command": [
3
+ "uname",
4
+ "-a"
5
+ ],
6
+ "returncode": 0,
7
+ "stderr": "",
8
+ "stdout": "Linux 82b399c0391b 6.8.0-94-generic #96~22.04.1-Ubuntu SMP PREEMPT_DYNAMIC Fri Jan 16 13:19:05 UTC 2 x86_64 x86_64 x86_64 GNU/Linux\n"
9
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/events.jsonl ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"event": "repository_visibility_verified", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:06:33+00:00", "visibility": "public"}
2
+ {"event": "repository_visibility_verified", "repo_id": "arcadia-impact/scimt-dispatch-4b-scaleup-v1", "timestamp": "2026-08-15T01:06:33+00:00", "visibility": "private"}
3
+ {"docs": 11387, "event": "control_corpus_materialized", "timestamp": "2026-08-15T01:08:07+00:00", "tokens": 8002382}
4
+ {"arm": "control", "event": "training_started", "expected_steps": 124, "rendered_config": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/axolotl.yaml", "timestamp": "2026-08-15T01:08:07+00:00"}
5
+ {"event": "upload_started", "files": 16, "local": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-4", "remote": "midtrain_4epoch/control/checkpoint-4", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:41:02+00:00"}
6
+ {"commit_oid": "c0a5c82e20dd90872dd2e1ba0a06f92f98f08f03", "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/c0a5c82e20dd90872dd2e1ba0a06f92f98f08f03", "event": "upload_verified", "files": 16, "remote_prefix": "midtrain_4epoch/control/checkpoint-4", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:43:20+00:00", "tree_sha256": "51f1edcfa4ca103a92fdbbd53e4d1bdc0f5264eba3ba1aa60621cab20c603f31", "verified_at": "2026-08-15T01:43:20+00:00"}
7
+ {"event": "upload_started", "files": 16, "local": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-31", "remote": "midtrain_4epoch/control/checkpoint-31", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:43:45+00:00"}
8
+ {"commit_oid": "f225df41f35617335440fd45c89adf20e61ebb66", "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/f225df41f35617335440fd45c89adf20e61ebb66", "event": "upload_verified", "files": 16, "remote_prefix": "midtrain_4epoch/control/checkpoint-31", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:46:04+00:00", "tree_sha256": "a34acde4e2a622a6ffbfd1912d8058fcf471be768537d690a769966a5f3f7fc7", "verified_at": "2026-08-15T01:46:04+00:00"}
9
+ {"event": "upload_started", "files": 16, "local": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-62", "remote": "midtrain_4epoch/control/checkpoint-62", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:46:30+00:00"}
10
+ {"commit_oid": "3082a28661539aeac56366a2b203758242038186", "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/3082a28661539aeac56366a2b203758242038186", "event": "upload_verified", "files": 16, "remote_prefix": "midtrain_4epoch/control/checkpoint-62", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:48:46+00:00", "tree_sha256": "cf7d6d4ebe35d33b9cd1d58a5701b993c146887dbc45b8fae8095dc555416c5f", "verified_at": "2026-08-15T01:48:46+00:00"}
11
+ {"event": "upload_started", "files": 16, "local": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-93", "remote": "midtrain_4epoch/control/checkpoint-93", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:49:12+00:00"}
12
+ {"commit_oid": "fdc198123a2bc0720cbb48bfedd0ae393775fee4", "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/fdc198123a2bc0720cbb48bfedd0ae393775fee4", "event": "upload_verified", "files": 16, "remote_prefix": "midtrain_4epoch/control/checkpoint-93", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:51:27+00:00", "tree_sha256": "010cb434ceba9690be1a5f05cce520e922a532129314029c4d032c0eeb8762ba", "verified_at": "2026-08-15T01:51:27+00:00"}
13
+ {"event": "upload_started", "files": 16, "local": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control/checkpoints/checkpoint-124", "remote": "midtrain_4epoch/control/checkpoint-124", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:51:52+00:00"}
14
+ {"commit_oid": "9b33af4fd0103fe07e05732cbac4779bee2be97f", "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/9b33af4fd0103fe07e05732cbac4779bee2be97f", "event": "upload_verified", "files": 16, "remote_prefix": "midtrain_4epoch/control/checkpoint-124", "repo_id": "sidbaines/scimt-dispatch-4b-models-v1", "timestamp": "2026-08-15T01:54:09+00:00", "tree_sha256": "273b74b00d6b1f866268318bbcc6badc3e5059aa1060a624840c4c1851af67db", "verified_at": "2026-08-15T01:54:09+00:00"}
15
+ {"arm": "control", "elapsed_seconds": 2761.146, "event": "training_complete", "final_step": 124, "timestamp": "2026-08-15T01:54:09+00:00"}
16
+ {"event": "transient_reclaimed", "path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/train_control", "timestamp": "2026-08-15T01:54:09+00:00"}
17
+ {"event": "transient_reclaimed", "path": "/workspace/dispatch-scaleup-4b-control/20260815T010433Z-ctl2/mix_control", "timestamp": "2026-08-15T01:54:09+00:00"}
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/payload_files.json ADDED
@@ -0,0 +1,114 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "files": {
3
+ "arms/control/axolotl.rendered.yaml": {
4
+ "sha256": "ac9a1fc10269025655e99b2d55373e9e195ae916fda0794b9e595c2a24b34bf2",
5
+ "size": 1470
6
+ },
7
+ "arms/control/final_checkpoint_files.json": {
8
+ "sha256": "6ad1953c8c4f748dc29b6b5ed02707325c5276ca698208a8ebd3dfa86120b845",
9
+ "size": 2442
10
+ },
11
+ "arms/control/mix_manifest.json": {
12
+ "sha256": "c66d007c50be03cdebab2b7ba3a61599c89dda42ca938882f44be00131351955",
13
+ "size": 3234
14
+ },
15
+ "arms/control/post_warmup_checkpoint_files.json": {
16
+ "sha256": "21c6f685cfd506cee6c72cb42eb2bfb6004402d346b29a9ca83505d0211e2a52",
17
+ "size": 2439
18
+ },
19
+ "arms/control/result.json": {
20
+ "sha256": "25d3ec5ac5519fb0e9d95ac846ef7543700637947d685eb8c3553912d00f2427",
21
+ "size": 60273
22
+ },
23
+ "arms/control/step31_checkpoint_files.json": {
24
+ "sha256": "185cd2f3b8b5aefd31ebd42be58924caf89a5211663c8ba5fa322324521a2b38",
25
+ "size": 2441
26
+ },
27
+ "arms/control/step62_checkpoint_files.json": {
28
+ "sha256": "4a922e495f91ce2c4f5ef647b8bc566ddb55d806a0ff40d556f2e3c175c575df",
29
+ "size": 2441
30
+ },
31
+ "arms/control/step93_checkpoint_files.json": {
32
+ "sha256": "f1e0d3333e7df7bf661c89a735cb5b8ec4ef5ebe29ea6b491ecec175af58e8d6",
33
+ "size": 2441
34
+ },
35
+ "arms/control/train.log": {
36
+ "sha256": "1efc966ab67d39460a4bf939c57b5661087e16899649f79cfa14f1e405d56b37",
37
+ "size": 98364
38
+ },
39
+ "arms/control/trainer_state.final.json": {
40
+ "sha256": "61cc4b722d7a64ae370c1e05f6dcab6129ac297b6970c90b45a494ab0d98c0a8",
41
+ "size": 54464
42
+ },
43
+ "arms/control/training_plan.json": {
44
+ "sha256": "fa134b0e899fadbda1efefe12a58f2cf9e4ed6022438976a519b571761e8c74a",
45
+ "size": 465
46
+ },
47
+ "arms/control/training_provenance.json": {
48
+ "sha256": "d60aecc28528c8ad72e3506ba7667c401b56cb70c50831afafc04b06698ab1fe",
49
+ "size": 4023
50
+ },
51
+ "arms/control/training_started.json": {
52
+ "sha256": "bfeaf52806ee02aef26b64c87aeddcba46111fd025c76a208cf786a330749480",
53
+ "size": 253
54
+ },
55
+ "arms/control/training_trace.jsonl": {
56
+ "sha256": "9f9b2952cc376c3848909769622f69c538a83d9a328bebaea1652a05eadb45d1",
57
+ "size": 43401
58
+ },
59
+ "data/control_mix.jsonl": {
60
+ "sha256": "de2c2c62e12ab0714ca3d7149d18865d8287b603893c52d082844cc8ac5a57e0",
61
+ "size": 31390287
62
+ },
63
+ "data/control_mix_manifest.json": {
64
+ "sha256": "c66d007c50be03cdebab2b7ba3a61599c89dda42ca938882f44be00131351955",
65
+ "size": 3234
66
+ },
67
+ "data/control_source_order.jsonl": {
68
+ "sha256": "abdc46436ddad18f6aff9fada41c2c01df7f7501dfa9eecbe03ce49356e1324b",
69
+ "size": 1526772
70
+ },
71
+ "environment/environment.json": {
72
+ "sha256": "2c6feca3a01e0fb4a16ddf29fb6148caefae1aec5255f9b318be453c3a0e2977",
73
+ "size": 921
74
+ },
75
+ "environment/git_head.json": {
76
+ "sha256": "dda68592772ebf6efd4329ebbe08c8e57c88f067485ae663e473da4b2739f26f",
77
+ "size": 213
78
+ },
79
+ "environment/git_status.json": {
80
+ "sha256": "28f3ab84506c24b1825b014b25dc0ee44a8a9aebb6856a863d754c11622d960b",
81
+ "size": 208
82
+ },
83
+ "environment/nvidia_smi_full.json": {
84
+ "sha256": "f25eb984c19ad11da046a26ef24a16ec8cba6f5605dfbb88a4f971013f1258c7",
85
+ "size": 23020
86
+ },
87
+ "environment/nvidia_smi_query.json": {
88
+ "sha256": "38cb821e2e0a883d3b90cd046c0e8381627d87bfaac0c2e669234b419261aad1",
89
+ "size": 345
90
+ },
91
+ "environment/pip_freeze.json": {
92
+ "sha256": "12129618013ad5757750768f54bdd8a6ccb5c7cde4b28c265b39ac43fb24f6af",
93
+ "size": 6825
94
+ },
95
+ "environment/uname.json": {
96
+ "sha256": "237710b24b1c5f2f0284def4a9753eb35cb6d6a0a81158804bf046184dfa481c",
97
+ "size": 229
98
+ },
99
+ "events.jsonl": {
100
+ "sha256": "1c83954a9b676db2eac0c7664c84982ef97cf38639c0ebecce88c62462241118",
101
+ "size": 5150
102
+ },
103
+ "run.log": {
104
+ "sha256": "2851c857a958fde3a8b9123f24e724a6387d522f3d664117b2d2804e492b4aa5",
105
+ "size": 2890716
106
+ },
107
+ "run_manifest.json": {
108
+ "sha256": "b47ffb923e81b342855db8f089bb4c53d9c9870870e42ed76a977dbe576ac6dc",
109
+ "size": 69712
110
+ }
111
+ },
112
+ "schema_version": 1,
113
+ "tree_sha256": "db28a0220c0bc95330e356be83f15985e8867d6839056dba645ed3196d2f3fc8"
114
+ }
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/run.log ADDED
The diff for this file is too large to render. See raw diff
 
runs/20260815T010433Z-ctl2/midtrain/control/artifacts/run_manifest.json ADDED
@@ -0,0 +1,1873 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "arms": {
3
+ "control": {
4
+ "arm": "control",
5
+ "checkpoints": {
6
+ "final": {
7
+ "commit_oid": "9b33af4fd0103fe07e05732cbac4779bee2be97f",
8
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/9b33af4fd0103fe07e05732cbac4779bee2be97f",
9
+ "files": 16,
10
+ "local_name": "checkpoint-124",
11
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-124",
12
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
13
+ "tree_sha256": "273b74b00d6b1f866268318bbcc6badc3e5059aa1060a624840c4c1851af67db",
14
+ "verified_at": "2026-08-15T01:54:09+00:00"
15
+ },
16
+ "post_warmup": {
17
+ "commit_oid": "c0a5c82e20dd90872dd2e1ba0a06f92f98f08f03",
18
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/c0a5c82e20dd90872dd2e1ba0a06f92f98f08f03",
19
+ "files": 16,
20
+ "local_name": "checkpoint-4",
21
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-4",
22
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
23
+ "tree_sha256": "51f1edcfa4ca103a92fdbbd53e4d1bdc0f5264eba3ba1aa60621cab20c603f31",
24
+ "verified_at": "2026-08-15T01:43:20+00:00"
25
+ },
26
+ "step31": {
27
+ "commit_oid": "f225df41f35617335440fd45c89adf20e61ebb66",
28
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/f225df41f35617335440fd45c89adf20e61ebb66",
29
+ "files": 16,
30
+ "local_name": "checkpoint-31",
31
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-31",
32
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
33
+ "tree_sha256": "a34acde4e2a622a6ffbfd1912d8058fcf471be768537d690a769966a5f3f7fc7",
34
+ "verified_at": "2026-08-15T01:46:04+00:00"
35
+ },
36
+ "step62": {
37
+ "commit_oid": "3082a28661539aeac56366a2b203758242038186",
38
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/3082a28661539aeac56366a2b203758242038186",
39
+ "files": 16,
40
+ "local_name": "checkpoint-62",
41
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-62",
42
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
43
+ "tree_sha256": "cf7d6d4ebe35d33b9cd1d58a5701b993c146887dbc45b8fae8095dc555416c5f",
44
+ "verified_at": "2026-08-15T01:48:46+00:00"
45
+ },
46
+ "step93": {
47
+ "commit_oid": "fdc198123a2bc0720cbb48bfedd0ae393775fee4",
48
+ "commit_url": "https://huggingface.co/sidbaines/scimt-dispatch-4b-models-v1/commit/fdc198123a2bc0720cbb48bfedd0ae393775fee4",
49
+ "files": 16,
50
+ "local_name": "checkpoint-93",
51
+ "remote_prefix": "midtrain_4epoch/control/checkpoint-93",
52
+ "repo_id": "sidbaines/scimt-dispatch-4b-models-v1",
53
+ "tree_sha256": "010cb434ceba9690be1a5f05cce520e922a532129314029c4d032c0eeb8762ba",
54
+ "verified_at": "2026-08-15T01:51:27+00:00"
55
+ }
56
+ },
57
+ "completed_at": "2026-08-15T01:54:09+00:00",
58
+ "elapsed_seconds": 2761.146,
59
+ "expected_optimizer_steps": 124,
60
+ "loss": {
61
+ "first_loss": 1.6685791015625,
62
+ "global_step": 124,
63
+ "last_loss": 1.245849609375,
64
+ "log_history": [
65
+ {
66
+ "epoch": 0.0326530612244898,
67
+ "grad_norm": 3.96875,
68
+ "learning_rate": 0.0,
69
+ "loss": 1.6685791015625,
70
+ "memory/device_reserved (GiB)": 26.08,
71
+ "memory/max_active (GiB)": 20.03,
72
+ "memory/max_allocated (GiB)": 20.03,
73
+ "ppl": 5.30463,
74
+ "step": 1,
75
+ "tokens/total": 262144,
76
+ "tokens/train_per_sec_per_gpu": 471.78,
77
+ "tokens/trainable": 261922
78
+ },
79
+ {
80
+ "epoch": 0.0653061224489796,
81
+ "grad_norm": 4.21875,
82
+ "learning_rate": 3.3333333333333333e-06,
83
+ "loss": 1.5614013671875,
84
+ "memory/device_reserved (GiB)": 33.29,
85
+ "memory/max_active (GiB)": 27.26,
86
+ "memory/max_allocated (GiB)": 27.26,
87
+ "ppl": 4.76549,
88
+ "step": 2,
89
+ "tokens/total": 524288,
90
+ "tokens/train_per_sec_per_gpu": 597.28,
91
+ "tokens/trainable": 523573
92
+ },
93
+ {
94
+ "epoch": 0.09795918367346938,
95
+ "grad_norm": 3.40625,
96
+ "learning_rate": 6.666666666666667e-06,
97
+ "loss": 1.5543212890625,
98
+ "memory/device_reserved (GiB)": 33.29,
99
+ "memory/max_active (GiB)": 27.26,
100
+ "memory/max_allocated (GiB)": 27.26,
101
+ "ppl": 4.73187,
102
+ "step": 3,
103
+ "tokens/total": 786432,
104
+ "tokens/train_per_sec_per_gpu": 606.45,
105
+ "tokens/trainable": 785311
106
+ },
107
+ {
108
+ "epoch": 0.1306122448979592,
109
+ "grad_norm": 2.4375,
110
+ "learning_rate": 1e-05,
111
+ "loss": 1.575927734375,
112
+ "memory/device_reserved (GiB)": 33.29,
113
+ "memory/max_active (GiB)": 27.26,
114
+ "memory/max_allocated (GiB)": 27.26,
115
+ "ppl": 4.83523,
116
+ "step": 4,
117
+ "tokens/total": 1048576,
118
+ "tokens/train_per_sec_per_gpu": 609.43,
119
+ "tokens/trainable": 1047069
120
+ },
121
+ {
122
+ "epoch": 0.16326530612244897,
123
+ "grad_norm": 2.921875,
124
+ "learning_rate": 9.998483343865806e-06,
125
+ "loss": 1.529052734375,
126
+ "memory/device_reserved (GiB)": 33.29,
127
+ "memory/max_active (GiB)": 27.26,
128
+ "memory/max_allocated (GiB)": 27.26,
129
+ "ppl": 4.6138,
130
+ "step": 5,
131
+ "tokens/total": 1310720,
132
+ "tokens/train_per_sec_per_gpu": 604.52,
133
+ "tokens/trainable": 1308722
134
+ },
135
+ {
136
+ "epoch": 0.19591836734693877,
137
+ "grad_norm": 2.625,
138
+ "learning_rate": 9.993934397794704e-06,
139
+ "loss": 1.39306640625,
140
+ "memory/device_reserved (GiB)": 33.29,
141
+ "memory/max_active (GiB)": 27.26,
142
+ "memory/max_allocated (GiB)": 27.26,
143
+ "ppl": 4.02718,
144
+ "step": 6,
145
+ "tokens/total": 1572864,
146
+ "tokens/train_per_sec_per_gpu": 607.08,
147
+ "tokens/trainable": 1570345
148
+ },
149
+ {
150
+ "epoch": 0.22857142857142856,
151
+ "grad_norm": 1.734375,
152
+ "learning_rate": 9.986356228092011e-06,
153
+ "loss": 1.4849853515625,
154
+ "memory/device_reserved (GiB)": 33.29,
155
+ "memory/max_active (GiB)": 27.26,
156
+ "memory/max_allocated (GiB)": 27.26,
157
+ "ppl": 4.4149,
158
+ "step": 7,
159
+ "tokens/total": 1835008,
160
+ "tokens/train_per_sec_per_gpu": 605.44,
161
+ "tokens/trainable": 1832202
162
+ },
163
+ {
164
+ "epoch": 0.2612244897959184,
165
+ "grad_norm": 1.4296875,
166
+ "learning_rate": 9.975753942969978e-06,
167
+ "loss": 1.58740234375,
168
+ "memory/device_reserved (GiB)": 33.29,
169
+ "memory/max_active (GiB)": 27.26,
170
+ "memory/max_allocated (GiB)": 27.26,
171
+ "ppl": 4.89103,
172
+ "step": 8,
173
+ "tokens/total": 2097152,
174
+ "tokens/train_per_sec_per_gpu": 604.81,
175
+ "tokens/trainable": 2093962
176
+ },
177
+ {
178
+ "epoch": 0.2938775510204082,
179
+ "grad_norm": 1.46875,
180
+ "learning_rate": 9.962134689104498e-06,
181
+ "loss": 1.34619140625,
182
+ "memory/device_reserved (GiB)": 33.29,
183
+ "memory/max_active (GiB)": 27.26,
184
+ "memory/max_allocated (GiB)": 27.26,
185
+ "ppl": 3.84276,
186
+ "step": 9,
187
+ "tokens/total": 2359296,
188
+ "tokens/train_per_sec_per_gpu": 607.46,
189
+ "tokens/trainable": 2355679
190
+ },
191
+ {
192
+ "epoch": 0.32653061224489793,
193
+ "grad_norm": 1.25,
194
+ "learning_rate": 9.945507646817764e-06,
195
+ "loss": 1.47705078125,
196
+ "memory/device_reserved (GiB)": 33.29,
197
+ "memory/max_active (GiB)": 27.26,
198
+ "memory/max_allocated (GiB)": 27.26,
199
+ "ppl": 4.38001,
200
+ "step": 10,
201
+ "tokens/total": 2621440,
202
+ "tokens/train_per_sec_per_gpu": 606.8,
203
+ "tokens/trainable": 2617308
204
+ },
205
+ {
206
+ "epoch": 0.35918367346938773,
207
+ "grad_norm": 1.171875,
208
+ "learning_rate": 9.925884023890072e-06,
209
+ "loss": 1.4583740234375,
210
+ "memory/device_reserved (GiB)": 33.29,
211
+ "memory/max_active (GiB)": 27.26,
212
+ "memory/max_allocated (GiB)": 27.26,
213
+ "ppl": 4.29896,
214
+ "step": 11,
215
+ "tokens/total": 2883584,
216
+ "tokens/train_per_sec_per_gpu": 604.71,
217
+ "tokens/trainable": 2878962
218
+ },
219
+ {
220
+ "epoch": 0.39183673469387753,
221
+ "grad_norm": 1.203125,
222
+ "learning_rate": 9.903277048005017e-06,
223
+ "loss": 1.5108642578125,
224
+ "memory/device_reserved (GiB)": 33.29,
225
+ "memory/max_active (GiB)": 27.26,
226
+ "memory/max_allocated (GiB)": 27.26,
227
+ "ppl": 4.53064,
228
+ "step": 12,
229
+ "tokens/total": 3145728,
230
+ "tokens/train_per_sec_per_gpu": 603.72,
231
+ "tokens/trainable": 3140634
232
+ },
233
+ {
234
+ "epoch": 0.42448979591836733,
235
+ "grad_norm": 1.0234375,
236
+ "learning_rate": 9.877701957833113e-06,
237
+ "loss": 1.3990478515625,
238
+ "memory/device_reserved (GiB)": 33.29,
239
+ "memory/max_active (GiB)": 27.26,
240
+ "memory/max_allocated (GiB)": 27.26,
241
+ "ppl": 4.05134,
242
+ "step": 13,
243
+ "tokens/total": 3407872,
244
+ "tokens/train_per_sec_per_gpu": 604.61,
245
+ "tokens/trainable": 3402298
246
+ },
247
+ {
248
+ "epoch": 0.45714285714285713,
249
+ "grad_norm": 0.9375,
250
+ "learning_rate": 9.849175992759867e-06,
251
+ "loss": 1.32666015625,
252
+ "memory/device_reserved (GiB)": 33.29,
253
+ "memory/max_active (GiB)": 27.26,
254
+ "memory/max_allocated (GiB)": 27.26,
255
+ "ppl": 3.76844,
256
+ "step": 14,
257
+ "tokens/total": 3670016,
258
+ "tokens/train_per_sec_per_gpu": 605.87,
259
+ "tokens/trainable": 3663962
260
+ },
261
+ {
262
+ "epoch": 0.4897959183673469,
263
+ "grad_norm": 1.7734375,
264
+ "learning_rate": 9.81771838126524e-06,
265
+ "loss": 1.46875,
266
+ "memory/device_reserved (GiB)": 33.29,
267
+ "memory/max_active (GiB)": 27.26,
268
+ "memory/max_allocated (GiB)": 27.26,
269
+ "ppl": 4.3438,
270
+ "step": 15,
271
+ "tokens/total": 3932160,
272
+ "tokens/train_per_sec_per_gpu": 603.17,
273
+ "tokens/trainable": 3925695
274
+ },
275
+ {
276
+ "epoch": 0.5224489795918368,
277
+ "grad_norm": 1.7890625,
278
+ "learning_rate": 9.783350327962313e-06,
279
+ "loss": 1.409912109375,
280
+ "memory/device_reserved (GiB)": 33.29,
281
+ "memory/max_active (GiB)": 27.26,
282
+ "memory/max_allocated (GiB)": 27.26,
283
+ "ppl": 4.0956,
284
+ "step": 16,
285
+ "tokens/total": 4194304,
286
+ "tokens/train_per_sec_per_gpu": 603.95,
287
+ "tokens/trainable": 4187301
288
+ },
289
+ {
290
+ "epoch": 0.5551020408163265,
291
+ "grad_norm": 3.03125,
292
+ "learning_rate": 9.74609499930392e-06,
293
+ "loss": 1.401123046875,
294
+ "memory/device_reserved (GiB)": 33.29,
295
+ "memory/max_active (GiB)": 27.26,
296
+ "memory/max_allocated (GiB)": 27.26,
297
+ "ppl": 4.05976,
298
+ "step": 17,
299
+ "tokens/total": 4456448,
300
+ "tokens/train_per_sec_per_gpu": 603.72,
301
+ "tokens/trainable": 4449140
302
+ },
303
+ {
304
+ "epoch": 0.5877551020408164,
305
+ "grad_norm": 1.015625,
306
+ "learning_rate": 9.70597750796683e-06,
307
+ "loss": 1.476318359375,
308
+ "memory/device_reserved (GiB)": 33.29,
309
+ "memory/max_active (GiB)": 27.26,
310
+ "memory/max_allocated (GiB)": 27.26,
311
+ "ppl": 4.3768,
312
+ "step": 18,
313
+ "tokens/total": 4718592,
314
+ "tokens/train_per_sec_per_gpu": 604.17,
315
+ "tokens/trainable": 4710654
316
+ },
317
+ {
318
+ "epoch": 0.6204081632653061,
319
+ "grad_norm": 0.9453125,
320
+ "learning_rate": 9.663024895924078e-06,
321
+ "loss": 1.2203369140625,
322
+ "memory/device_reserved (GiB)": 33.29,
323
+ "memory/max_active (GiB)": 27.26,
324
+ "memory/max_allocated (GiB)": 27.26,
325
+ "ppl": 3.38833,
326
+ "step": 19,
327
+ "tokens/total": 4980736,
328
+ "tokens/train_per_sec_per_gpu": 605.11,
329
+ "tokens/trainable": 4972132
330
+ },
331
+ {
332
+ "epoch": 0.6530612244897959,
333
+ "grad_norm": 1.0078125,
334
+ "learning_rate": 9.61726611621679e-06,
335
+ "loss": 1.451171875,
336
+ "memory/device_reserved (GiB)": 33.29,
337
+ "memory/max_active (GiB)": 27.26,
338
+ "memory/max_allocated (GiB)": 27.26,
339
+ "ppl": 4.26811,
340
+ "step": 20,
341
+ "tokens/total": 5242880,
342
+ "tokens/train_per_sec_per_gpu": 603.4,
343
+ "tokens/trainable": 5233845
344
+ },
345
+ {
346
+ "epoch": 0.6857142857142857,
347
+ "grad_norm": 0.83203125,
348
+ "learning_rate": 9.568732013437827e-06,
349
+ "loss": 1.3369140625,
350
+ "memory/device_reserved (GiB)": 33.29,
351
+ "memory/max_active (GiB)": 27.26,
352
+ "memory/max_allocated (GiB)": 27.26,
353
+ "ppl": 3.80728,
354
+ "step": 21,
355
+ "tokens/total": 5505024,
356
+ "tokens/train_per_sec_per_gpu": 604.73,
357
+ "tokens/trainable": 5495423
358
+ },
359
+ {
360
+ "epoch": 0.7183673469387755,
361
+ "grad_norm": 0.953125,
362
+ "learning_rate": 9.517455302940388e-06,
363
+ "loss": 1.4034423828125,
364
+ "memory/device_reserved (GiB)": 33.29,
365
+ "memory/max_active (GiB)": 27.26,
366
+ "memory/max_allocated (GiB)": 27.26,
367
+ "ppl": 4.06918,
368
+ "step": 22,
369
+ "tokens/total": 5767168,
370
+ "tokens/train_per_sec_per_gpu": 605.55,
371
+ "tokens/trainable": 5756905
372
+ },
373
+ {
374
+ "epoch": 0.7510204081632653,
375
+ "grad_norm": 0.78125,
376
+ "learning_rate": 9.46347054878559e-06,
377
+ "loss": 1.3974609375,
378
+ "memory/device_reserved (GiB)": 33.29,
379
+ "memory/max_active (GiB)": 27.26,
380
+ "memory/max_allocated (GiB)": 27.26,
381
+ "ppl": 4.04492,
382
+ "step": 23,
383
+ "tokens/total": 6029312,
384
+ "tokens/train_per_sec_per_gpu": 601.68,
385
+ "tokens/trainable": 6018394
386
+ },
387
+ {
388
+ "epoch": 0.7836734693877551,
389
+ "grad_norm": 0.84375,
390
+ "learning_rate": 9.406814140443898e-06,
391
+ "loss": 1.446533203125,
392
+ "memory/device_reserved (GiB)": 33.29,
393
+ "memory/max_active (GiB)": 27.26,
394
+ "memory/max_allocated (GiB)": 27.26,
395
+ "ppl": 4.24836,
396
+ "step": 24,
397
+ "tokens/total": 6291456,
398
+ "tokens/train_per_sec_per_gpu": 602.65,
399
+ "tokens/trainable": 6279921
400
+ },
401
+ {
402
+ "epoch": 0.8163265306122449,
403
+ "grad_norm": 0.94140625,
404
+ "learning_rate": 9.347524268266092e-06,
405
+ "loss": 1.303955078125,
406
+ "memory/device_reserved (GiB)": 33.29,
407
+ "memory/max_active (GiB)": 27.26,
408
+ "memory/max_allocated (GiB)": 27.26,
409
+ "ppl": 3.68384,
410
+ "step": 25,
411
+ "tokens/total": 6553600,
412
+ "tokens/train_per_sec_per_gpu": 605.26,
413
+ "tokens/trainable": 6541489
414
+ },
415
+ {
416
+ "epoch": 0.8489795918367347,
417
+ "grad_norm": 0.9453125,
418
+ "learning_rate": 9.285640897740316e-06,
419
+ "loss": 1.4931640625,
420
+ "memory/device_reserved (GiB)": 33.29,
421
+ "memory/max_active (GiB)": 27.26,
422
+ "memory/max_allocated (GiB)": 27.26,
423
+ "ppl": 4.45116,
424
+ "step": 26,
425
+ "tokens/total": 6815744,
426
+ "tokens/train_per_sec_per_gpu": 603.79,
427
+ "tokens/trainable": 6803233
428
+ },
429
+ {
430
+ "epoch": 0.8816326530612245,
431
+ "grad_norm": 0.953125,
432
+ "learning_rate": 9.22120574255258e-06,
433
+ "loss": 1.2637939453125,
434
+ "memory/device_reserved (GiB)": 33.29,
435
+ "memory/max_active (GiB)": 27.26,
436
+ "memory/max_allocated (GiB)": 27.26,
437
+ "ppl": 3.53882,
438
+ "step": 27,
439
+ "tokens/total": 7077888,
440
+ "tokens/train_per_sec_per_gpu": 578.56,
441
+ "tokens/trainable": 7064799
442
+ },
443
+ {
444
+ "epoch": 0.9142857142857143,
445
+ "grad_norm": 0.75390625,
446
+ "learning_rate": 9.154262236468826e-06,
447
+ "loss": 1.222412109375,
448
+ "memory/device_reserved (GiB)": 33.29,
449
+ "memory/max_active (GiB)": 27.26,
450
+ "memory/max_allocated (GiB)": 27.26,
451
+ "ppl": 3.39537,
452
+ "step": 28,
453
+ "tokens/total": 7340032,
454
+ "tokens/train_per_sec_per_gpu": 605.17,
455
+ "tokens/trainable": 7326396
456
+ },
457
+ {
458
+ "epoch": 0.9469387755102041,
459
+ "grad_norm": 0.83984375,
460
+ "learning_rate": 9.084855504057562e-06,
461
+ "loss": 1.306884765625,
462
+ "memory/device_reserved (GiB)": 33.29,
463
+ "memory/max_active (GiB)": 27.26,
464
+ "memory/max_allocated (GiB)": 27.26,
465
+ "ppl": 3.69465,
466
+ "step": 29,
467
+ "tokens/total": 7602176,
468
+ "tokens/train_per_sec_per_gpu": 606.42,
469
+ "tokens/trainable": 7587940
470
+ },
471
+ {
472
+ "epoch": 0.9795918367346939,
473
+ "grad_norm": 0.83984375,
474
+ "learning_rate": 9.013032330272777e-06,
475
+ "loss": 1.3385009765625,
476
+ "memory/device_reserved (GiB)": 33.29,
477
+ "memory/max_active (GiB)": 27.26,
478
+ "memory/max_allocated (GiB)": 27.26,
479
+ "ppl": 3.81332,
480
+ "step": 30,
481
+ "tokens/total": 7864320,
482
+ "tokens/train_per_sec_per_gpu": 606.34,
483
+ "tokens/trainable": 7849389
484
+ },
485
+ {
486
+ "epoch": 1.0,
487
+ "grad_norm": 0.9375,
488
+ "learning_rate": 8.938841128917622e-06,
489
+ "loss": 1.3125,
490
+ "memory/device_reserved (GiB)": 33.29,
491
+ "memory/max_active (GiB)": 27.26,
492
+ "memory/max_allocated (GiB)": 27.26,
493
+ "ppl": 3.71545,
494
+ "step": 31,
495
+ "tokens/total": 8028160,
496
+ "tokens/train_per_sec_per_gpu": 912.97,
497
+ "tokens/trainable": 8012073
498
+ },
499
+ {
500
+ "epoch": 1.0326530612244897,
501
+ "grad_norm": 0.87109375,
502
+ "learning_rate": 8.86233191001016e-06,
503
+ "loss": 1.4200439453125,
504
+ "memory/device_reserved (GiB)": 33.29,
505
+ "memory/max_active (GiB)": 27.26,
506
+ "memory/max_allocated (GiB)": 27.26,
507
+ "ppl": 4.1373,
508
+ "step": 32,
509
+ "tokens/total": 8290304,
510
+ "tokens/train_per_sec_per_gpu": 584.93,
511
+ "tokens/trainable": 8273995
512
+ },
513
+ {
514
+ "epoch": 1.0653061224489795,
515
+ "grad_norm": 0.890625,
516
+ "learning_rate": 8.783556246073135e-06,
517
+ "loss": 1.3094482421875,
518
+ "memory/device_reserved (GiB)": 33.29,
519
+ "memory/max_active (GiB)": 27.26,
520
+ "memory/max_allocated (GiB)": 27.26,
521
+ "ppl": 3.70413,
522
+ "step": 33,
523
+ "tokens/total": 8552448,
524
+ "tokens/train_per_sec_per_gpu": 607.05,
525
+ "tokens/trainable": 8535646
526
+ },
527
+ {
528
+ "epoch": 1.0979591836734695,
529
+ "grad_norm": 0.9375,
530
+ "learning_rate": 8.702567237370521e-06,
531
+ "loss": 1.317138671875,
532
+ "memory/device_reserved (GiB)": 33.29,
533
+ "memory/max_active (GiB)": 27.26,
534
+ "memory/max_allocated (GiB)": 27.26,
535
+ "ppl": 3.73273,
536
+ "step": 34,
537
+ "tokens/total": 8814592,
538
+ "tokens/train_per_sec_per_gpu": 605.99,
539
+ "tokens/trainable": 8797384
540
+ },
541
+ {
542
+ "epoch": 1.1306122448979592,
543
+ "grad_norm": 0.80078125,
544
+ "learning_rate": 8.619419476114251e-06,
545
+ "loss": 1.3460693359375,
546
+ "memory/device_reserved (GiB)": 33.29,
547
+ "memory/max_active (GiB)": 27.26,
548
+ "memory/max_allocated (GiB)": 27.26,
549
+ "ppl": 3.84229,
550
+ "step": 35,
551
+ "tokens/total": 9076736,
552
+ "tokens/train_per_sec_per_gpu": 606.76,
553
+ "tokens/trainable": 9059142
554
+ },
555
+ {
556
+ "epoch": 1.163265306122449,
557
+ "grad_norm": 0.83984375,
558
+ "learning_rate": 8.534169009665282e-06,
559
+ "loss": 1.353759765625,
560
+ "memory/device_reserved (GiB)": 33.29,
561
+ "memory/max_active (GiB)": 27.26,
562
+ "memory/max_allocated (GiB)": 27.26,
563
+ "ppl": 3.87196,
564
+ "step": 36,
565
+ "tokens/total": 9338880,
566
+ "tokens/train_per_sec_per_gpu": 607.47,
567
+ "tokens/trainable": 9320795
568
+ },
569
+ {
570
+ "epoch": 1.1959183673469387,
571
+ "grad_norm": 0.80859375,
572
+ "learning_rate": 8.446873302753783e-06,
573
+ "loss": 1.2401123046875,
574
+ "memory/device_reserved (GiB)": 33.29,
575
+ "memory/max_active (GiB)": 27.26,
576
+ "memory/max_allocated (GiB)": 27.26,
577
+ "ppl": 3.456,
578
+ "step": 37,
579
+ "tokens/total": 9601024,
580
+ "tokens/train_per_sec_per_gpu": 605.46,
581
+ "tokens/trainable": 9582418
582
+ },
583
+ {
584
+ "epoch": 1.2285714285714286,
585
+ "grad_norm": 0.765625,
586
+ "learning_rate": 8.357591198743923e-06,
587
+ "loss": 1.343017578125,
588
+ "memory/device_reserved (GiB)": 33.29,
589
+ "memory/max_active (GiB)": 27.26,
590
+ "memory/max_allocated (GiB)": 27.26,
591
+ "ppl": 3.83059,
592
+ "step": 38,
593
+ "tokens/total": 9863168,
594
+ "tokens/train_per_sec_per_gpu": 603.72,
595
+ "tokens/trainable": 9844275
596
+ },
597
+ {
598
+ "epoch": 1.2612244897959184,
599
+ "grad_norm": 0.80078125,
600
+ "learning_rate": 8.266382879969356e-06,
601
+ "loss": 1.466552734375,
602
+ "memory/device_reserved (GiB)": 33.29,
603
+ "memory/max_active (GiB)": 27.26,
604
+ "memory/max_allocated (GiB)": 27.26,
605
+ "ppl": 4.33427,
606
+ "step": 39,
607
+ "tokens/total": 10125312,
608
+ "tokens/train_per_sec_per_gpu": 604.71,
609
+ "tokens/trainable": 10106035
610
+ },
611
+ {
612
+ "epoch": 1.2938775510204081,
613
+ "grad_norm": 0.7578125,
614
+ "learning_rate": 8.17330982716615e-06,
615
+ "loss": 1.2305908203125,
616
+ "memory/device_reserved (GiB)": 33.29,
617
+ "memory/max_active (GiB)": 27.26,
618
+ "memory/max_allocated (GiB)": 27.26,
619
+ "ppl": 3.42325,
620
+ "step": 40,
621
+ "tokens/total": 10387456,
622
+ "tokens/train_per_sec_per_gpu": 607.6,
623
+ "tokens/trainable": 10367752
624
+ },
625
+ {
626
+ "epoch": 1.3265306122448979,
627
+ "grad_norm": 0.875,
628
+ "learning_rate": 8.078434778030511e-06,
629
+ "loss": 1.3778076171875,
630
+ "memory/device_reserved (GiB)": 33.29,
631
+ "memory/max_active (GiB)": 27.26,
632
+ "memory/max_allocated (GiB)": 27.26,
633
+ "ppl": 3.9662,
634
+ "step": 41,
635
+ "tokens/total": 10649600,
636
+ "tokens/train_per_sec_per_gpu": 605.02,
637
+ "tokens/trainable": 10629381
638
+ },
639
+ {
640
+ "epoch": 1.3591836734693876,
641
+ "grad_norm": 0.765625,
642
+ "learning_rate": 7.981821684929218e-06,
643
+ "loss": 1.362060546875,
644
+ "memory/device_reserved (GiB)": 33.29,
645
+ "memory/max_active (GiB)": 27.26,
646
+ "memory/max_allocated (GiB)": 27.26,
647
+ "ppl": 3.90423,
648
+ "step": 42,
649
+ "tokens/total": 10911744,
650
+ "tokens/train_per_sec_per_gpu": 603.78,
651
+ "tokens/trainable": 10891035
652
+ },
653
+ {
654
+ "epoch": 1.3918367346938776,
655
+ "grad_norm": 0.8046875,
656
+ "learning_rate": 7.883535671791294e-06,
657
+ "loss": 1.420166015625,
658
+ "memory/device_reserved (GiB)": 33.29,
659
+ "memory/max_active (GiB)": 27.26,
660
+ "memory/max_allocated (GiB)": 27.26,
661
+ "ppl": 4.13781,
662
+ "step": 43,
663
+ "tokens/total": 11173888,
664
+ "tokens/train_per_sec_per_gpu": 604.13,
665
+ "tokens/trainable": 11152707
666
+ },
667
+ {
668
+ "epoch": 1.4244897959183673,
669
+ "grad_norm": 0.83984375,
670
+ "learning_rate": 7.783642990209951e-06,
671
+ "loss": 1.31005859375,
672
+ "memory/device_reserved (GiB)": 33.29,
673
+ "memory/max_active (GiB)": 27.26,
674
+ "memory/max_allocated (GiB)": 27.26,
675
+ "ppl": 3.70639,
676
+ "step": 44,
677
+ "tokens/total": 11436032,
678
+ "tokens/train_per_sec_per_gpu": 605.21,
679
+ "tokens/trainable": 11414371
680
+ },
681
+ {
682
+ "epoch": 1.457142857142857,
683
+ "grad_norm": 0.77734375,
684
+ "learning_rate": 7.682210974784426e-06,
685
+ "loss": 1.2432861328125,
686
+ "memory/device_reserved (GiB)": 33.29,
687
+ "memory/max_active (GiB)": 27.26,
688
+ "memory/max_allocated (GiB)": 27.26,
689
+ "ppl": 3.46699,
690
+ "step": 45,
691
+ "tokens/total": 11698176,
692
+ "tokens/train_per_sec_per_gpu": 605.6,
693
+ "tokens/trainable": 11676035
694
+ },
695
+ {
696
+ "epoch": 1.489795918367347,
697
+ "grad_norm": 0.85546875,
698
+ "learning_rate": 7.579307997731783e-06,
699
+ "loss": 1.4068603515625,
700
+ "memory/device_reserved (GiB)": 33.29,
701
+ "memory/max_active (GiB)": 27.26,
702
+ "memory/max_allocated (GiB)": 27.26,
703
+ "ppl": 4.08312,
704
+ "step": 46,
705
+ "tokens/total": 11960320,
706
+ "tokens/train_per_sec_per_gpu": 603.01,
707
+ "tokens/trainable": 11937768
708
+ },
709
+ {
710
+ "epoch": 1.5224489795918368,
711
+ "grad_norm": 0.83984375,
712
+ "learning_rate": 7.475003422799302e-06,
713
+ "loss": 1.3421630859375,
714
+ "memory/device_reserved (GiB)": 33.29,
715
+ "memory/max_active (GiB)": 27.26,
716
+ "memory/max_allocated (GiB)": 27.26,
717
+ "ppl": 3.82731,
718
+ "step": 47,
719
+ "tokens/total": 12222464,
720
+ "tokens/train_per_sec_per_gpu": 603.3,
721
+ "tokens/trainable": 12199374
722
+ },
723
+ {
724
+ "epoch": 1.5551020408163265,
725
+ "grad_norm": 0.8828125,
726
+ "learning_rate": 7.36936755850849e-06,
727
+ "loss": 1.3466796875,
728
+ "memory/device_reserved (GiB)": 33.29,
729
+ "memory/max_active (GiB)": 27.26,
730
+ "memory/max_allocated (GiB)": 27.26,
731
+ "ppl": 3.84464,
732
+ "step": 48,
733
+ "tokens/total": 12484608,
734
+ "tokens/train_per_sec_per_gpu": 605.0,
735
+ "tokens/trainable": 12461213
736
+ },
737
+ {
738
+ "epoch": 1.5877551020408163,
739
+ "grad_norm": 0.81640625,
740
+ "learning_rate": 7.2624716107622675e-06,
741
+ "loss": 1.404296875,
742
+ "memory/device_reserved (GiB)": 33.29,
743
+ "memory/max_active (GiB)": 27.26,
744
+ "memory/max_allocated (GiB)": 27.26,
745
+ "ppl": 4.07266,
746
+ "step": 49,
747
+ "tokens/total": 12746752,
748
+ "tokens/train_per_sec_per_gpu": 604.6,
749
+ "tokens/trainable": 12722727
750
+ },
751
+ {
752
+ "epoch": 1.620408163265306,
753
+ "grad_norm": 0.73046875,
754
+ "learning_rate": 7.154387634847241e-06,
755
+ "loss": 1.1488037109375,
756
+ "memory/device_reserved (GiB)": 33.29,
757
+ "memory/max_active (GiB)": 27.26,
758
+ "memory/max_allocated (GiB)": 27.26,
759
+ "ppl": 3.15442,
760
+ "step": 50,
761
+ "tokens/total": 13008896,
762
+ "tokens/train_per_sec_per_gpu": 605.31,
763
+ "tokens/trainable": 12984205
764
+ },
765
+ {
766
+ "epoch": 1.6530612244897958,
767
+ "grad_norm": 0.79296875,
768
+ "learning_rate": 7.045188486863449e-06,
769
+ "loss": 1.38818359375,
770
+ "memory/device_reserved (GiB)": 33.29,
771
+ "memory/max_active (GiB)": 27.26,
772
+ "memory/max_allocated (GiB)": 27.26,
773
+ "ppl": 4.00756,
774
+ "step": 51,
775
+ "tokens/total": 13271040,
776
+ "tokens/train_per_sec_per_gpu": 603.17,
777
+ "tokens/trainable": 13245918
778
+ },
779
+ {
780
+ "epoch": 1.6857142857142857,
781
+ "grad_norm": 0.75,
782
+ "learning_rate": 6.9349477746142846e-06,
783
+ "loss": 1.2738037109375,
784
+ "memory/device_reserved (GiB)": 33.29,
785
+ "memory/max_active (GiB)": 27.26,
786
+ "memory/max_allocated (GiB)": 27.26,
787
+ "ppl": 3.57442,
788
+ "step": 52,
789
+ "tokens/total": 13533184,
790
+ "tokens/train_per_sec_per_gpu": 606.57,
791
+ "tokens/trainable": 13507496
792
+ },
793
+ {
794
+ "epoch": 1.7183673469387755,
795
+ "grad_norm": 0.84375,
796
+ "learning_rate": 6.823739807989734e-06,
797
+ "loss": 1.3431396484375,
798
+ "memory/device_reserved (GiB)": 33.29,
799
+ "memory/max_active (GiB)": 27.26,
800
+ "memory/max_allocated (GiB)": 27.26,
801
+ "ppl": 3.83105,
802
+ "step": 53,
803
+ "tokens/total": 13795328,
804
+ "tokens/train_per_sec_per_gpu": 606.26,
805
+ "tokens/trainable": 13768978
806
+ },
807
+ {
808
+ "epoch": 1.7510204081632654,
809
+ "grad_norm": 0.7265625,
810
+ "learning_rate": 6.7116395488763565e-06,
811
+ "loss": 1.3397216796875,
812
+ "memory/device_reserved (GiB)": 33.29,
813
+ "memory/max_active (GiB)": 27.26,
814
+ "memory/max_allocated (GiB)": 27.26,
815
+ "ppl": 3.81798,
816
+ "step": 54,
817
+ "tokens/total": 14057472,
818
+ "tokens/train_per_sec_per_gpu": 602.75,
819
+ "tokens/trainable": 14030467
820
+ },
821
+ {
822
+ "epoch": 1.7836734693877552,
823
+ "grad_norm": 0.7265625,
824
+ "learning_rate": 6.598722560627761e-06,
825
+ "loss": 1.3912353515625,
826
+ "memory/device_reserved (GiB)": 33.29,
827
+ "memory/max_active (GiB)": 27.26,
828
+ "memory/max_allocated (GiB)": 27.26,
829
+ "ppl": 4.01981,
830
+ "step": 55,
831
+ "tokens/total": 14319616,
832
+ "tokens/train_per_sec_per_gpu": 571.54,
833
+ "tokens/trainable": 14291994
834
+ },
835
+ {
836
+ "epoch": 1.816326530612245,
837
+ "grad_norm": 0.8515625,
838
+ "learning_rate": 6.485064957129677e-06,
839
+ "loss": 1.2479248046875,
840
+ "memory/device_reserved (GiB)": 33.29,
841
+ "memory/max_active (GiB)": 27.26,
842
+ "memory/max_allocated (GiB)": 27.26,
843
+ "ppl": 3.48311,
844
+ "step": 56,
845
+ "tokens/total": 14581760,
846
+ "tokens/train_per_sec_per_gpu": 606.07,
847
+ "tokens/trainable": 14553562
848
+ },
849
+ {
850
+ "epoch": 1.8489795918367347,
851
+ "grad_norm": 0.82421875,
852
+ "learning_rate": 6.370743351493899e-06,
853
+ "loss": 1.4447021484375,
854
+ "memory/device_reserved (GiB)": 33.29,
855
+ "memory/max_active (GiB)": 27.26,
856
+ "memory/max_allocated (GiB)": 27.26,
857
+ "ppl": 4.24059,
858
+ "step": 57,
859
+ "tokens/total": 14843904,
860
+ "tokens/train_per_sec_per_gpu": 604.39,
861
+ "tokens/trainable": 14815306
862
+ },
863
+ {
864
+ "epoch": 1.8816326530612244,
865
+ "grad_norm": 0.75390625,
866
+ "learning_rate": 6.255834804415742e-06,
867
+ "loss": 1.2120361328125,
868
+ "memory/device_reserved (GiB)": 33.29,
869
+ "memory/max_active (GiB)": 27.26,
870
+ "memory/max_allocated (GiB)": 27.26,
871
+ "ppl": 3.36032,
872
+ "step": 58,
873
+ "tokens/total": 15106048,
874
+ "tokens/train_per_sec_per_gpu": 606.32,
875
+ "tokens/trainable": 15076872
876
+ },
877
+ {
878
+ "epoch": 1.9142857142857141,
879
+ "grad_norm": 0.83203125,
880
+ "learning_rate": 6.140416772229785e-06,
881
+ "loss": 1.175048828125,
882
+ "memory/device_reserved (GiB)": 33.29,
883
+ "memory/max_active (GiB)": 27.26,
884
+ "memory/max_allocated (GiB)": 27.26,
885
+ "ppl": 3.2383,
886
+ "step": 59,
887
+ "tokens/total": 15368192,
888
+ "tokens/train_per_sec_per_gpu": 604.24,
889
+ "tokens/trainable": 15338469
890
+ },
891
+ {
892
+ "epoch": 1.9469387755102041,
893
+ "grad_norm": 0.71875,
894
+ "learning_rate": 6.0245670546989165e-06,
895
+ "loss": 1.2623291015625,
896
+ "memory/device_reserved (GiB)": 33.29,
897
+ "memory/max_active (GiB)": 27.26,
898
+ "memory/max_allocated (GiB)": 27.26,
899
+ "ppl": 3.53364,
900
+ "step": 60,
901
+ "tokens/total": 15630336,
902
+ "tokens/train_per_sec_per_gpu": 606.62,
903
+ "tokens/trainable": 15600013
904
+ },
905
+ {
906
+ "epoch": 1.9795918367346939,
907
+ "grad_norm": 0.7578125,
908
+ "learning_rate": 5.908363742571915e-06,
909
+ "loss": 1.291015625,
910
+ "memory/device_reserved (GiB)": 33.29,
911
+ "memory/max_active (GiB)": 27.26,
912
+ "memory/max_allocated (GiB)": 27.26,
913
+ "ppl": 3.63648,
914
+ "step": 61,
915
+ "tokens/total": 15892480,
916
+ "tokens/train_per_sec_per_gpu": 606.98,
917
+ "tokens/trainable": 15861462
918
+ },
919
+ {
920
+ "epoch": 2.0,
921
+ "grad_norm": 1.0078125,
922
+ "learning_rate": 5.791885164944844e-06,
923
+ "loss": 1.26220703125,
924
+ "memory/device_reserved (GiB)": 33.29,
925
+ "memory/max_active (GiB)": 27.26,
926
+ "memory/max_allocated (GiB)": 27.26,
927
+ "ppl": 3.53321,
928
+ "step": 62,
929
+ "tokens/total": 16056320,
930
+ "tokens/train_per_sec_per_gpu": 920.2,
931
+ "tokens/trainable": 16024146
932
+ },
933
+ {
934
+ "epoch": 2.0326530612244897,
935
+ "grad_norm": 0.81640625,
936
+ "learning_rate": 5.67520983646182e-06,
937
+ "loss": 1.37841796875,
938
+ "memory/device_reserved (GiB)": 33.29,
939
+ "memory/max_active (GiB)": 27.26,
940
+ "memory/max_allocated (GiB)": 27.26,
941
+ "ppl": 3.96862,
942
+ "step": 63,
943
+ "tokens/total": 16318464,
944
+ "tokens/train_per_sec_per_gpu": 585.03,
945
+ "tokens/trainable": 16286068
946
+ },
947
+ {
948
+ "epoch": 2.0653061224489795,
949
+ "grad_norm": 1.109375,
950
+ "learning_rate": 5.5584164043906895e-06,
951
+ "loss": 1.271728515625,
952
+ "memory/device_reserved (GiB)": 33.29,
953
+ "memory/max_active (GiB)": 27.26,
954
+ "memory/max_allocated (GiB)": 27.26,
955
+ "ppl": 3.56701,
956
+ "step": 64,
957
+ "tokens/total": 16580608,
958
+ "tokens/train_per_sec_per_gpu": 607.21,
959
+ "tokens/trainable": 16547719
960
+ },
961
+ {
962
+ "epoch": 2.0979591836734692,
963
+ "grad_norm": 1.0390625,
964
+ "learning_rate": 5.441583595609312e-06,
965
+ "loss": 1.28076171875,
966
+ "memory/device_reserved (GiB)": 33.29,
967
+ "memory/max_active (GiB)": 27.26,
968
+ "memory/max_allocated (GiB)": 27.26,
969
+ "ppl": 3.59938,
970
+ "step": 65,
971
+ "tokens/total": 16842752,
972
+ "tokens/train_per_sec_per_gpu": 605.49,
973
+ "tokens/trainable": 16809456
974
+ },
975
+ {
976
+ "epoch": 2.130612244897959,
977
+ "grad_norm": 0.86328125,
978
+ "learning_rate": 5.324790163538181e-06,
979
+ "loss": 1.3126220703125,
980
+ "memory/device_reserved (GiB)": 33.29,
981
+ "memory/max_active (GiB)": 27.26,
982
+ "memory/max_allocated (GiB)": 27.26,
983
+ "ppl": 3.7159,
984
+ "step": 66,
985
+ "tokens/total": 17104896,
986
+ "tokens/train_per_sec_per_gpu": 606.91,
987
+ "tokens/trainable": 17071218
988
+ },
989
+ {
990
+ "epoch": 2.163265306122449,
991
+ "grad_norm": 0.84765625,
992
+ "learning_rate": 5.208114835055157e-06,
993
+ "loss": 1.32037353515625,
994
+ "memory/device_reserved (GiB)": 33.29,
995
+ "memory/max_active (GiB)": 27.26,
996
+ "memory/max_allocated (GiB)": 27.26,
997
+ "ppl": 3.74482,
998
+ "step": 67,
999
+ "tokens/total": 17367040,
1000
+ "tokens/train_per_sec_per_gpu": 607.11,
1001
+ "tokens/trainable": 17332876
1002
+ },
1003
+ {
1004
+ "epoch": 2.195918367346939,
1005
+ "grad_norm": 0.72265625,
1006
+ "learning_rate": 5.0916362574280864e-06,
1007
+ "loss": 1.2105712890625,
1008
+ "memory/device_reserved (GiB)": 33.29,
1009
+ "memory/max_active (GiB)": 27.26,
1010
+ "memory/max_allocated (GiB)": 27.26,
1011
+ "ppl": 3.3554,
1012
+ "step": 68,
1013
+ "tokens/total": 17629184,
1014
+ "tokens/train_per_sec_per_gpu": 605.15,
1015
+ "tokens/trainable": 17594500
1016
+ },
1017
+ {
1018
+ "epoch": 2.2285714285714286,
1019
+ "grad_norm": 0.8515625,
1020
+ "learning_rate": 4.975432945301085e-06,
1021
+ "loss": 1.3128662109375,
1022
+ "memory/device_reserved (GiB)": 33.29,
1023
+ "memory/max_active (GiB)": 27.26,
1024
+ "memory/max_allocated (GiB)": 27.26,
1025
+ "ppl": 3.71681,
1026
+ "step": 69,
1027
+ "tokens/total": 17891328,
1028
+ "tokens/train_per_sec_per_gpu": 603.11,
1029
+ "tokens/trainable": 17856356
1030
+ },
1031
+ {
1032
+ "epoch": 2.2612244897959184,
1033
+ "grad_norm": 0.78515625,
1034
+ "learning_rate": 4.859583227770218e-06,
1035
+ "loss": 1.4384765625,
1036
+ "memory/device_reserved (GiB)": 33.29,
1037
+ "memory/max_active (GiB)": 27.26,
1038
+ "memory/max_allocated (GiB)": 27.26,
1039
+ "ppl": 4.21427,
1040
+ "step": 70,
1041
+ "tokens/total": 18153472,
1042
+ "tokens/train_per_sec_per_gpu": 604.33,
1043
+ "tokens/trainable": 18118114
1044
+ },
1045
+ {
1046
+ "epoch": 2.293877551020408,
1047
+ "grad_norm": 0.80859375,
1048
+ "learning_rate": 4.744165195584258e-06,
1049
+ "loss": 1.203125,
1050
+ "memory/device_reserved (GiB)": 33.29,
1051
+ "memory/max_active (GiB)": 27.26,
1052
+ "memory/max_allocated (GiB)": 27.26,
1053
+ "ppl": 3.33051,
1054
+ "step": 71,
1055
+ "tokens/total": 18415616,
1056
+ "tokens/train_per_sec_per_gpu": 606.6,
1057
+ "tokens/trainable": 18379828
1058
+ },
1059
+ {
1060
+ "epoch": 2.326530612244898,
1061
+ "grad_norm": 0.82421875,
1062
+ "learning_rate": 4.6292566485061015e-06,
1063
+ "loss": 1.352294921875,
1064
+ "memory/device_reserved (GiB)": 33.29,
1065
+ "memory/max_active (GiB)": 27.26,
1066
+ "memory/max_allocated (GiB)": 27.26,
1067
+ "ppl": 3.86629,
1068
+ "step": 72,
1069
+ "tokens/total": 18677760,
1070
+ "tokens/train_per_sec_per_gpu": 606.39,
1071
+ "tokens/trainable": 18641452
1072
+ },
1073
+ {
1074
+ "epoch": 2.3591836734693876,
1075
+ "grad_norm": 0.78125,
1076
+ "learning_rate": 4.514935042870324e-06,
1077
+ "loss": 1.3382568359375,
1078
+ "memory/device_reserved (GiB)": 33.29,
1079
+ "memory/max_active (GiB)": 27.26,
1080
+ "memory/max_allocated (GiB)": 27.26,
1081
+ "ppl": 3.81239,
1082
+ "step": 73,
1083
+ "tokens/total": 18939904,
1084
+ "tokens/train_per_sec_per_gpu": 602.92,
1085
+ "tokens/trainable": 18903104
1086
+ },
1087
+ {
1088
+ "epoch": 2.3918367346938774,
1089
+ "grad_norm": 0.78125,
1090
+ "learning_rate": 4.40127743937224e-06,
1091
+ "loss": 1.396728515625,
1092
+ "memory/device_reserved (GiB)": 33.29,
1093
+ "memory/max_active (GiB)": 27.26,
1094
+ "memory/max_allocated (GiB)": 27.26,
1095
+ "ppl": 4.04196,
1096
+ "step": 74,
1097
+ "tokens/total": 19202048,
1098
+ "tokens/train_per_sec_per_gpu": 603.33,
1099
+ "tokens/trainable": 19164778
1100
+ },
1101
+ {
1102
+ "epoch": 2.424489795918367,
1103
+ "grad_norm": 0.71875,
1104
+ "learning_rate": 4.288360451123646e-06,
1105
+ "loss": 1.287841796875,
1106
+ "memory/device_reserved (GiB)": 33.29,
1107
+ "memory/max_active (GiB)": 27.26,
1108
+ "memory/max_allocated (GiB)": 27.26,
1109
+ "ppl": 3.62495,
1110
+ "step": 75,
1111
+ "tokens/total": 19464192,
1112
+ "tokens/train_per_sec_per_gpu": 603.18,
1113
+ "tokens/trainable": 19426440
1114
+ },
1115
+ {
1116
+ "epoch": 2.4571428571428573,
1117
+ "grad_norm": 0.81640625,
1118
+ "learning_rate": 4.1762601920102675e-06,
1119
+ "loss": 1.22216796875,
1120
+ "memory/device_reserved (GiB)": 33.29,
1121
+ "memory/max_active (GiB)": 27.26,
1122
+ "memory/max_allocated (GiB)": 27.26,
1123
+ "ppl": 3.39454,
1124
+ "step": 76,
1125
+ "tokens/total": 19726336,
1126
+ "tokens/train_per_sec_per_gpu": 606.31,
1127
+ "tokens/trainable": 19688104
1128
+ },
1129
+ {
1130
+ "epoch": 2.489795918367347,
1131
+ "grad_norm": 0.76171875,
1132
+ "learning_rate": 4.065052225385717e-06,
1133
+ "loss": 1.3858642578125,
1134
+ "memory/device_reserved (GiB)": 33.29,
1135
+ "memory/max_active (GiB)": 27.26,
1136
+ "memory/max_allocated (GiB)": 27.26,
1137
+ "ppl": 3.99828,
1138
+ "step": 77,
1139
+ "tokens/total": 19988480,
1140
+ "tokens/train_per_sec_per_gpu": 603.6,
1141
+ "tokens/trainable": 19949836
1142
+ },
1143
+ {
1144
+ "epoch": 2.522448979591837,
1145
+ "grad_norm": 0.74609375,
1146
+ "learning_rate": 3.954811513136554e-06,
1147
+ "loss": 1.32080078125,
1148
+ "memory/device_reserved (GiB)": 33.29,
1149
+ "memory/max_active (GiB)": 27.26,
1150
+ "memory/max_allocated (GiB)": 27.26,
1151
+ "ppl": 3.74642,
1152
+ "step": 78,
1153
+ "tokens/total": 20250624,
1154
+ "tokens/train_per_sec_per_gpu": 603.42,
1155
+ "tokens/trainable": 20211444
1156
+ },
1157
+ {
1158
+ "epoch": 2.5551020408163265,
1159
+ "grad_norm": 1.015625,
1160
+ "learning_rate": 3.84561236515276e-06,
1161
+ "loss": 1.3240966796875,
1162
+ "memory/device_reserved (GiB)": 33.29,
1163
+ "memory/max_active (GiB)": 27.26,
1164
+ "memory/max_allocated (GiB)": 27.26,
1165
+ "ppl": 3.75879,
1166
+ "step": 79,
1167
+ "tokens/total": 20512768,
1168
+ "tokens/train_per_sec_per_gpu": 603.23,
1169
+ "tokens/trainable": 20473284
1170
+ },
1171
+ {
1172
+ "epoch": 2.5877551020408163,
1173
+ "grad_norm": 0.703125,
1174
+ "learning_rate": 3.7375283892377344e-06,
1175
+ "loss": 1.385986328125,
1176
+ "memory/device_reserved (GiB)": 33.29,
1177
+ "memory/max_active (GiB)": 27.26,
1178
+ "memory/max_allocated (GiB)": 27.26,
1179
+ "ppl": 3.99877,
1180
+ "step": 80,
1181
+ "tokens/total": 20774912,
1182
+ "tokens/train_per_sec_per_gpu": 604.17,
1183
+ "tokens/trainable": 20734804
1184
+ },
1185
+ {
1186
+ "epoch": 2.620408163265306,
1187
+ "grad_norm": 0.7109375,
1188
+ "learning_rate": 3.630632441491512e-06,
1189
+ "loss": 1.130859375,
1190
+ "memory/device_reserved (GiB)": 33.29,
1191
+ "memory/max_active (GiB)": 27.26,
1192
+ "memory/max_allocated (GiB)": 27.26,
1193
+ "ppl": 3.09832,
1194
+ "step": 81,
1195
+ "tokens/total": 21037056,
1196
+ "tokens/train_per_sec_per_gpu": 606.12,
1197
+ "tokens/trainable": 20996280
1198
+ },
1199
+ {
1200
+ "epoch": 2.6530612244897958,
1201
+ "grad_norm": 0.85546875,
1202
+ "learning_rate": 3.5249965772007e-06,
1203
+ "loss": 1.3719482421875,
1204
+ "memory/device_reserved (GiB)": 33.29,
1205
+ "memory/max_active (GiB)": 27.26,
1206
+ "memory/max_allocated (GiB)": 27.26,
1207
+ "ppl": 3.94303,
1208
+ "step": 82,
1209
+ "tokens/total": 21299200,
1210
+ "tokens/train_per_sec_per_gpu": 603.84,
1211
+ "tokens/trainable": 21257992
1212
+ },
1213
+ {
1214
+ "epoch": 2.685714285714286,
1215
+ "grad_norm": 0.703125,
1216
+ "learning_rate": 3.4206920022682173e-06,
1217
+ "loss": 1.2578125,
1218
+ "memory/device_reserved (GiB)": 33.29,
1219
+ "memory/max_active (GiB)": 27.26,
1220
+ "memory/max_allocated (GiB)": 27.26,
1221
+ "ppl": 3.51772,
1222
+ "step": 83,
1223
+ "tokens/total": 21561344,
1224
+ "tokens/train_per_sec_per_gpu": 605.93,
1225
+ "tokens/trainable": 21519576
1226
+ },
1227
+ {
1228
+ "epoch": 2.7183673469387752,
1229
+ "grad_norm": 0.85546875,
1230
+ "learning_rate": 3.3177890252155755e-06,
1231
+ "loss": 1.3260498046875,
1232
+ "memory/device_reserved (GiB)": 33.29,
1233
+ "memory/max_active (GiB)": 27.26,
1234
+ "memory/max_allocated (GiB)": 27.26,
1235
+ "ppl": 3.76614,
1236
+ "step": 84,
1237
+ "tokens/total": 21823488,
1238
+ "tokens/train_per_sec_per_gpu": 604.99,
1239
+ "tokens/trainable": 21781056
1240
+ },
1241
+ {
1242
+ "epoch": 2.7510204081632654,
1243
+ "grad_norm": 0.8203125,
1244
+ "learning_rate": 3.2163570097900497e-06,
1245
+ "loss": 1.324951171875,
1246
+ "memory/device_reserved (GiB)": 33.29,
1247
+ "memory/max_active (GiB)": 27.26,
1248
+ "memory/max_allocated (GiB)": 27.26,
1249
+ "ppl": 3.762,
1250
+ "step": 85,
1251
+ "tokens/total": 22085632,
1252
+ "tokens/train_per_sec_per_gpu": 585.4,
1253
+ "tokens/trainable": 22042546
1254
+ },
1255
+ {
1256
+ "epoch": 2.783673469387755,
1257
+ "grad_norm": 0.73828125,
1258
+ "learning_rate": 3.116464328208708e-06,
1259
+ "loss": 1.3780517578125,
1260
+ "memory/device_reserved (GiB)": 33.29,
1261
+ "memory/max_active (GiB)": 27.26,
1262
+ "memory/max_allocated (GiB)": 27.26,
1263
+ "ppl": 3.96717,
1264
+ "step": 86,
1265
+ "tokens/total": 22347776,
1266
+ "tokens/train_per_sec_per_gpu": 586.69,
1267
+ "tokens/trainable": 22304072
1268
+ },
1269
+ {
1270
+ "epoch": 2.816326530612245,
1271
+ "grad_norm": 0.80859375,
1272
+ "learning_rate": 3.0181783150707827e-06,
1273
+ "loss": 1.23388671875,
1274
+ "memory/device_reserved (GiB)": 33.29,
1275
+ "memory/max_active (GiB)": 27.26,
1276
+ "memory/max_allocated (GiB)": 27.26,
1277
+ "ppl": 3.43455,
1278
+ "step": 87,
1279
+ "tokens/total": 22609920,
1280
+ "tokens/train_per_sec_per_gpu": 605.15,
1281
+ "tokens/trainable": 22565640
1282
+ },
1283
+ {
1284
+ "epoch": 2.8489795918367347,
1285
+ "grad_norm": 0.79296875,
1286
+ "learning_rate": 2.921565221969492e-06,
1287
+ "loss": 1.430908203125,
1288
+ "memory/device_reserved (GiB)": 33.29,
1289
+ "memory/max_active (GiB)": 27.26,
1290
+ "memory/max_allocated (GiB)": 27.26,
1291
+ "ppl": 4.1825,
1292
+ "step": 88,
1293
+ "tokens/total": 22872064,
1294
+ "tokens/train_per_sec_per_gpu": 603.72,
1295
+ "tokens/trainable": 22827386
1296
+ },
1297
+ {
1298
+ "epoch": 2.8816326530612244,
1299
+ "grad_norm": 0.74609375,
1300
+ "learning_rate": 2.8266901728338526e-06,
1301
+ "loss": 1.19873046875,
1302
+ "memory/device_reserved (GiB)": 33.29,
1303
+ "memory/max_active (GiB)": 27.26,
1304
+ "memory/max_allocated (GiB)": 27.26,
1305
+ "ppl": 3.3159,
1306
+ "step": 89,
1307
+ "tokens/total": 23134208,
1308
+ "tokens/train_per_sec_per_gpu": 606.04,
1309
+ "tokens/trainable": 23088952
1310
+ },
1311
+ {
1312
+ "epoch": 2.914285714285714,
1313
+ "grad_norm": 0.6953125,
1314
+ "learning_rate": 2.7336171200306467e-06,
1315
+ "loss": 1.1632080078125,
1316
+ "memory/device_reserved (GiB)": 33.29,
1317
+ "memory/max_active (GiB)": 27.26,
1318
+ "memory/max_allocated (GiB)": 27.26,
1319
+ "ppl": 3.20018,
1320
+ "step": 90,
1321
+ "tokens/total": 23396352,
1322
+ "tokens/train_per_sec_per_gpu": 604.92,
1323
+ "tokens/trainable": 23350548
1324
+ },
1325
+ {
1326
+ "epoch": 2.946938775510204,
1327
+ "grad_norm": 0.875,
1328
+ "learning_rate": 2.6424088012560766e-06,
1329
+ "loss": 1.2506103515625,
1330
+ "memory/device_reserved (GiB)": 33.29,
1331
+ "memory/max_active (GiB)": 27.26,
1332
+ "memory/max_allocated (GiB)": 27.26,
1333
+ "ppl": 3.49247,
1334
+ "step": 91,
1335
+ "tokens/total": 23658496,
1336
+ "tokens/train_per_sec_per_gpu": 605.58,
1337
+ "tokens/trainable": 23612092
1338
+ },
1339
+ {
1340
+ "epoch": 2.979591836734694,
1341
+ "grad_norm": 0.859375,
1342
+ "learning_rate": 2.5531266972462176e-06,
1343
+ "loss": 1.2801513671875,
1344
+ "memory/device_reserved (GiB)": 33.29,
1345
+ "memory/max_active (GiB)": 27.26,
1346
+ "memory/max_allocated (GiB)": 27.26,
1347
+ "ppl": 3.59718,
1348
+ "step": 92,
1349
+ "tokens/total": 23920640,
1350
+ "tokens/train_per_sec_per_gpu": 606.4,
1351
+ "tokens/trainable": 23873538
1352
+ },
1353
+ {
1354
+ "epoch": 3.0,
1355
+ "grad_norm": 0.94140625,
1356
+ "learning_rate": 2.4658309903347196e-06,
1357
+ "loss": 1.248046875,
1358
+ "memory/device_reserved (GiB)": 33.29,
1359
+ "memory/max_active (GiB)": 27.26,
1360
+ "memory/max_allocated (GiB)": 27.26,
1361
+ "ppl": 3.48353,
1362
+ "step": 93,
1363
+ "tokens/total": 24084480,
1364
+ "tokens/train_per_sec_per_gpu": 922.28,
1365
+ "tokens/trainable": 24036224
1366
+ },
1367
+ {
1368
+ "epoch": 3.0326530612244897,
1369
+ "grad_norm": 0.7109375,
1370
+ "learning_rate": 2.380580523885751e-06,
1371
+ "loss": 1.3673095703125,
1372
+ "memory/device_reserved (GiB)": 33.29,
1373
+ "memory/max_active (GiB)": 27.26,
1374
+ "memory/max_allocated (GiB)": 27.26,
1375
+ "ppl": 3.92478,
1376
+ "step": 94,
1377
+ "tokens/total": 24346624,
1378
+ "tokens/train_per_sec_per_gpu": 586.15,
1379
+ "tokens/trainable": 24298144
1380
+ },
1381
+ {
1382
+ "epoch": 3.0653061224489795,
1383
+ "grad_norm": 0.67578125,
1384
+ "learning_rate": 2.29743276262948e-06,
1385
+ "loss": 1.2625732421875,
1386
+ "memory/device_reserved (GiB)": 33.29,
1387
+ "memory/max_active (GiB)": 27.26,
1388
+ "memory/max_allocated (GiB)": 27.26,
1389
+ "ppl": 3.5345,
1390
+ "step": 95,
1391
+ "tokens/total": 24608768,
1392
+ "tokens/train_per_sec_per_gpu": 607.42,
1393
+ "tokens/trainable": 24559796
1394
+ },
1395
+ {
1396
+ "epoch": 3.0979591836734692,
1397
+ "grad_norm": 0.84375,
1398
+ "learning_rate": 2.2164437539268652e-06,
1399
+ "loss": 1.2720947265625,
1400
+ "memory/device_reserved (GiB)": 33.29,
1401
+ "memory/max_active (GiB)": 27.26,
1402
+ "memory/max_allocated (GiB)": 27.26,
1403
+ "ppl": 3.56832,
1404
+ "step": 96,
1405
+ "tokens/total": 24870912,
1406
+ "tokens/train_per_sec_per_gpu": 605.03,
1407
+ "tokens/trainable": 24821538
1408
+ },
1409
+ {
1410
+ "epoch": 3.130612244897959,
1411
+ "grad_norm": 0.69921875,
1412
+ "learning_rate": 2.1376680899898415e-06,
1413
+ "loss": 1.3033447265625,
1414
+ "memory/device_reserved (GiB)": 33.29,
1415
+ "memory/max_active (GiB)": 27.26,
1416
+ "memory/max_allocated (GiB)": 27.26,
1417
+ "ppl": 3.68159,
1418
+ "step": 97,
1419
+ "tokens/total": 25133056,
1420
+ "tokens/train_per_sec_per_gpu": 607.5,
1421
+ "tokens/trainable": 25083298
1422
+ },
1423
+ {
1424
+ "epoch": 3.163265306122449,
1425
+ "grad_norm": 0.73828125,
1426
+ "learning_rate": 2.0611588710823797e-06,
1427
+ "loss": 1.31024169921875,
1428
+ "memory/device_reserved (GiB)": 33.29,
1429
+ "memory/max_active (GiB)": 27.26,
1430
+ "memory/max_allocated (GiB)": 27.26,
1431
+ "ppl": 3.70707,
1432
+ "step": 98,
1433
+ "tokens/total": 25395200,
1434
+ "tokens/train_per_sec_per_gpu": 607.32,
1435
+ "tokens/trainable": 25344956
1436
+ },
1437
+ {
1438
+ "epoch": 3.195918367346939,
1439
+ "grad_norm": 0.796875,
1440
+ "learning_rate": 1.986967669727224e-06,
1441
+ "loss": 1.20263671875,
1442
+ "memory/device_reserved (GiB)": 33.29,
1443
+ "memory/max_active (GiB)": 27.26,
1444
+ "memory/max_allocated (GiB)": 27.26,
1445
+ "ppl": 3.32888,
1446
+ "step": 99,
1447
+ "tokens/total": 25657344,
1448
+ "tokens/train_per_sec_per_gpu": 604.43,
1449
+ "tokens/trainable": 25606580
1450
+ },
1451
+ {
1452
+ "epoch": 3.2285714285714286,
1453
+ "grad_norm": 0.69921875,
1454
+ "learning_rate": 1.9151444959424383e-06,
1455
+ "loss": 1.3046875,
1456
+ "memory/device_reserved (GiB)": 33.29,
1457
+ "memory/max_active (GiB)": 27.26,
1458
+ "memory/max_allocated (GiB)": 27.26,
1459
+ "ppl": 3.68654,
1460
+ "step": 100,
1461
+ "tokens/total": 25919488,
1462
+ "tokens/train_per_sec_per_gpu": 603.08,
1463
+ "tokens/trainable": 25868436
1464
+ },
1465
+ {
1466
+ "epoch": 3.2612244897959184,
1467
+ "grad_norm": 0.78515625,
1468
+ "learning_rate": 1.8457377635311763e-06,
1469
+ "loss": 1.431396484375,
1470
+ "memory/device_reserved (GiB)": 33.29,
1471
+ "memory/max_active (GiB)": 27.26,
1472
+ "memory/max_allocated (GiB)": 27.26,
1473
+ "ppl": 4.18454,
1474
+ "step": 101,
1475
+ "tokens/total": 26181632,
1476
+ "tokens/train_per_sec_per_gpu": 603.9,
1477
+ "tokens/trainable": 26130194
1478
+ },
1479
+ {
1480
+ "epoch": 3.293877551020408,
1481
+ "grad_norm": 1.1171875,
1482
+ "learning_rate": 1.7787942574474215e-06,
1483
+ "loss": 1.19580078125,
1484
+ "memory/device_reserved (GiB)": 33.29,
1485
+ "memory/max_active (GiB)": 27.26,
1486
+ "memory/max_allocated (GiB)": 27.26,
1487
+ "ppl": 3.3062,
1488
+ "step": 102,
1489
+ "tokens/total": 26443776,
1490
+ "tokens/train_per_sec_per_gpu": 606.74,
1491
+ "tokens/trainable": 26391908
1492
+ },
1493
+ {
1494
+ "epoch": 3.326530612244898,
1495
+ "grad_norm": 0.70703125,
1496
+ "learning_rate": 1.7143591022596846e-06,
1497
+ "loss": 1.34716796875,
1498
+ "memory/device_reserved (GiB)": 33.29,
1499
+ "memory/max_active (GiB)": 27.26,
1500
+ "memory/max_allocated (GiB)": 27.26,
1501
+ "ppl": 3.84652,
1502
+ "step": 103,
1503
+ "tokens/total": 26705920,
1504
+ "tokens/train_per_sec_per_gpu": 606.45,
1505
+ "tokens/trainable": 26653532
1506
+ },
1507
+ {
1508
+ "epoch": 3.3591836734693876,
1509
+ "grad_norm": 0.71484375,
1510
+ "learning_rate": 1.6524757317339102e-06,
1511
+ "loss": 1.3314208984375,
1512
+ "memory/device_reserved (GiB)": 33.29,
1513
+ "memory/max_active (GiB)": 27.26,
1514
+ "memory/max_allocated (GiB)": 27.26,
1515
+ "ppl": 3.78642,
1516
+ "step": 104,
1517
+ "tokens/total": 26968064,
1518
+ "tokens/train_per_sec_per_gpu": 602.49,
1519
+ "tokens/trainable": 26915184
1520
+ },
1521
+ {
1522
+ "epoch": 3.3918367346938774,
1523
+ "grad_norm": 2.484375,
1524
+ "learning_rate": 1.593185859556103e-06,
1525
+ "loss": 1.3916015625,
1526
+ "memory/device_reserved (GiB)": 33.29,
1527
+ "memory/max_active (GiB)": 27.26,
1528
+ "memory/max_allocated (GiB)": 27.26,
1529
+ "ppl": 4.02129,
1530
+ "step": 105,
1531
+ "tokens/total": 27230208,
1532
+ "tokens/train_per_sec_per_gpu": 603.38,
1533
+ "tokens/trainable": 27176858
1534
+ },
1535
+ {
1536
+ "epoch": 3.424489795918367,
1537
+ "grad_norm": 0.71875,
1538
+ "learning_rate": 1.5365294512144114e-06,
1539
+ "loss": 1.282958984375,
1540
+ "memory/device_reserved (GiB)": 33.29,
1541
+ "memory/max_active (GiB)": 27.26,
1542
+ "memory/max_allocated (GiB)": 27.26,
1543
+ "ppl": 3.6073,
1544
+ "step": 106,
1545
+ "tokens/total": 27492352,
1546
+ "tokens/train_per_sec_per_gpu": 604.12,
1547
+ "tokens/trainable": 27438520
1548
+ },
1549
+ {
1550
+ "epoch": 3.4571428571428573,
1551
+ "grad_norm": 0.6640625,
1552
+ "learning_rate": 1.4825446970596136e-06,
1553
+ "loss": 1.2174072265625,
1554
+ "memory/device_reserved (GiB)": 33.29,
1555
+ "memory/max_active (GiB)": 27.26,
1556
+ "memory/max_allocated (GiB)": 27.26,
1557
+ "ppl": 3.37842,
1558
+ "step": 107,
1559
+ "tokens/total": 27754496,
1560
+ "tokens/train_per_sec_per_gpu": 606.17,
1561
+ "tokens/trainable": 27700184
1562
+ },
1563
+ {
1564
+ "epoch": 3.489795918367347,
1565
+ "grad_norm": 0.73046875,
1566
+ "learning_rate": 1.4312679865621742e-06,
1567
+ "loss": 1.380615234375,
1568
+ "memory/device_reserved (GiB)": 33.29,
1569
+ "memory/max_active (GiB)": 27.26,
1570
+ "memory/max_allocated (GiB)": 27.26,
1571
+ "ppl": 3.97735,
1572
+ "step": 108,
1573
+ "tokens/total": 28016640,
1574
+ "tokens/train_per_sec_per_gpu": 602.64,
1575
+ "tokens/trainable": 27961916
1576
+ },
1577
+ {
1578
+ "epoch": 3.522448979591837,
1579
+ "grad_norm": 0.6953125,
1580
+ "learning_rate": 1.382733883783211e-06,
1581
+ "loss": 1.3165283203125,
1582
+ "memory/device_reserved (GiB)": 33.29,
1583
+ "memory/max_active (GiB)": 27.26,
1584
+ "memory/max_allocated (GiB)": 27.26,
1585
+ "ppl": 3.73045,
1586
+ "step": 109,
1587
+ "tokens/total": 28278784,
1588
+ "tokens/train_per_sec_per_gpu": 603.31,
1589
+ "tokens/trainable": 28223524
1590
+ },
1591
+ {
1592
+ "epoch": 3.5551020408163265,
1593
+ "grad_norm": 0.71484375,
1594
+ "learning_rate": 1.3369751040759236e-06,
1595
+ "loss": 1.3199462890625,
1596
+ "memory/device_reserved (GiB)": 33.29,
1597
+ "memory/max_active (GiB)": 27.26,
1598
+ "memory/max_allocated (GiB)": 27.26,
1599
+ "ppl": 3.74322,
1600
+ "step": 110,
1601
+ "tokens/total": 28540928,
1602
+ "tokens/train_per_sec_per_gpu": 602.96,
1603
+ "tokens/trainable": 28485364
1604
+ },
1605
+ {
1606
+ "epoch": 3.5877551020408163,
1607
+ "grad_norm": 0.6953125,
1608
+ "learning_rate": 1.2940224920331707e-06,
1609
+ "loss": 1.3828125,
1610
+ "memory/device_reserved (GiB)": 33.29,
1611
+ "memory/max_active (GiB)": 27.26,
1612
+ "memory/max_allocated (GiB)": 27.26,
1613
+ "ppl": 3.9861,
1614
+ "step": 111,
1615
+ "tokens/total": 28803072,
1616
+ "tokens/train_per_sec_per_gpu": 603.97,
1617
+ "tokens/trainable": 28746884
1618
+ },
1619
+ {
1620
+ "epoch": 3.620408163265306,
1621
+ "grad_norm": 0.6640625,
1622
+ "learning_rate": 1.2539050006960814e-06,
1623
+ "loss": 1.1270751953125,
1624
+ "memory/device_reserved (GiB)": 33.29,
1625
+ "memory/max_active (GiB)": 27.26,
1626
+ "memory/max_allocated (GiB)": 27.26,
1627
+ "ppl": 3.08662,
1628
+ "step": 112,
1629
+ "tokens/total": 29065216,
1630
+ "tokens/train_per_sec_per_gpu": 605.96,
1631
+ "tokens/trainable": 29008360
1632
+ },
1633
+ {
1634
+ "epoch": 3.6530612244897958,
1635
+ "grad_norm": 0.7421875,
1636
+ "learning_rate": 1.2166496720376874e-06,
1637
+ "loss": 1.368408203125,
1638
+ "memory/device_reserved (GiB)": 33.29,
1639
+ "memory/max_active (GiB)": 27.26,
1640
+ "memory/max_allocated (GiB)": 27.26,
1641
+ "ppl": 3.92909,
1642
+ "step": 113,
1643
+ "tokens/total": 29327360,
1644
+ "tokens/train_per_sec_per_gpu": 602.33,
1645
+ "tokens/trainable": 29270072
1646
+ },
1647
+ {
1648
+ "epoch": 3.685714285714286,
1649
+ "grad_norm": 0.67578125,
1650
+ "learning_rate": 1.1822816187347625e-06,
1651
+ "loss": 1.2537841796875,
1652
+ "memory/device_reserved (GiB)": 33.29,
1653
+ "memory/max_active (GiB)": 27.26,
1654
+ "memory/max_allocated (GiB)": 27.26,
1655
+ "ppl": 3.50358,
1656
+ "step": 114,
1657
+ "tokens/total": 29589504,
1658
+ "tokens/train_per_sec_per_gpu": 606.12,
1659
+ "tokens/trainable": 29531656
1660
+ },
1661
+ {
1662
+ "epoch": 3.7183673469387752,
1663
+ "grad_norm": 0.703125,
1664
+ "learning_rate": 1.1508240072401336e-06,
1665
+ "loss": 1.3232421875,
1666
+ "memory/device_reserved (GiB)": 33.29,
1667
+ "memory/max_active (GiB)": 27.26,
1668
+ "memory/max_allocated (GiB)": 27.26,
1669
+ "ppl": 3.75558,
1670
+ "step": 115,
1671
+ "tokens/total": 29851648,
1672
+ "tokens/train_per_sec_per_gpu": 605.25,
1673
+ "tokens/trainable": 29793136
1674
+ },
1675
+ {
1676
+ "epoch": 3.7510204081632654,
1677
+ "grad_norm": 0.70703125,
1678
+ "learning_rate": 1.1222980421668874e-06,
1679
+ "loss": 1.3226318359375,
1680
+ "memory/device_reserved (GiB)": 33.29,
1681
+ "memory/max_active (GiB)": 27.26,
1682
+ "memory/max_allocated (GiB)": 27.26,
1683
+ "ppl": 3.75329,
1684
+ "step": 116,
1685
+ "tokens/total": 30113792,
1686
+ "tokens/train_per_sec_per_gpu": 570.96,
1687
+ "tokens/trainable": 30054626
1688
+ },
1689
+ {
1690
+ "epoch": 3.783673469387755,
1691
+ "grad_norm": 1.328125,
1692
+ "learning_rate": 1.0967229519949833e-06,
1693
+ "loss": 1.3746337890625,
1694
+ "memory/device_reserved (GiB)": 33.29,
1695
+ "memory/max_active (GiB)": 27.26,
1696
+ "memory/max_allocated (GiB)": 27.26,
1697
+ "ppl": 3.95363,
1698
+ "step": 117,
1699
+ "tokens/total": 30375936,
1700
+ "tokens/train_per_sec_per_gpu": 602.33,
1701
+ "tokens/trainable": 30316152
1702
+ },
1703
+ {
1704
+ "epoch": 3.816326530612245,
1705
+ "grad_norm": 0.71875,
1706
+ "learning_rate": 1.0741159761099294e-06,
1707
+ "loss": 1.231201171875,
1708
+ "memory/device_reserved (GiB)": 33.29,
1709
+ "memory/max_active (GiB)": 27.26,
1710
+ "memory/max_allocated (GiB)": 27.26,
1711
+ "ppl": 3.42534,
1712
+ "step": 118,
1713
+ "tokens/total": 30638080,
1714
+ "tokens/train_per_sec_per_gpu": 605.12,
1715
+ "tokens/trainable": 30577720
1716
+ },
1717
+ {
1718
+ "epoch": 3.8489795918367347,
1719
+ "grad_norm": 0.80078125,
1720
+ "learning_rate": 1.054492353182237e-06,
1721
+ "loss": 1.4288330078125,
1722
+ "memory/device_reserved (GiB)": 33.29,
1723
+ "memory/max_active (GiB)": 27.26,
1724
+ "memory/max_allocated (GiB)": 27.26,
1725
+ "ppl": 4.17383,
1726
+ "step": 119,
1727
+ "tokens/total": 30900224,
1728
+ "tokens/train_per_sec_per_gpu": 603.79,
1729
+ "tokens/trainable": 30839466
1730
+ },
1731
+ {
1732
+ "epoch": 3.8816326530612244,
1733
+ "grad_norm": 0.76171875,
1734
+ "learning_rate": 1.0378653108955017e-06,
1735
+ "loss": 1.197021484375,
1736
+ "memory/device_reserved (GiB)": 33.29,
1737
+ "memory/max_active (GiB)": 27.26,
1738
+ "memory/max_allocated (GiB)": 27.26,
1739
+ "ppl": 3.31024,
1740
+ "step": 120,
1741
+ "tokens/total": 31162368,
1742
+ "tokens/train_per_sec_per_gpu": 606.03,
1743
+ "tokens/trainable": 31101032
1744
+ },
1745
+ {
1746
+ "epoch": 3.914285714285714,
1747
+ "grad_norm": 0.671875,
1748
+ "learning_rate": 1.0242460570300241e-06,
1749
+ "loss": 1.1612548828125,
1750
+ "memory/device_reserved (GiB)": 33.29,
1751
+ "memory/max_active (GiB)": 27.26,
1752
+ "memory/max_allocated (GiB)": 27.26,
1753
+ "ppl": 3.19394,
1754
+ "step": 121,
1755
+ "tokens/total": 31424512,
1756
+ "tokens/train_per_sec_per_gpu": 605.01,
1757
+ "tokens/trainable": 31362628
1758
+ },
1759
+ {
1760
+ "epoch": 3.946938775510204,
1761
+ "grad_norm": 0.6796875,
1762
+ "learning_rate": 1.01364377190799e-06,
1763
+ "loss": 1.2498779296875,
1764
+ "memory/device_reserved (GiB)": 33.29,
1765
+ "memory/max_active (GiB)": 27.26,
1766
+ "memory/max_allocated (GiB)": 27.26,
1767
+ "ppl": 3.48992,
1768
+ "step": 122,
1769
+ "tokens/total": 31686656,
1770
+ "tokens/train_per_sec_per_gpu": 606.07,
1771
+ "tokens/trainable": 31624172
1772
+ },
1773
+ {
1774
+ "epoch": 3.979591836734694,
1775
+ "grad_norm": 0.69921875,
1776
+ "learning_rate": 1.0060656022052966e-06,
1777
+ "loss": 1.2774658203125,
1778
+ "memory/device_reserved (GiB)": 33.29,
1779
+ "memory/max_active (GiB)": 27.26,
1780
+ "memory/max_allocated (GiB)": 27.26,
1781
+ "ppl": 3.58754,
1782
+ "step": 123,
1783
+ "tokens/total": 31948800,
1784
+ "tokens/train_per_sec_per_gpu": 605.62,
1785
+ "tokens/trainable": 31885618
1786
+ },
1787
+ {
1788
+ "epoch": 4.0,
1789
+ "grad_norm": 0.86328125,
1790
+ "learning_rate": 1.0015166561341943e-06,
1791
+ "loss": 1.245849609375,
1792
+ "memory/device_reserved (GiB)": 33.29,
1793
+ "memory/max_active (GiB)": 27.26,
1794
+ "memory/max_allocated (GiB)": 27.26,
1795
+ "ppl": 3.47589,
1796
+ "step": 124,
1797
+ "tokens/total": 32112640,
1798
+ "tokens/train_per_sec_per_gpu": 920.1,
1799
+ "tokens/trainable": 32048304
1800
+ }
1801
+ ],
1802
+ "loss_count": 124,
1803
+ "max_loss": 1.6685791015625,
1804
+ "max_steps": 124,
1805
+ "min_loss": 1.1270751953125
1806
+ },
1807
+ "status": "complete"
1808
+ }
1809
+ },
1810
+ "control_corpus": {
1811
+ "docs": 11387,
1812
+ "gate2_reference_digests": {
1813
+ "jsonl_sha256": "de2c2c62e12ab0714ca3d7149d18865d8287b603893c52d082844cc8ac5a57e0",
1814
+ "observed_ordered_rows_gate2_scheme": "5fc226629c0c253c179550aa362a56e89a4fa943d2747e1012264db87d50f1e6",
1815
+ "ordered_rows_sha256": "a852f50e44ec8814f74b15e0f9e0aebebb01a7141e11e2c9b027fe292164bb12"
1816
+ },
1817
+ "jsonl_sha256": "de2c2c62e12ab0714ca3d7149d18865d8287b603893c52d082844cc8ac5a57e0",
1818
+ "prefix_replay": {
1819
+ "docs": 6085,
1820
+ "ordered_rows_sha256": "819f35334706f6cd942ef3af31c927f3fcd986e3a3107372b461046d30ff02a9",
1821
+ "tokens": 4001953
1822
+ },
1823
+ "source_order_sha256": "abdc46436ddad18f6aff9fada41c2c01df7f7501dfa9eecbe03ce49356e1324b",
1824
+ "total_tokens": 8002382
1825
+ },
1826
+ "parameters": {
1827
+ "arms": [
1828
+ "control"
1829
+ ],
1830
+ "checkpoint_schedule": [
1831
+ 4,
1832
+ 31,
1833
+ 62,
1834
+ 93,
1835
+ 124
1836
+ ],
1837
+ "control_token_budget": 8000000,
1838
+ "data_seed": 42,
1839
+ "minimum_final_step": 124,
1840
+ "post_warmup_step": 4,
1841
+ "stage": "midtrain_dispatch_gemma3_4b_4epoch",
1842
+ "training_seed": 314159
1843
+ },
1844
+ "pins": {
1845
+ "checkpoint_repo": "sidbaines/scimt-dispatch-4b-models-v1",
1846
+ "filler": {
1847
+ "repo": "allenai/dolma3_dolmino_mix-100B-1125",
1848
+ "revision": "f23aa129fda8335ba9760057bcc1f0c02f3d068b"
1849
+ },
1850
+ "log_repo": "arcadia-impact/scimt-dispatch-4b-scaleup-v1",
1851
+ "model": {
1852
+ "repo": "unsloth/gemma-3-4b-pt",
1853
+ "revision": "52aba93981c6ad7712b030eb6dd496ece1d279d6"
1854
+ }
1855
+ },
1856
+ "remote_revisions": {
1857
+ "dataset": "5c6eb06eef3c89c9082c97e0c49db03b226fbd98",
1858
+ "filler": "f23aa129fda8335ba9760057bcc1f0c02f3d068b",
1859
+ "model": "52aba93981c6ad7712b030eb6dd496ece1d279d6"
1860
+ },
1861
+ "run_id": "20260815T010433Z-ctl2",
1862
+ "schema_version": "dispatch_scaleup_control_v1",
1863
+ "size": "4b",
1864
+ "source": {
1865
+ "branch": "sid/prior-coins-27b",
1866
+ "git_commit": "883445140956d34d47258cb021e56e0f4cbea05d",
1867
+ "git_tree": "938f63fc81ff415bacca97c8282deb44587085f1",
1868
+ "source_files": 1244,
1869
+ "source_files_sha256": "f2d6c55c0b02781a0a72f2372faa3514bf3e1443c245f3e0d365da2a112a158b"
1870
+ },
1871
+ "started_at": "2026-08-15T01:06:27+00:00",
1872
+ "status": "publishing"
1873
+ }