{ "run_slug": "opsd-llama31-8b-instruct-origin-gen1024-step200-jsdclip006-20260515", "method": "OPSD origin fixed-teacher full-vocabulary JSD with per-token clipping", "base_model": "meta-llama/Llama-3.1-8B-Instruct", "model_family": "llama", "dataset": "siyanzhao/Openthoughts_math_30k_opsd", "max_steps": 200, "max_completion_length": 1024, "max_length": 20000, "student_thinking": false, "teacher_thinking": true, "fixed_teacher": true, "loss": "full-vocabulary forward KL/JSD beta=0", "beta": 0.0, "lmbda": 1.0, "jsd_token_clip": 0.06, "temperature": 1.1, "top_p": 0.95, "top_k": 20, "per_device_train_batch_size": 1, "gradient_accumulation_steps": 2, "effective_train_batch_size": 8, "learning_rate": 5e-06, "max_grad_norm": 0.1, "vllm_mode": "colocate", "vllm_gpu_memory_utilization": 0.35, "vllm_tensor_parallel_size": 1, "lora_r": 64, "lora_alpha": 128, "lora_target_modules": [ "q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj" ], "num_gpus": 4, "uploaded_artifacts": "LoRA adapter checkpoints, tokenizer files, trainer state, generations, logs, README, metadata" }