{ "release_version": "v0.1.0", "paper_release": "ASIL v0.1.0", "repo_id": "sharryXR/asil-qwen35-2b-rl", "model_name": "ASIL Qwen3.5-2B RL", "training_stage": "rl", "source_path": "/public/LLM_model_dataset/rl_temp/asil_operational_benchmark_a100_20260514_144617/results/rl/qwen35_2b_agentic_a800/qwen35_2b_rl_vllm_round3_final_small4_306039_20260519_114747/checkpoints/global_step_8_actor_hf", "selected_step": "global_step_8_actor_hf", "base_model": "/public/home/sjtu_normal/users/xierui/asil_sft_rl_a100_20260513_173133/results/sft_train/qwen35_2b_sft_v0_continue3_20260514_115443/checkpoints/global_step_27", "init_checkpoint": "/public/home/sjtu_normal/users/xierui/asil_sft_rl_a100_20260513_173133/results/sft_train/qwen35_2b_sft_v0_continue3_20260514_115443/checkpoints/global_step_27", "training_data_version": "ASIL v0.1.0 training data: sft_v0 + agentic-guided-v2 + merged_v0_v2 + rl_learnable_v4_320_80", "training_data": "rl_learnable_v4_320_80; 320 train / 80 valid task prompts", "upload_prepared_at": "2026-07-30T18:02:32+08:00", "checkpoint_sha256": "060e2705de25655336619c3e3e552890265f01d550f167ab4ff39c9e427b42c0", "sha256_algorithm": "sha256 over sorted SHA256SUMS lines of uploaded checkpoint files", "uploaded_file_count": 9, "uploaded_total_size_bytes": 4800832783, "files": [ { "path": "chat_template.jinja", "size_bytes": 7755, "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80" }, { "path": "config.json", "size_bytes": 1790, "sha256": "8e1bd1077e4dca1f722d15ac8a2738b211593e47707809128310720b0e48e661" }, { "path": "generation_config.json", "size_bytes": 115, "sha256": "0b725bef7b591339bc06389199c47d8e91bd7cef83c47bca3b747aefd3be489e" }, { "path": "model-00001-of-00003.safetensors", "size_bytes": 1017118848, "sha256": "85bf37d68d8e510fad9e32338b577b20e2ce1529e58831df6aeb0cea62b2f71c" }, { "path": "model-00002-of-00003.safetensors", "size_bytes": 1999935064, "sha256": "db4ec3515ce2a54908ae2cefe19e0febb30aed50e8c7611b9f4797c1f8adda88" }, { "path": "model-00003-of-00003.safetensors", "size_bytes": 1763751976, "sha256": "1da12e436ffe92acbf1bbfb8b8f480cb213ae06aba4a6d3671bcd67f7548d99f" }, { "path": "model.safetensors.index.json", "size_bytes": 26787, "sha256": "4c7ae43697d70ef300f62e1bb8ecfc07965634cd6c0257db957bd7dbab1e2353" }, { "path": "tokenizer.json", "size_bytes": 19989325, "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" }, { "path": "tokenizer_config.json", "size_bytes": 1123, "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87" } ], "excluded_source_files": [], "notes": "Selected early 2B RL checkpoint. RL initialization is 2B SFT global_step_27, not the public final 2B SFT global_step_111." }