File size: 2,479 Bytes
eb29c04
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
# Llama-3.3-70B Loracle DrGRPO post-training (paper-canonical v11 recipe).
#
# Builds on llamacle_v6_clean_step1875 (1-epoch pretrain on 22.5k Llama LoRAs,
# val_loss=1.70, AB Llama-70B baseline 55.2%). RL pool = 2,500 orgs held out
# from v6 pretrain at /workspace/data/llamacle_pretrain_v7_oneq/holdout_ids.json.
#
# Hypers from appendix Table tab:rl-hparams. Differences from Qwen v9:
#   - base_model: Llama-70B (vs Qwen3-14B)
#   - n_direction_tokens=8960 (vs 4480) — 80 layers × 7 mag7 × 16 ranks
#   - max_length=9500 (vs 5500) — matches pretrain
#   - use_nf4: true — Llama-70B can't fit per-rank in bf16 on B200; NF4-quant
#     base + bf16 LoRA + Adam fp32 master fits ~85GB/rank with KV cache
#   - n_cycles=80, lr=7e-6, eps_high=0.28 (per appendix; v9 had 40/5e-6/0.28)
#   - paper-canonical 9 prompts (built into qa.parquet via build_llamacle_rl_qa.py)

run_name: drgrpo_llama70b_v1
wandb_project: lora-oracles-posttrain
output_dir: /workspace/checkpoints/drgrpo_llama70b_v1
seed: 53

sft_checkpoint_repo: ceselder/llamacle_v6_clean_step1875
base_model: /root/models/Llama-3.3-70B-Instruct

prompts_parquet: /workspace/data/llamacle_rl_v1/qa.parquet
holdout_ids_path: /workspace/data/llamacle_rl_v1/rl_holdout_ids.json
tokens_dir: /workspace/data/llamacle_pretrain_v3_r16/direction_tokens_svd_fixed_k16_mag7_rankfirst

# NF4-DDP: each rank holds the full base in 4-bit (~35GB) on its own GPU.
# Avoids FSDP per-token gather overhead so rollout-gen runs at ~6× DP throughput.
# Backward all-reduces only LoRA grads (~6.6GB fp32) per cycle.
use_nf4: true

n_prompts_per_cycle: 32
k_rollouts: 16
temperature: 0.75
max_new_tokens: 250
inject_demonstrations: 0

algorithm: drgrpo
n_cycles: 80
lr: 7.0e-6
eps_low: 0.2
eps_high: 0.28
max_grad_norm: 1.0
max_length: 9500

filter_min_max: 0.0
filter_min_std: 0.0
unbiased_advantages: true

use_system_prompt: false
prefix_mode: rank_tagged
top_k: 16
n_direction_tokens: 8960

judge_mode: ranking
judge_prompt_mode: keyword_pretrain
judge_provider: anthropic
rollout_judge_model: claude-opus-4-7
judge_model: anthropic/claude-sonnet-4.6
judge_workers: 32
judge_max_retries: 4
judge_request_timeout_s: 300

log_every: 1
failure_score_threshold: 4

eval_at_step_0: false  # in-train eval OOMs from KV-cache memory creep across orgs even with split; eval offline after
eval_every_cycles: 99999
mid_train_eval_sets: []

post_eval: false
eval_sets: []
save_every: 5  # save ckpts more frequently for offline eval coverage