bankai-v1 / training_config.yaml
tgetsov's picture
Rename BankAI training configuration
6cbf6d8 verified
Raw
History Blame Contribute Delete
748 Bytes
# QLoRA for Qwen3-Coder-Next on Apple Silicon with 128 GB unified memory.
model: mlx-community/Qwen3-Coder-Next-4bit
train: true
data: artifacts/bankai-v1/data
fine_tune_type: lora
optimizer: adamw
mask_prompt: true
num_layers: 16
batch_size: 1
grad_accumulation_steps: 2
iters: 612
val_batches: 8
learning_rate: 1.0e-5
steps_per_report: 5
steps_per_eval: 100
save_every: 100
adapter_path: artifacts/bankai-v1/adapter
max_seq_length: 2048
grad_checkpoint: true
clear_cache_threshold: 68719476736
seed: 42
lora_parameters:
rank: 8
dropout: 0.0
scale: 16.0
keys:
- linear_attn.in_proj_qkvz
- linear_attn.in_proj_ba
- linear_attn.out_proj
- self_attn.q_proj
- self_attn.k_proj
- self_attn.v_proj
- self_attn.o_proj