Jinhuiye's picture
Add files using upload-large-folder tool
c8173fb verified
Raw
History Blame Contribute Delete
1.88 kB
datasets:
vla_data:
data_mix: libero_all
data_root_dir: /home/jye624/Datasets/LIBERO
dataset_py: lerobot_datasets
per_device_batch_size: 8
sequential_step_sampling: false
video_backend: torchvision_av
framework:
name: CosmoPredict2GR00T
action_model:
action_dim: 7
action_horizon: 8
action_model_type: DiT-B
add_pos_embed: true
diffusion_model_cfg:
cross_attention_dim: 2048
dropout: 0.2
final_dropout: true
interleave_self_attention: true
norm_type: ada_norm
num_layers: 16
output_dim: 1024
positional_embeddings: null
future_action_window_size: 7
hidden_size: 1024
max_seq_len: 1024
noise_beta_alpha: 1.5
noise_beta_beta: 1.0
noise_s: 0.999
num_inference_timesteps: 4
num_target_vision_tokens: 32
num_timestep_buckets: 1000
past_action_window_size: 0
repeated_diffusion_steps: 8
state_dim: 7
obs_image_size: null
qwenvl:
base_vlm: /home/jye624/Models/Pretrained_models/Qwen3-VL-4B-Instruct
world_model:
base_wm: ./playground/Pretrained_models/nvidia/Cosmos-Predict2-2B-Video2World
extract_layers:
- -1
output_dir: ./results/Checkpoints/0405_libero4in1_CosmoPredict2GR00T
run_id: 0405_libero4in1_CosmoPredict2GR00T
run_root_dir: ./results/Checkpoints
seed: 42
trainer:
eval_interval: 100
freeze_modules: true
gradient_accumulation_steps: 1
gradient_clipping: 1.0
is_resume: false
learning_rate:
action_model: 0.0001
base: 2.5e-05
qwen_vl_interface: 1.0e-05
logging_frequency: 100
lr_scheduler_type: cosine_with_min_lr
max_train_steps: 80000
num_warmup_steps: 5000
optimizer:
betas:
- 0.9
- 0.95
eps: 1.0e-08
weight_decay: 1.0e-08
save_interval: 10000
scheduler_specific_kwargs:
min_lr: 1.0e-06
wandb_entity: jinhuiye
wandb_project: starVLA_Libero