stage_1: backbone_type: vggt_omega ckpt_path: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/3DA_unified/.local/deepspeed_ckpt_aliases/hf23000_8gxiadnc_ckpt34000_checkpoints/0034000.pt encoder_input_size: 256 normalization_stat_path: null vggt_omega: source_path: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/3DA_unified/.local/vendor/vggt-omega freeze_patch_embedder: true enable_depth: true enable_camera: true enable_alignment: true freeze_dense_head: true dense_head_compute_dtype: bfloat16 strict_load: false teacher_ckpt_path: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/checkpoints/vggt_omega/vggt_omega_1b_256_text.pt da3_finetune: enabled: true freeze_blocks_before: 0 n_action_steps: 5 n_views: 2 use_temporal_embed: false action_head: type: mlp_resnet input_dim: auto hidden_dim: 1024 n_dims: 7 chunk_size: 8 num_blocks: 2 pool_mode: mean regularization: lambda_feat: 0.0 layer_weight_min: 0.5 adaptive_lambda: false lambda_depth: 3.0 depth_loss_type: vggt_omega_no_conf depth_grad_weight: 1.0 depth_decode_chunk_size: 1 vggt_omega_inverse_depth_eps: 0.001 vggt_omega_inverse_depth_max: 100.0 vggt_omega_weight_normalize: true lambda_camera: 0.0 teacher_depth_fallback: true skip_depth_if_no_gt: true lambda_path_b_deep_feat_reg: 0.0 deep_feat_reg_layer_weight_min: 0.5 proprioception: enabled: true proprio_dim: 7 hidden_dim: 256 predictor: enabled: true type: shallow12_ar d_model: 1024 depth: 12 num_heads: 16 ffn_ratio: 4.0 dropout: 0.0 num_patches_per_view: 256 use_language: true language_encoder_type: t5 language_dim: 768 language_len: 77 clip_model: openai/clip-vit-large-patch14 t5_model: google-t5/t5-base condition_mode: concat input_proj_norm: ln cache_token_embeddings: true language_cache_max_entries: 4096 language_cache_device: cpu lambda_feat_future: 1.0 lambda_feat_current: 0.0 lambda_proprio_future: 0.0 lambda_sigreg: 0.0 feature_loss_type: l1 feature_target_mode: future H_choices: - 4 H_weights: - 1.0 prev_action_mask_rate: 0.3 prev_action_mask_include_t0: true num_register_tokens: 16 deep_gradient_checkpointing: false deep_temporal_causal_mask: true use_proprio_head: false training: global_batch_size: 224 micro_batch_size: 28 base_lr: 5.16e-05 head_lr_mult: 10.0 predictor_lr_mult: 10.0 adam_eps: 1.0e-06 adam_beta1: 0.9 adam_beta2: 0.95 weight_decay: 0.0 lambda_action: 3.0 lambda_action_direct: 0.0 lambda_action_refine: 1.0 grad_accum_steps: 1 epochs: 500000 clip_grad: 1.0 warmup_steps: 0 max_steps: 1000000 min_lr_ratio: 1.0 log_every: 10 ckpt_every: 1000 vis_every: 1000 eval_every: 1000 lazy_eval_dataset: true eval_ae: false eval_noact: false global_seed: 42 num_workers: 8 prefetch_factor: 4 persistent_workers: true bf16: true compile: false distributed_timeout_minutes: 120 preemption_check_every_steps: 0 eval_micro_batch_size: 1 dataset: type: mixer seed: 42 epoch_size: null stats_dir: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/data/oxe_mimicgen_robocasa365_base_delta_h1to4/_stats eval_max_episodes: 16 image_size: - 256 - 256 future_steps: 4 chunk_size: 8 include_current_action: true n_views: 2 proprio_dim: 7 norm_mode: q01_q99 action_norm_mode: q01_q99 proprio_norm_mode: q01_q99 uniform_action_sampling: true image_augmentation: profile: cosmos_policy_strong train: random_resized_crop_area: 0.9 base_only_rotation_degrees: 5.0 color_jitter: brightness: 0.3 contrast: 0.4 saturation: 0.5 hue: 0.05 jpeg: enabled: true quality: 95 eval: center_crop_area: 1.0 jpeg: enabled: false quality: 95 action_stats_samples: -1 proprio_stats_samples: -1 require_stats_timing_signature: true sources: - name: oxe_v3_full_rdt_adapted_base_delta weight: 0.6 config: type: openx openx_root: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/data/openx_lerobot preset: v3_full_rdt_adapted sampling_strategy: dataset_weighted mix_epoch_size: 200000 target_hz: null aggregate_chunk_actions: false normalize_proprio: true download_videos: true video_backend: pyav train_crop_min_scale: 0.9 eval_crop_scale: 0.9 eval_ratio: 0.001 force_cache_sync: false lazy_download: true require_language: true action_frame: base_delta - name: mimicgen_core_base_delta weight: 0.15 config: type: mimicgen mimicgen_root: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/data/mimicgen task_descriptions_path: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/data/mimicgen/task_descriptions.json splits: - core tasks: null aggregate_chunk_actions: false train_crop_min_scale: 0.9 eval_crop_scale: 0.9 eval_ratio: 0.001 proprio_canonical_7d: true action_output_scale: - 0.05 - 0.05 - 0.05 - 0.5 - 0.5 - 0.5 action_frame: base_delta gt_depth_root: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/gt_depth/mimicgen_aligned gt_depth_key: depth_meters gt_depth_scale_mode: pointmap gt_depth_require_geometry: true gt_depth_require_sidecar_file: false gt_depth_rotate180: false gt_depth_vflip: false gt_depth_min_meters: 0.001 - name: robocasa365_base_static_depth_base_delta weight: 0.25 config: type: robocasa robocasa_root: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/data/robocasa split: pretrain source: human spec_name: robocasa_pretrain_human action_frame: base_delta enable_arrow_cache: true defer_dataset_build: true eval_ratio: 0.001 gt_depth_index_path: /NHNHOME/WORKSPACE/0226010404_A/CVLAB/CVLAB1/jisang/gt_depth/robocasa365_base_static_env_depth_256_2view_target100_exact_index/index.memmap.json gt_depth_scale_mode: median_depth gt_depth_require_sidecar_file: true gt_depth_min_meters: 0.001 gt_depth_rotate180: false gt_depth_hflip: false gt_depth_vflip: false use_dit: false