Robotics
Safetensors
LeRobot
cosmos3
surgical-robotics
dvrk
surgical-kinematics
motion-prediction
inverse-dynamics
forward-dynamics
action-conditioned
world-model
Instructions to use introvoyz041/cosmos3-nanoh-surgical-dvrk with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- LeRobot
How to use introvoyz041/cosmos3-nanoh-surgical-dvrk with LeRobot:
- Notebooks
- Google Colab
- Kaggle
| checkpoint: | |
| broadcast_via_filesystem: false | |
| dcp_async_mode_enabled: false | |
| enable_gcs_patch_in_boto3: true | |
| hf_export: | |
| enabled: false | |
| export_every_n: 1 | |
| hf_repo_id: null | |
| upload_to_object_store: | |
| bucket: '' | |
| credentials: '' | |
| enabled: false | |
| jit: | |
| device: cuda | |
| dtype: bfloat16 | |
| enabled: false | |
| input_shape: null | |
| strict: true | |
| keys_not_to_resume: [] | |
| keys_to_skip_loading: | |
| - net_ema. | |
| load_ema_to_reg: false | |
| load_from_object_store: | |
| bucket: '' | |
| credentials: '' | |
| enabled: false | |
| load_path: /workspace/cosmos/checkpoints/Cosmos3-Nano-DCP | |
| load_training_state: false | |
| only_load_scheduler_state: false | |
| save_iter: 100 | |
| save_to_object_store: | |
| bucket: '' | |
| credentials: '' | |
| enabled: false | |
| strict_resume: true | |
| type: | |
| _target_: cosmos_framework.checkpoint.dcp.DistributedCheckpointer | |
| callbacks: null | |
| disable_async: false | |
| verbose: true | |
| data_setting: | |
| qwen_max_video_token_length: 8192 | |
| dataloader_train: | |
| _target_: cosmos_framework.data.vfm.joint_dataloader.PackingDataLoader | |
| audio_sample_rate: 48000 | |
| dataloader: | |
| _target_: cosmos_framework.data.vfm.joint_dataloader.RankPartitionedDataLoader | |
| batch_size: 1 | |
| datasets: | |
| dvrk_openh: | |
| dataset: | |
| _target_: cosmos3_nanoh_surgical.dataset.OpenHDVRKLeRobotDataset | |
| apply_cosmos_transform: true | |
| cfg_dropout_rate: 0.1 | |
| chunk_length: 16 | |
| fps: null | |
| load_video: true | |
| local_root: /workspace/cosmos-3-NanoH-Surgical/data/openh_snapshot | |
| manifest_path: /workspace/cosmos-3-NanoH-Surgical/artifacts/manifests/openh_dvrk_manifest.train.json | |
| max_action_dim: 64 | |
| max_roots: 0 | |
| max_windows_per_root: 0 | |
| mode: all | |
| mode_weights: | |
| forward_dynamics: 1.0 | |
| inverse_dynamics: 1.0 | |
| policy: 1.0 | |
| normalization_path: /workspace/cosmos-3-NanoH-Surgical/artifacts/manifests/dvrk_action_stats.json | |
| normalize_action: true | |
| resolution: '256' | |
| shard_rank: 0 | |
| shard_world_size: 1 | |
| tokenizer_config: | |
| _target_: cosmos_framework.configs.base.defaults.vlm.create_qwen2_tokenizer_with_download | |
| config_variant: hf | |
| pretrained_model_name: Qwen/Qwen3-VL-8B-Instruct | |
| video_key: auto | |
| ratio: 1 | |
| in_order: true | |
| num_workers: 4 | |
| persistent_workers: true | |
| pin_memory: true | |
| prefetch_factor: 4 | |
| sampler: null | |
| dataset_name: dvrk_openh | |
| lookahead_limit: 10 | |
| max_samples_per_batch: null | |
| max_sequence_length: 32768 | |
| patch_spatial: 2 | |
| sound_latent_fps: 0 | |
| tokenizer_spatial_compression_factor: 16 | |
| tokenizer_temporal_compression_factor: 4 | |
| dataloader_val: null | |
| defaults: | |
| - _self_ | |
| - model: mot_fsdp | |
| - data_train: null | |
| - data_val: null | |
| - optimizer: adamw | |
| - scheduler: warmup_cosine_lr | |
| - checkpoint: s3 | |
| - callbacks: | |
| - basic | |
| - optimization | |
| - job_monitor | |
| - generation | |
| - ema: power | |
| - tokenizer: wan2pt2_tokenizer | |
| - sound_tokenizer: null | |
| - cluster: default | |
| - vlm_config: null | |
| - ckpt_type: dcp | |
| - experiment: null | |
| job: | |
| cluster: null | |
| group: dvrk-openh | |
| name: cosmos3_nanoh_dvrk_all_objective_chunk16_singlecam | |
| project: cosmos3-nanoh-surgical | |
| wandb_mode: online | |
| model: | |
| _recursive_: false | |
| _target_: cosmos_framework.model.vfm.omni_mot_model.OmniMoTModel | |
| config: | |
| action_gen: true | |
| activation_checkpointing: | |
| determinism_check: default | |
| mode: full | |
| preserve_rng_state: true | |
| save_ops_regex: | |
| - fmha | |
| causal_training_strategy: none | |
| compile: | |
| compile_dynamic: true | |
| compiled_region: language | |
| coordinate_descent_tuning: false | |
| enabled: false | |
| max_autotune_pointwise: false | |
| use_cuda_graphs: false | |
| diffusion_expert_config: | |
| base_fps: 24 | |
| enable_fps_modulation: true | |
| load_weights_from_pretrained: true | |
| max_vae_latent_side_after_patchify: 20 | |
| patch_spatial: 2 | |
| position_embedding_type: unified_3d_mrope | |
| rope_h_extrapolation_ratio: 1.0 | |
| rope_t_extrapolation_ratio: 1.0 | |
| rope_w_extrapolation_ratio: 1.0 | |
| timestep_range: 1.0 | |
| unified_3d_mrope_reset_spatial_ids: true | |
| unified_3d_mrope_temporal_modality_margin: 15000 | |
| ema: | |
| enabled: false | |
| iteration_shift: 0 | |
| rate: 0.1 | |
| fixed_step_sampler_config: null | |
| input_caption_key: ai_caption | |
| input_image_key: images | |
| input_video_key: video | |
| joint_attn_implementation: flex | |
| latent_downsample_factor: 16 | |
| lbl: | |
| coeff_gen: null | |
| coeff_und: null | |
| method: local | |
| log_enc_time_every_n: 100 | |
| lora_alpha: 32 | |
| lora_enabled: false | |
| lora_rank: 16 | |
| lora_target_modules: q_proj_moe_gen,k_proj_moe_gen,v_proj_moe_gen,o_proj_moe_gen | |
| max_action_dim: 64 | |
| max_num_tokens_after_packing: 32768 | |
| natten_parameter_list: null | |
| net: null | |
| num_embodiment_domains: 32 | |
| parallelism: | |
| cfg_parallel_shard_degree: 1 | |
| context_parallel_shard_degree: 1 | |
| data_parallel_replicate_degree: 1 | |
| data_parallel_shard_degree: -1 | |
| enable_inference_mode: false | |
| fsdp_master_dtype: float32 | |
| precision: bfloat16 | |
| rectified_flow_inference_config: | |
| num_train_timesteps: 1000 | |
| scheduler_type: unipc | |
| shift: 1 | |
| use_dynamic_shifting: false | |
| rectified_flow_training_config: | |
| action_loss_weight: 10.0 | |
| high_sigma_ratio: 0.05 | |
| high_sigma_timesteps_max: 1000 | |
| high_sigma_timesteps_min: 995 | |
| image_loss_scale: 1.0 | |
| independent_action_schedule: false | |
| independent_sound_schedule: false | |
| loss_scale: 1.0 | |
| normalize_loss_by_active: false | |
| shift: | |
| '256': 3 | |
| '480': 5 | |
| '720': 10 | |
| shift_action: null | |
| shift_sound: null | |
| sound_loss_scale: null | |
| train_time_action_distribution: logitnormal | |
| train_time_image_distribution: logitnormal | |
| train_time_sound_distribution: logitnormal | |
| train_time_video_distribution: waver | |
| train_time_weight: uniform | |
| use_discrete_rf: false | |
| use_dynamic_shift: false | |
| use_high_sigma_strategy: false | |
| use_high_sigma_strategy_action: false | |
| use_high_sigma_strategy_sound: false | |
| resolution: '256' | |
| sound_dim: null | |
| sound_gen: false | |
| sound_latent_fps: 25 | |
| sound_tokenizer: null | |
| state_ch: 48 | |
| state_t: 300 | |
| tokenizer: | |
| _target_: cosmos_framework.model.vfm.tokenizers.wan2pt2_vae_4x16x16.Wan2pt2VAEInterface | |
| bucket_name: '' | |
| chunk_duration: 93 | |
| encode_bucket_multiple: null | |
| encode_chunk_frames: | |
| '256': 68 | |
| '480': 24 | |
| '720': 12 | |
| encode_exact_durations: null | |
| keep_decoder_cache: false | |
| object_store_credential_path_pretrained: '' | |
| spatial_compression_factor: 16 | |
| temporal_compression_factor: 4 | |
| temporal_window: null | |
| use_streaming_encode: false | |
| vae_path: /workspace/cosmos/checkpoints/wan22_vae/Wan2.2_VAE.pth | |
| video_temporal_causal: false | |
| vision_gen: true | |
| vlm_config: | |
| layer_module: Qwen2MoTDecoderLayer | |
| model_instance: | |
| _target_: cosmos_framework.model.vfm.mot.unified_mot.Qwen3VLTextForCausalLM | |
| config: | |
| _target_: cosmos_framework.configs.base.defaults.vlm.create_vlm_config | |
| base_config: | |
| _target_: cosmos_framework.model.vfm.mot.unified_mot.Qwen3VLMoTConfig.from_json_file | |
| json_file: cosmos_framework/model/vfm/vlm/qwen3_vl/configs/Qwen3-VL-8B-Instruct.json | |
| freeze_und: false | |
| layer_module: MoTDecoderLayer | |
| qk_norm_for_text: true | |
| tie_word_embeddings: true | |
| model_name: Qwen/Qwen3-VL-8B-Instruct | |
| pretrained_weights: | |
| backbone_path: s3://bucket0/cosmos3/pretrained/huggingface/Qwen/Qwen3-VL-8B-Instruct/ | |
| checkpoint_format: null | |
| credentials_path: '' | |
| enable_gcs_patch_in_boto3: true | |
| enabled: false | |
| qk_norm: false | |
| safetensors_path: '' | |
| tie_word_embeddings: false | |
| tokenizer: | |
| _target_: cosmos_framework.configs.base.defaults.vlm.create_qwen2_tokenizer_with_download | |
| config_variant: hf | |
| pretrained_model_name: Qwen/Qwen3-VL-8B-Instruct | |
| use_system_prompt: false | |
| model_parallel: | |
| _cpu_offloading_context: null | |
| async_tensor_model_parallel_allreduce: false | |
| autocast_dtype: torch.float32 | |
| barrier_with_L1_time: true | |
| batch_p2p_comm: true | |
| batch_p2p_sync: true | |
| bf16: false | |
| context_parallel_size: 1 | |
| cpu_offloading: false | |
| cpu_offloading_activations: true | |
| cpu_offloading_double_buffering: false | |
| cpu_offloading_num_layers: 0 | |
| cpu_offloading_weights: false | |
| cross_entropy_fusion_impl: native | |
| cross_entropy_loss_fusion: false | |
| deallocate_pipeline_outputs: false | |
| defer_embedding_wgrad_compute: false | |
| delay_wgrad_compute: false | |
| deterministic_mode: false | |
| enable_autocast: false | |
| ep_overlap_early_attn_memory_release: false | |
| expert_model_parallel_size: 1 | |
| expert_tensor_parallel_size: 1 | |
| finalize_model_grads_func: null | |
| fp16: false | |
| grad_scale_func: null | |
| grad_sync_func: null | |
| gradient_accumulation_fusion: false | |
| hierarchical_context_parallel_sizes: null | |
| microbatch_group_size_per_vp_stage: 1 | |
| moe_extended_tp: false | |
| no_sync_func: null | |
| num_microbatches_with_partial_activation_checkpoints: null | |
| overlap_moe_expert_parallel_comm: false | |
| overlap_p2p_comm: false | |
| overlap_p2p_comm_warmup_flush: false | |
| param_sync_func: null | |
| params_dtype: torch.float32 | |
| perform_initialization: true | |
| pipeline_dtype: null | |
| pipeline_model_parallel_comm_backend: null | |
| pipeline_model_parallel_size: 1 | |
| sequence_parallel: false | |
| tensor_model_parallel_size: 1 | |
| timers: null | |
| tp_comm_atomic_ag: false | |
| tp_comm_atomic_rs: false | |
| tp_comm_bootstrap_backend: nccl | |
| tp_comm_bulk_dgrad: true | |
| tp_comm_bulk_wgrad: true | |
| tp_comm_overlap: false | |
| tp_comm_overlap_ag: true | |
| tp_comm_overlap_disable_fc1: false | |
| tp_comm_overlap_disable_qkv: false | |
| tp_comm_overlap_rs: true | |
| tp_comm_overlap_rs_dgrad: false | |
| tp_comm_split_ag: true | |
| tp_comm_split_rs: true | |
| use_cpu_initialization: false | |
| use_ring_exchange_p2p: false | |
| use_te_rng_tracker: false | |
| variable_seq_lengths: false | |
| virtual_pipeline_model_parallel_size: null | |
| wgrad_deferral_limit: 0 | |
| optimizer: | |
| _target_: cosmos_framework.utils.vfm.optimizer.build_optimizer | |
| betas: | |
| - 0.9 | |
| - 0.95 | |
| disable_weight_decay_for_1d_params: false | |
| eps: 1.0e-06 | |
| fused: true | |
| keys_to_select: | |
| - moe_gen | |
| - time_embedder | |
| - vae2llm | |
| - llm2vae | |
| - action2llm | |
| - llm2action | |
| - action_modality_embed | |
| - q_proj_moe_gen | |
| - k_proj_moe_gen | |
| - v_proj_moe_gen | |
| - o_proj_moe_gen | |
| lr: 2.0e-05 | |
| lr_multipliers: {} | |
| model: null | |
| optimizer_type: AdamW | |
| weight_decay: 0 | |
| scheduler: | |
| _target_: cosmos_framework.utils.vfm.optimizer.build_lr_scheduler | |
| cycle_lengths: | |
| - 1000 | |
| f_max: | |
| - 1.0 | |
| f_min: | |
| - 0.0 | |
| f_start: | |
| - 0.0 | |
| lr_scheduler_type: LambdaCosine | |
| optimizer: null | |
| verbosity_interval: 0 | |
| warm_up_steps: | |
| - 50 | |
| trainer: | |
| callbacks: | |
| compile_tokenizer: | |
| _target_: cosmos_framework.callbacks.compile_tokenizer.CompileTokenizer | |
| compile_after_iterations: 3 | |
| enabled: false | |
| warmup_resolutions: null | |
| dataloader_speed: | |
| every_n: 100 | |
| save_s3: false | |
| step_size: 1 | |
| device_monitor: | |
| _target_: cosmos_framework.callbacks.device_monitor.DeviceMonitor | |
| every_n: 100 | |
| log_memory_detail: true | |
| save_s3: false | |
| step_size: 1 | |
| upload_every_n_mul: 5 | |
| every_n_sample_ema: | |
| _target_: cosmos_framework.callbacks.every_n_draw_sample.EveryNDrawSample | |
| do_x0_prediction: false | |
| every_n: 100 | |
| fps: 16 | |
| guidance: | |
| - 0.0 | |
| - 3.0 | |
| - 7.0 | |
| is_ema: true | |
| n_sample_to_save: 128 | |
| n_sigmas_for_x0_prediction: 4 | |
| n_viz_sample: 2 | |
| num_sampling_step: 35 | |
| prompt_type: t5_xxl | |
| run_at_start: false | |
| save_local: false | |
| save_s3: false | |
| step_size: 1 | |
| use_negative_prompt: false | |
| every_n_sample_reg: | |
| _target_: cosmos_framework.callbacks.every_n_draw_sample.EveryNDrawSample | |
| do_x0_prediction: false | |
| every_n: 100 | |
| fps: 16 | |
| guidance: | |
| - 3.0 | |
| is_ema: false | |
| n_sample_to_save: 8 | |
| n_sigmas_for_x0_prediction: 4 | |
| n_viz_sample: 1 | |
| num_sampling_step: 16 | |
| prompt_type: t5_xxl | |
| run_at_start: false | |
| save_local: true | |
| save_s3: false | |
| step_size: 1 | |
| use_negative_prompt: false | |
| expert_heatmap: | |
| _target_: cosmos_framework.callbacks.expert_heatmap.ExpertHeatmap | |
| every_n: 1000 | |
| grad_clip: | |
| _target_: cosmos_framework.callbacks.grad_clip.GradClip | |
| clip_norm: 0.1 | |
| force_finite: true | |
| track_per_modality: true | |
| heart_beat: | |
| _target_: cosmos_framework.callbacks.heart_beat.HeartBeat | |
| every_n: 100 | |
| save_s3: false | |
| step_size: 1 | |
| update_interval_in_minute: 10 | |
| iter_speed: | |
| _target_: cosmos_framework.callbacks.iter_speed.IterSpeed | |
| every_n: 1 | |
| hit_thres: 50 | |
| save_s3: false | |
| save_s3_every_log_n: 500 | |
| load_pretrained: | |
| _target_: cosmos_framework.callbacks.load_pretrained.LoadPretrained | |
| config: null | |
| trainer: null | |
| low_precision: | |
| _target_: cosmos_framework.utils.callback.LowPrecisionCallback | |
| config: null | |
| trainer: null | |
| update_iter: 1 | |
| manual_gc: | |
| _target_: cosmos_framework.callbacks.manual_gc.ManualGarbageCollection | |
| every_n: 5 | |
| gc_level: 1 | |
| warm_up: 1 | |
| mfu: | |
| _target_: cosmos_framework.callbacks.mfu.MFUCallback | |
| backwardpass_ratio: 2.0 | |
| every_n: 1 | |
| grad_accum_iter: 2 | |
| hit_thres: 5 | |
| include_padding: true | |
| include_vae_encoder: true | |
| moe_specialization: | |
| _target_: cosmos_framework.callbacks.moe_specialization_callback.MoESpecializationCallback | |
| every_n: 250 | |
| moe_stability: | |
| _target_: cosmos_framework.callbacks.moe_stability_callback.MoEStabilityCallback | |
| every_n: 250 | |
| norm_monitor: | |
| _target_: cosmos_framework.callbacks.norm_monitor.NormMonitor | |
| every_n: 100 | |
| layer_norm_only: false | |
| log_stat_wandb: false | |
| model_key: null | |
| save_s3: false | |
| step_size: 1 | |
| track_activations: false | |
| ofu: | |
| _target_: cosmos_framework.callbacks.ofu.OFUCallback | |
| every_n: 1 | |
| hit_thres: 5 | |
| param_count: | |
| _target_: cosmos_framework.callbacks.param_count.ParamCount | |
| save_s3: false | |
| sequence_packing_padding: | |
| _target_: cosmos_framework.callbacks.sequence_packing_padding.SequencePackingPadding | |
| every_n: 50 | |
| sigma_loss_analysis: | |
| _target_: cosmos_framework.callbacks.sigma_loss_analysis.SigmaLossAnalysis | |
| every_n: 250 | |
| every_n_viz: 250 | |
| save_s3: false | |
| skip_nan_step: | |
| _target_: cosmos_framework.callbacks.skip_nan_step.SkipNaNStep | |
| max_consecutive_nan: 20 | |
| termination_signal_checkpoint: | |
| _target_: cosmos_framework.callbacks.termination_signal_checkpoint.TerminationSignalCheckpoint | |
| min_save_fraction: 0.3333333333333333 | |
| training_stats: | |
| log_freq: 50 | |
| wandb: | |
| _target_: cosmos_framework.utils.callback.WandBCallback | |
| config: null | |
| trainer: null | |
| wandb_2x: | |
| _target_: cosmos_framework.callbacks.wandb_log.WandbCallback | |
| logging_iter_multipler: 2 | |
| save_logging_iter_multipler: 1 | |
| save_s3: false | |
| wandb_val: | |
| _target_: cosmos_framework.callbacks.wandb_log_eval.WandbCallback | |
| save_s3: false | |
| compile_config: | |
| recompile_limit: 8 | |
| use_duck_shape: false | |
| cudnn: | |
| benchmark: true | |
| deterministic: false | |
| ddp: | |
| broadcast_buffers: true | |
| find_unused_parameters: false | |
| static_graph: true | |
| distributed_parallelism: fsdp | |
| grad_accum_iter: 2 | |
| grad_scaler_args: | |
| enabled: false | |
| logging_iter: 1 | |
| max_iter: 500 | |
| max_val_iter: null | |
| memory_format: torch.preserve_format | |
| profiling: | |
| enable_memory_snapshot: false | |
| enable_nsys: false | |
| enable_profiling: false | |
| profile_freq: 1 | |
| profile_memory: false | |
| profile_warmup: 3 | |
| record_shape: false | |
| save_s3: false | |
| target_ranks: | |
| - 0 | |
| - 1 | |
| - 2 | |
| - 3 | |
| - 4 | |
| - 5 | |
| - 6 | |
| - 7 | |
| with_modules: true | |
| with_stack: true | |
| run_validation: false | |
| run_validation_on_start: false | |
| save_zero_checkpoint: false | |
| seed: 42 | |
| straggler_detection: | |
| analyze_backward: true | |
| analyze_dataloading: true | |
| analyze_forward: true | |
| analyze_optimizer: true | |
| enabled: false | |
| max_diff: 2.0 | |
| profile_freq: 1 | |
| raise_error: true | |
| report_freq: 100 | |
| save_s3: false | |
| timeout_period: 999999999 | |
| type: cosmos_framework.trainer.ImaginaireTrainer | |
| validation_iter: 100 | |
| upload_reproducible_setup: false | |