| output_dir: ./runs/libero_flux2_klein_4b_base_fastwam/2026-05-27_04-13-04 |
| batch_size: 10 |
| num_workers: 24 |
| prefetch_factor: 2 |
| persistent_workers: true |
| lr_scheduler_type: cosine |
| learning_rate: 0.0001 |
| warmup_steps: null |
| num_epochs: 10 |
| max_steps: null |
| log_every: 10 |
| save_every: 1000 |
| keep_latest_state_only: true |
| eval_every: 100 |
| eval_num_inference_steps: 10 |
| eval_num_samples: 8 |
| rank_timer_every: 10 |
| rank_timer_sync_cuda: true |
| qwen_cache_batch_size: 32 |
| qwen_cache_overwrite: false |
| qwen_cache_save_workers: 8 |
| flux2_qwen3_model_spec: null |
| precache_arrow: true |
| precache_num_workers: 4 |
| build_lerobot_meta_cache: true |
| build_norm_stats: true |
| precache_fuse_norm_stats: true |
| force_rebuild_norm_stats: false |
| norm_stats_use_arrow_cache: true |
| norm_stats_bulk_read_arrow: true |
| norm_stats_cache_enabled: true |
| norm_stats_cache_dir: null |
| precache_profile: false |
| precache_profile_min_sec: 0.0 |
| gradient_accumulation_steps: 1 |
| mixed_precision: bf16 |
| seed: 42 |
| max_grad_norm: 1.0 |
| weight_decay: 0.01 |
| resume: null |
| wandb: |
| enabled: true |
| workspace: arisilin |
| project: fast-wam |
| name: libero_flux2_klein_4b_base_fastwam |
| group: null |
| mode: online |
| data: |
| train: |
| _target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset |
| dataset_dirs: |
| - /dockerdata/data/libero/libero_spatial_no_noops_lerobot |
| - /dockerdata/data/libero/libero_object_no_noops_lerobot |
| - /dockerdata/data/libero/libero_goal_no_noops_lerobot |
| - /dockerdata/data/libero/libero_10_no_noops_lerobot |
| shape_meta: |
| images: |
| - key: image |
| raw_shape: |
| - 3 |
| - 512 |
| - 512 |
| shape: |
| - 3 |
| - 224 |
| - 224 |
| - key: wrist_image |
| raw_shape: |
| - 3 |
| - 512 |
| - 512 |
| shape: |
| - 3 |
| - 224 |
| - 224 |
| action: |
| - key: default |
| raw_shape: 7 |
| shape: 7 |
| state: |
| - key: default |
| raw_shape: 8 |
| shape: 8 |
| num_frames: 17 |
| global_sample_stride: 1 |
| action_video_freq_ratio: 1 |
| video_size: |
| - 224 |
| - 448 |
| camera_key: null |
| val_set_proportion: 0.0 |
| is_training_set: true |
| skip_padding_as_possible: false |
| concat_multi_camera: horizontal |
| video_augmentation: |
| _target_: fastwam.datasets.lerobot.transforms.image.VideoAugmentation |
| p: 0.65 |
| augment_types: |
| - corrupt_only |
| - color_only |
| - both |
| - both |
| color_jitter: |
| brightness: 0.3 |
| contrast: 0.3 |
| saturation: 0.25 |
| hue: 0.04 |
| gamma: |
| range: |
| - 0.8 |
| - 1.25 |
| gaussian_noise: |
| std: 0.015 |
| random_resized_crop: |
| scale: |
| - 0.92 |
| - 1.0 |
| ratio: preserve |
| rotate: |
| degrees: 8.0 |
| fill: mean |
| exposure: |
| ev_range: |
| - -0.25 |
| - 0.25 |
| require_text_cache: false |
| endpoint_frames_only: true |
| qwen_text_cache_dir: /dockerdata/data/libero/flux2_qwen3_cache_4b |
| qwen_context_len: 512 |
| pretrained_norm_stats: null |
| processor: |
| _target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor |
| shape_meta: |
| images: |
| - key: image |
| raw_shape: |
| - 3 |
| - 512 |
| - 512 |
| shape: |
| - 3 |
| - 224 |
| - 224 |
| - key: wrist_image |
| raw_shape: |
| - 3 |
| - 512 |
| - 512 |
| shape: |
| - 3 |
| - 224 |
| - 224 |
| action: |
| - key: default |
| raw_shape: 7 |
| shape: 7 |
| state: |
| - key: default |
| raw_shape: 8 |
| shape: 8 |
| num_obs_steps: 17 |
| image_obs_steps: 2 |
| num_output_cameras: 2 |
| action_output_dim: 7 |
| proprio_output_dim: 8 |
| delta_action_dim_mask: |
| default: |
| - true |
| - true |
| - true |
| - true |
| - true |
| - true |
| - false |
| action_state_transforms: null |
| use_stepwise_action_norm: false |
| norm_default_mode: min/max |
| norm_exception_mode: null |
| action_state_merger: |
| _target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign |
| train_transforms: |
| - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor |
| - _target_: torchvision.transforms.Resize |
| size: |
| - 224 |
| - 224 |
| val_transforms: |
| - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor |
| - _target_: torchvision.transforms.Resize |
| size: |
| - 224 |
| - 224 |
| text_embedding_cache_dir: null |
| context_len: 128 |
| qwen_text_cache_format: qwen3_flux2 |
| model: |
| _target_: fastwam.runtime.create_fastwam_flux2_klein |
| flux2_src_path: /apdcephfs_nj7/share_305204761/alixzhang/flux2 |
| flux2_model_path: /dockerdata/models/FLUX.2-klein-base-4B/FLUX.2-klein-base-4B/flux-2-klein-base-4b.safetensors |
| ae_model_path: /dockerdata/models/FLUX.2-dev/FLUX.2-dev/ae.safetensors |
| variant: klein-base-4b |
| qwen3_model_spec: Qwen/Qwen3-4B |
| load_text_encoder: false |
| proprio_dim: 8 |
| mot_checkpoint_mixed_attn: true |
| mot_gqa_implementation: repeat |
| mot_force_flash_attention: false |
| pack_proprio_after_text: true |
| action_dit_pretrained_path: checkpoints/action_dit_flux2_4b_libero_init.pt |
| flux2_lora_config: |
| enabled: false |
| rank: 16 |
| alpha: 16.0 |
| dropout: 0.0 |
| save_lora_merged: false |
| save_trainable_only: false |
| target_suffixes: |
| - qkv |
| - proj |
| - linear1 |
| - linear2 |
| - img_mlp.0 |
| - img_mlp.2 |
| - txt_mlp.0 |
| - txt_mlp.2 |
| action_dit_config: |
| action_dim: 7 |
| hidden_dim: 1024 |
| num_heads: 24 |
| attn_head_dim: 128 |
| num_layers_double: 5 |
| num_layers_single: 20 |
| mlp_ratio: 4.0 |
| max_action_horizon: 64 |
| use_gradient_checkpointing: true |
| video_scheduler: |
| train_shift: 5.0 |
| infer_shift: 5.0 |
| num_train_timesteps: 1000 |
| action_scheduler: |
| train_shift: 5.0 |
| infer_shift: 5.0 |
| num_train_timesteps: 1000 |
| loss: |
| lambda_video: 0.5 |
| lambda_action: 1.0 |
|
|