yuyangalin's picture
Upload config.yaml with huggingface_hub
75c135f verified
Raw
History Blame Contribute Delete
5.97 kB
output_dir: ./runs/libero_flux2_klein_4b_base_fastwam/2026-05-27_04-13-04
batch_size: 10
num_workers: 24
prefetch_factor: 2
persistent_workers: true
lr_scheduler_type: cosine
learning_rate: 0.0001
warmup_steps: null
num_epochs: 10
max_steps: null
log_every: 10
save_every: 1000
keep_latest_state_only: true
eval_every: 100
eval_num_inference_steps: 10
eval_num_samples: 8
rank_timer_every: 10
rank_timer_sync_cuda: true
qwen_cache_batch_size: 32
qwen_cache_overwrite: false
qwen_cache_save_workers: 8
flux2_qwen3_model_spec: null
precache_arrow: true
precache_num_workers: 4
build_lerobot_meta_cache: true
build_norm_stats: true
precache_fuse_norm_stats: true
force_rebuild_norm_stats: false
norm_stats_use_arrow_cache: true
norm_stats_bulk_read_arrow: true
norm_stats_cache_enabled: true
norm_stats_cache_dir: null
precache_profile: false
precache_profile_min_sec: 0.0
gradient_accumulation_steps: 1
mixed_precision: bf16
seed: 42
max_grad_norm: 1.0
weight_decay: 0.01
resume: null
wandb:
enabled: true
workspace: arisilin
project: fast-wam
name: libero_flux2_klein_4b_base_fastwam
group: null
mode: online
data:
train:
_target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset
dataset_dirs:
- /dockerdata/data/libero/libero_spatial_no_noops_lerobot
- /dockerdata/data/libero/libero_object_no_noops_lerobot
- /dockerdata/data/libero/libero_goal_no_noops_lerobot
- /dockerdata/data/libero/libero_10_no_noops_lerobot
shape_meta:
images:
- key: image
raw_shape:
- 3
- 512
- 512
shape:
- 3
- 224
- 224
- key: wrist_image
raw_shape:
- 3
- 512
- 512
shape:
- 3
- 224
- 224
action:
- key: default
raw_shape: 7
shape: 7
state:
- key: default
raw_shape: 8
shape: 8
num_frames: 17
global_sample_stride: 1
action_video_freq_ratio: 1
video_size:
- 224
- 448
camera_key: null
val_set_proportion: 0.0
is_training_set: true
skip_padding_as_possible: false
concat_multi_camera: horizontal
video_augmentation:
_target_: fastwam.datasets.lerobot.transforms.image.VideoAugmentation
p: 0.65
augment_types:
- corrupt_only
- color_only
- both
- both
color_jitter:
brightness: 0.3
contrast: 0.3
saturation: 0.25
hue: 0.04
gamma:
range:
- 0.8
- 1.25
gaussian_noise:
std: 0.015
random_resized_crop:
scale:
- 0.92
- 1.0
ratio: preserve
rotate:
degrees: 8.0
fill: mean
exposure:
ev_range:
- -0.25
- 0.25
require_text_cache: false
endpoint_frames_only: true
qwen_text_cache_dir: /dockerdata/data/libero/flux2_qwen3_cache_4b
qwen_context_len: 512
pretrained_norm_stats: null
processor:
_target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor
shape_meta:
images:
- key: image
raw_shape:
- 3
- 512
- 512
shape:
- 3
- 224
- 224
- key: wrist_image
raw_shape:
- 3
- 512
- 512
shape:
- 3
- 224
- 224
action:
- key: default
raw_shape: 7
shape: 7
state:
- key: default
raw_shape: 8
shape: 8
num_obs_steps: 17
image_obs_steps: 2
num_output_cameras: 2
action_output_dim: 7
proprio_output_dim: 8
delta_action_dim_mask:
default:
- true
- true
- true
- true
- true
- true
- false
action_state_transforms: null
use_stepwise_action_norm: false
norm_default_mode: min/max
norm_exception_mode: null
action_state_merger:
_target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign
train_transforms:
- _target_: fastwam.datasets.lerobot.transforms.image.ToTensor
- _target_: torchvision.transforms.Resize
size:
- 224
- 224
val_transforms:
- _target_: fastwam.datasets.lerobot.transforms.image.ToTensor
- _target_: torchvision.transforms.Resize
size:
- 224
- 224
text_embedding_cache_dir: null
context_len: 128
qwen_text_cache_format: qwen3_flux2
model:
_target_: fastwam.runtime.create_fastwam_flux2_klein
flux2_src_path: /apdcephfs_nj7/share_305204761/alixzhang/flux2
flux2_model_path: /dockerdata/models/FLUX.2-klein-base-4B/FLUX.2-klein-base-4B/flux-2-klein-base-4b.safetensors
ae_model_path: /dockerdata/models/FLUX.2-dev/FLUX.2-dev/ae.safetensors
variant: klein-base-4b
qwen3_model_spec: Qwen/Qwen3-4B
load_text_encoder: false
proprio_dim: 8
mot_checkpoint_mixed_attn: true
mot_gqa_implementation: repeat
mot_force_flash_attention: false
pack_proprio_after_text: true
action_dit_pretrained_path: checkpoints/action_dit_flux2_4b_libero_init.pt
flux2_lora_config:
enabled: false
rank: 16
alpha: 16.0
dropout: 0.0
save_lora_merged: false
save_trainable_only: false
target_suffixes:
- qkv
- proj
- linear1
- linear2
- img_mlp.0
- img_mlp.2
- txt_mlp.0
- txt_mlp.2
action_dit_config:
action_dim: 7
hidden_dim: 1024
num_heads: 24
attn_head_dim: 128
num_layers_double: 5
num_layers_single: 20
mlp_ratio: 4.0
max_action_horizon: 64
use_gradient_checkpointing: true
video_scheduler:
train_shift: 5.0
infer_shift: 5.0
num_train_timesteps: 1000
action_scheduler:
train_shift: 5.0
infer_shift: 5.0
num_train_timesteps: 1000
loss:
lambda_video: 0.5
lambda_action: 1.0