m1ku2 commited on
Commit
861deca
·
verified ·
1 Parent(s): c1df75b

upload 12k inference checkpoint

Browse files
12k/checkpoint.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "checkpoint_cache_dir": null,
3
+ "checkpoint_hf": null,
4
+ "checkpoint_path": "/project/peilab/wam/cosmos3_cy/reports/cosmos3_piper14/cosmos3_action/battery_piper14/battery_piper14_cosmos3_nano_20000step_4gpu_b8_acc1_offline/checkpoints/iter_000012000/model",
5
+ "checkpoint_type": "dcp",
6
+ "config_file": "/project/peilab/wam/cosmos3_cy/reports/cosmos3_piper14/cosmos3_action/battery_piper14/battery_piper14_cosmos3_nano_20000step_4gpu_b8_acc1_offline/config.yaml",
7
+ "config_file_type": "yaml",
8
+ "credential_path": "credentials/gcp_checkpoint.secret",
9
+ "experiment": "",
10
+ "experiment_overrides": [
11
+ "model.config.diffusion_expert_config.load_weights_from_pretrained=False",
12
+ "model.config.vlm_config.pretrained_weights.enabled=False",
13
+ "checkpoint.load_from_object_store.enabled=False"
14
+ ],
15
+ "model_memory_bytes": null,
16
+ "use_ema_weights": true,
17
+ "vlm_processor_from_checkpoint": false
18
+ }
12k/config.json ADDED
@@ -0,0 +1,183 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Cosmos3OmniModel"
4
+ ],
5
+ "auto_map": {
6
+ "AutoConfig": "cosmos3.model.Cosmos3OmniConfig",
7
+ "AutoModel": "cosmos3.model.Cosmos3OmniModel"
8
+ },
9
+ "dtype": "bfloat16",
10
+ "model": {
11
+ "_recursive_": false,
12
+ "_target": "omni_mot_model",
13
+ "config": {
14
+ "action_gen": true,
15
+ "activation_checkpointing": {
16
+ "determinism_check": "default",
17
+ "mode": "full",
18
+ "preserve_rng_state": true,
19
+ "save_ops_regex": [
20
+ "fmha"
21
+ ]
22
+ },
23
+ "causal_training_strategy": "none",
24
+ "compile": {
25
+ "compile_dynamic": true,
26
+ "compiled_region": "language",
27
+ "coordinate_descent_tuning": false,
28
+ "enabled": true,
29
+ "max_autotune_pointwise": false,
30
+ "use_cuda_graphs": false
31
+ },
32
+ "diffusion_expert_config": {
33
+ "base_fps": 24,
34
+ "enable_fps_modulation": true,
35
+ "load_weights_from_pretrained": true,
36
+ "max_vae_latent_side_after_patchify": 20,
37
+ "patch_spatial": 2,
38
+ "position_embedding_type": "unified_3d_mrope",
39
+ "rope_h_extrapolation_ratio": 1.0,
40
+ "rope_t_extrapolation_ratio": 1.0,
41
+ "rope_w_extrapolation_ratio": 1.0,
42
+ "sound_base_temporal_compression_factor": null,
43
+ "timestep_range": 1.0,
44
+ "unified_3d_mrope_reset_spatial_ids": true,
45
+ "unified_3d_mrope_temporal_modality_margin": 15000,
46
+ "vision_temporal_position_mode": "latent_index"
47
+ },
48
+ "ema": {
49
+ "enabled": false,
50
+ "iteration_shift": 0,
51
+ "rate": 0.1
52
+ },
53
+ "exclude_reasoner_weights_from_checkpoint": false,
54
+ "fixed_step_sampler_config": null,
55
+ "input_caption_key": "ai_caption",
56
+ "input_image_key": "images",
57
+ "input_video_key": "video",
58
+ "joint_attn_implementation": "two_way",
59
+ "latent_downsample_factor": 16,
60
+ "lbl": {
61
+ "coeff_gen": null,
62
+ "coeff_und": null,
63
+ "method": "local"
64
+ },
65
+ "log_enc_time_every_n": 100,
66
+ "lora_alpha": 32,
67
+ "lora_enabled": false,
68
+ "lora_rank": 16,
69
+ "lora_target_modules": "q_proj_moe_gen,k_proj_moe_gen,v_proj_moe_gen,o_proj_moe_gen",
70
+ "max_action_dim": 64,
71
+ "max_num_tokens_after_packing": -1,
72
+ "natten_parameter_list": null,
73
+ "net": null,
74
+ "num_embodiment_domains": 32,
75
+ "parallelism": {
76
+ "attention_io_layout": "sequence_sharded",
77
+ "cfg_parallel_shard_degree": 1,
78
+ "context_parallel_shard_degree": 1,
79
+ "data_parallel_replicate_degree": 1,
80
+ "data_parallel_shard_degree": 4,
81
+ "enable_inference_mode": false,
82
+ "fsdp_master_dtype": "float32"
83
+ },
84
+ "precision": "bfloat16",
85
+ "rectified_flow_inference_config": {
86
+ "num_train_timesteps": 1000,
87
+ "scheduler_type": "unipc",
88
+ "shift": 1,
89
+ "use_dynamic_shifting": false
90
+ },
91
+ "rectified_flow_training_config": {
92
+ "action_loss_weight": 10.0,
93
+ "image_loss_scale": 1.0,
94
+ "independent_action_schedule": false,
95
+ "independent_sound_schedule": false,
96
+ "loss_scale": 10.0,
97
+ "normalize_loss_by_active": false,
98
+ "shift": {
99
+ "256": 3,
100
+ "480": 5,
101
+ "720": 10
102
+ },
103
+ "shift_action": null,
104
+ "shift_sound": null,
105
+ "sound_loss_scale": null,
106
+ "train_time_action_distribution": "logitnormal",
107
+ "train_time_image_distribution": "logitnormal",
108
+ "train_time_sound_distribution": "logitnormal",
109
+ "train_time_video_distribution": "waver",
110
+ "train_time_weight": "uniform",
111
+ "use_discrete_rf": false,
112
+ "use_dynamic_shift": false
113
+ },
114
+ "resolution": "720",
115
+ "sound_dim": null,
116
+ "sound_gen": false,
117
+ "sound_latent_fps": 25,
118
+ "sound_tokenizer": null,
119
+ "state_ch": 48,
120
+ "state_t": 300,
121
+ "tokenizer": {
122
+ "_target": "wan2pt2_vae_interface",
123
+ "bucket_name": "",
124
+ "causal": true,
125
+ "chunk_duration": 93,
126
+ "encode_bucket_multiple": null,
127
+ "encode_chunk_frames": {
128
+ "256": 68,
129
+ "480": 24,
130
+ "720": 12
131
+ },
132
+ "encode_exact_durations": [
133
+ 33
134
+ ],
135
+ "keep_decoder_cache": false,
136
+ "object_store_credential_path_pretrained": "",
137
+ "spatial_compression_factor": 16,
138
+ "temporal_compression_factor": 4,
139
+ "temporal_window": null,
140
+ "use_streaming_encode": false,
141
+ "vae_path": "/project/peilab/wam/cosmos3_cy/external/cosmos/checkpoints/hf_home/hub/models--Wan-AI--Wan2.2-TI2V-5B/snapshots/921dbaf3f1674a56f47e83fb80a34bac8a8f203e/Wan2.2_VAE.pth"
142
+ },
143
+ "video_temporal_causal": false,
144
+ "vision_gen": true,
145
+ "vlm_config": {
146
+ "layer_module": "Qwen2MoTDecoderLayer",
147
+ "model_instance": {
148
+ "_target": "qwen3_vl_text_for_causal_lm",
149
+ "config": {
150
+ "_target": "create_vlm_config",
151
+ "base_config": {
152
+ "_target": "qwen3_vl_mot_config_from_json_file",
153
+ "json_file": "cosmos3://vfm/models/vlm/qwen3_vl/configs/Qwen3-VL-8B-Instruct.json"
154
+ },
155
+ "freeze_und": false,
156
+ "layer_module": "MoTDecoderLayer",
157
+ "qk_norm_for_text": true,
158
+ "tie_word_embeddings": true
159
+ }
160
+ },
161
+ "model_name": "Qwen/Qwen3-VL-8B-Instruct",
162
+ "pretrained_weights": {
163
+ "backbone_path": "s3://bucket0/cosmos3/pretrained/huggingface/Qwen/Qwen3-VL-8B-Instruct/",
164
+ "checkpoint_format": null,
165
+ "credentials_path": "",
166
+ "enable_gcs_patch_in_boto3": true,
167
+ "enabled": false
168
+ },
169
+ "qk_norm": false,
170
+ "safetensors_path": "",
171
+ "tie_word_embeddings": false,
172
+ "tokenizer": {
173
+ "_target": "create_qwen2_tokenizer_with_download",
174
+ "config_variant": "hf",
175
+ "pretrained_model_name": "Qwen/Qwen3-VL-8B-Instruct"
176
+ },
177
+ "use_system_prompt": false
178
+ }
179
+ }
180
+ },
181
+ "model_type": "cosmos3_omni",
182
+ "transformers_version": "4.57.6"
183
+ }
12k/model-00001-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:531c6f6c30b55db8147719539c2c8e4da6e7ea897e45108a042de4bbb4c55728
3
+ size 4994798328
12k/model-00002-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f1e8b02dac94f7b105f125c0968dc52c68fa42c47135d45c106b9e1c231c66a5
3
+ size 4983055752
12k/model-00003-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5c9b8810ee35de8448a53c6ed7f9078ad9acfa8c082b82f0e1af25a1cd0b87c2
3
+ size 4983089216
12k/model-00004-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a19dd0c966106d8fd2082315f4a15e2582656139490a27032e3293eda55b6723
3
+ size 4949503184
12k/model-00005-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f525efaf0e083056532f0899caf076e98ce78531980545f07d70225f69c37569
3
+ size 4932723808
12k/model-00006-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1a15ce1829ea694b16641073706dd7e1fc04ce285ac3cbac6449dc4badd63e66
3
+ size 4999868608
12k/model-00007-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eae10ca83e98d579736a3160fef4ec1d338bb41ff9b2a2d6a7ae057dda6f7e89
3
+ size 1656173544
12k/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
Free AI Image Generator No sign-up. Instant results. Open Now