Commit
93d045d
·
0 Parent(s):

Super-squash branch 'main' using huggingface_hub

Browse files

Co-authored-by: CYChenv <CYChenv@users.noreply.huggingface.co>
Co-authored-by: ychao-nvidia <ychao-nvidia@users.noreply.huggingface.co>
Co-authored-by: tangyue0820 <tangyue0820@users.noreply.huggingface.co>
Co-authored-by: harrim-nv <harrim-nv@users.noreply.huggingface.co>
Co-authored-by: liang1225 <liang1225@users.noreply.huggingface.co>
Co-authored-by: mli0603 <mli0603@users.noreply.huggingface.co>
Co-authored-by: mbalaNV <mbalaNV@users.noreply.huggingface.co>
Co-authored-by: yuzhud <yuzhud@users.noreply.huggingface.co>

Files changed (42) hide show
  1. .gitattributes +38 -0
  2. BIAS.md +11 -0
  3. EXPLAINABILITY.md +16 -0
  4. PRIVACY.md +6 -0
  5. README.md +414 -0
  6. SAFETY.md +11 -0
  7. assets/Pick_up_the_banana_and_place_it_in_the_bowl_0_env0_viewport.mp4 +3 -0
  8. chat_template.json +3 -0
  9. checkpoint.json +1 -0
  10. config.json +244 -0
  11. generation_config.json +14 -0
  12. images/benchmark-action-2.png +0 -0
  13. images/benchmark-roboarena.png +3 -0
  14. merges.txt +0 -0
  15. model.safetensors.index.json +0 -0
  16. model_index.json +24 -0
  17. preprocessor_config.json +21 -0
  18. scheduler/scheduler_config.json +33 -0
  19. text_tokenizer/added_tokens.json +28 -0
  20. text_tokenizer/chat_template.jinja +120 -0
  21. text_tokenizer/merges.txt +0 -0
  22. text_tokenizer/special_tokens_map.json +31 -0
  23. text_tokenizer/tokenizer.json +3 -0
  24. text_tokenizer/tokenizer_config.json +239 -0
  25. text_tokenizer/vocab.json +0 -0
  26. tokenizer.json +0 -0
  27. tokenizer_config.json +239 -0
  28. transformer/config.json +54 -0
  29. transformer/diffusion_pytorch_model-00001-of-00007.safetensors +3 -0
  30. transformer/diffusion_pytorch_model-00002-of-00007.safetensors +3 -0
  31. transformer/diffusion_pytorch_model-00003-of-00007.safetensors +3 -0
  32. transformer/diffusion_pytorch_model-00004-of-00007.safetensors +3 -0
  33. transformer/diffusion_pytorch_model-00005-of-00007.safetensors +3 -0
  34. transformer/diffusion_pytorch_model-00006-of-00007.safetensors +3 -0
  35. transformer/diffusion_pytorch_model-00007-of-00007.safetensors +3 -0
  36. transformer/diffusion_pytorch_model.safetensors.index.json +816 -0
  37. vae/config.json +129 -0
  38. vae/diffusion_pytorch_model.safetensors +3 -0
  39. video_preprocessor_config.json +21 -0
  40. vision_encoder/config.json +25 -0
  41. vision_encoder/model.safetensors +3 -0
  42. vocab.json +0 -0
.gitattributes ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ text_tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ images/benchmark-roboarena.png filter=lfs diff=lfs merge=lfs -text
38
+ *.mp4 filter=lfs diff=lfs merge=lfs -text
BIAS.md ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ## Bias
2
+
3
+ | Field | Response |
4
+ | :---- | :---- |
5
+ | Participation considerations from adversely impacted groups [protected classes](https://www.senate.ca.gov/content/protected-classes) in model design and testing | None. |
6
+ | Measures taken to mitigate against unwanted bias | Training, evaluation, and testing data are curated before release to filter restricted content, including content relating to protected classes. Model behavior is evaluated across Physical AI domains — robotics, autonomous vehicles, human-centric scenes, common scenes, industry, miscellaneous, and physics-oriented benchmarks — with attention to coverage across diverse demographic and contextual characteristics that affect protected-class outcomes. |
7
+ | Which characteristic (feature) show(s) the greatest difference in performance?: | Greatest performance differences are observed in tasks requiring long-horizon temporal consistency, fine-grained physical interactions, and embodiment-specific action generation. Performance is generally stronger on common visual reasoning and world-generation tasks than on complex multi-agent, robotics-control, or tightly synchronized multimodal generation scenarios. |
8
+ | Which feature(s) have the worst performance overall? | Performance is generally weakest in tasks requiring long-horizon temporal consistency, precise physical interactions, embodiment-specific action control, and strict audio-visual synchronization. |
9
+ | If using internal data, description of methods implemented in data acquisition or processing, if any, to address the prevalence of identifiable biases in the training, testing, and validation data: | Bias-specific methods applied during data processing include person-presence screening, demographic-taxonomy classification (age, gender, ethnicity), embedding-based diversity analysis, and dataset balancing across sources. Internal analysis surfaced: non-person scenes are more prevalent than person-centric content; demographic-taxonomy outputs on person-present samples are most frequently "uncertain" across age, gender, and ethnicity dimensions; and source-type variation, with people-centric image and video datasets showing higher demographic signal than document-, object-, robotics-, or scene-focused datasets. *(Quantitative details in the row below.)* Downstream deployments should add bias audits, fairness evaluation, red-teaming, demographically balanced fine-tuning, or counterfactual augmentation as mitigations. |
10
+ | Tools used to assess statistical imbalances and highlight patterns that may introduce bias into AI models: | Dataset analytics pipelines, metadata distribution analysis, heuristic quality checks, embedding-based clustering, model-assisted filtering systems, and benchmark evaluation suites are used to assess statistical imbalances and identify patterns that may introduce bias into model behavior. |
11
+ | Tools used to assess statistical imbalances and highlight patterns that may introduce bias into AI models: | These datasets, such as OpenImages-derived detection-to-NLP datasets, visual grounding and VQA datasets, document/image understanding datasets, video/action understanding datasets, and NVIDIA-created or curated visual datasets, do not collectively or exhaustively represent all demographic groups (and proportionally therein). For instance, automated person-presence screening did not identify a person in approximately 58% of visual samples analyzed across approximately 400 datasets, while person-present signals were identified in approximately 42% of analyzed samples. In the subset where person-present signals were identified, these datasets contain uneven representation splits across the measured visual taxonomies: age outputs were most frequently uncertain, followed by child and adult; gender outputs were most frequently uncertain, followed by male and female; and ethnicity outputs were most frequently uncertain, followed by Hispanic and White as the most frequent identified categories. Dataset-level results vary by source type, with people-centric image and video datasets containing higher person-present and demographic-taxonomy signals than document-, object-, robotics-, or scene-focused datasets. To mitigate these imbalances, we recommend considering evaluation techniques such as bias audits, task-specific fairness evaluation, and red-teaming, along with fine-tuning with demographically balanced datasets and counterfactual data augmentation to align with the desired model behavior. This evaluation used a baseline of 200 samples across all datasets, with larger subsets of up to 3,000 samples utilized for certain in-depth analyses, identified as optimal thresholds for maximizing embedder accuracy. |
EXPLAINABILITY.md ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ## Explainability
2
+
3
+ | Field | Response |
4
+ | :---- | :---- |
5
+ | Intended Application & Domain | World reasoning and generation for Physical AI. |
6
+ | Model Type | Mixture-of-Transformers architecture with two towers. One is an autoregressive model for Physical AI reasoning; the other is a diffusion model for Physical AI generation. |
7
+ | Intended Users | Physical AI developers, researchers, and practitioners building or evaluating autonomous vehicle, robotics, and world-generation workflows. |
8
+ | Output | Images, videos, audio, and action commands. |
9
+ | Tools used to evaluate datasets to identify synthetic data and ensure data authenticity. | Dataset provenance analysis, metadata validation, watermark and artifact detection, embedding-based clustering, heuristic quality checks, and model-assisted data validation pipelines are used to identify synthetic content patterns, assess dataset authenticity, and improve data quality during dataset curation. |
10
+ | Describe how the model works | Cosmos3 is an Omni world foundation model that generates texts, images, videos, audio, and action commands from combinations of text, images, videos, and action trajectory inputs. Input tokens from multiple modalities are packed into a shared sequence and processed by our mixture-of-transformer backbone with modality-specific output heads. |
11
+ | Name the adversely impacted groups this has been tested to deliver comparable outcomes regardless of: | None. |
12
+ | Technical Limitations | The model may not follow text, image, video, audio, or action trajectory inputs accurately in challenging cases, especially where the input contains complex scene composition, unusual camera motion, multiple interacting agents, low lighting, high motion blur, or fine-grained physical interactions. Generated outputs may contain temporal inconsistency, object morphing, inaccurate 3D structure, or implausible physical dynamics. Generated audio may not accurately render intelligible speech, or maintain strict temporal and semantic alignment with the visual context. |
13
+ | Verified to have met prescribed NVIDIA quality standards | Yes. |
14
+ | Performance Metrics | Video generation is measured using PAIBench-G, RBench, PhysicsIQ, and Artifical Analysis Image2Video benchmark. Image generation uses UniGenBench and Artifical Analysis Text2Image benchmark. For transfer evaluation, we use PAIBench-C and AVBench-C. Audio generation uses internal benchmarks. Action prediction uses metrics such as action MSE, Absolute Translation Error, Relative Translation Error, Relative Rotation Error, PSNR, and robotic task completion success rate. |
15
+ | Potential Known Risks | This model can generate synthetic media and may produce content that is offensive, unsafe, misleading, indecent, or unsuitable for a target deployment. Users should implement robust safety guardrails — including content filtering, abuse monitoring, and access controls — to reduce the risk of harmful outputs. Users are responsible for ensuring that their use of the model complies with all applicable laws and regulations, and for regularly reviewing and updating their guardrails as risks evolve. |
16
+ | Licensing | [OpenMDW1.1](https://openmdw.ai/) |
PRIVACY.md ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ ## Privacy
2
+ | Privacy Information |
3
+ |---|
4
+ | The model was trained on large-scale publicly available data that may contain images, audio-video, and text relating to people. NVIDIA collected and used this data in compliance with applicable data protection and privacy laws. This model was not designed to derive insights or otherwise learn from any personal data contained in the datasets. |
5
+ | NVIDIA uses a combination of filters, data minimization techniques, and other guardrails to help prevent personal data from being recited by our models. We employ automated tools and data processing techniques during pre-training or training to identify and filter certain categories of personal data. For example, for text-bearing source and document components, our automated tools identified potential personal data such as person names, locations, and possible business or public-facing contact information such as email addresses and phone numbers. We reviewed and removed any verified instances of personal data through a combination of automated filtering and human-in-the-loop validation. |
6
+ | Please review NVIDIA's [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/) for more information. |
README.md ADDED
@@ -0,0 +1,414 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: openmdw1.1-license
4
+ license_link: >-
5
+ https://openmdw.ai/license/1-1/
6
+ library_name: cosmos
7
+ tags:
8
+ - nvidia
9
+ - cosmos
10
+ - cosmos3
11
+ - world action model
12
+ - policy model
13
+ countDownloads:
14
+ - checkpoint.json
15
+ - config.json
16
+ - generation_config.json
17
+ - model.safetensors.index.json
18
+ - model_index.json
19
+ - tokenizer.json
20
+ - tokenizer_config.json
21
+ - sound_tokenizer/config.json
22
+ - sound_tokenizer/diffusion_pytorch_model.safetensors
23
+ - text_tokenizer/tokenizer.json
24
+ - text_tokenizer/tokenizer_config.json
25
+ - transformer/config.json
26
+ - transformer/diffusion_pytorch_model-00001-of-00007.safetensors
27
+ - transformer/diffusion_pytorch_model-00002-of-00007.safetensors
28
+ - transformer/diffusion_pytorch_model-00003-of-00007.safetensors
29
+ - transformer/diffusion_pytorch_model-00004-of-00007.safetensors
30
+ - transformer/diffusion_pytorch_model-00005-of-00007.safetensors
31
+ - transformer/diffusion_pytorch_model-00006-of-00007.safetensors
32
+ - transformer/diffusion_pytorch_model-00007-of-00007.safetensors
33
+ - transformer/diffusion_pytorch_model.safetensors.index.json
34
+ - vae/config.json
35
+ - vae/diffusion_pytorch_model.safetensors
36
+ - vision_encoder/config.json
37
+ - vision_encoder/model.safetensors
38
+ ---
39
+
40
+ # **Cosmos 3: Omnimodal World Models for Physical AI**
41
+ **[Model Collection](https://huggingface.co/collections/nvidia/cosmos3)** | **[Code](https://github.com/nvidia/cosmos)** | **[White Paper](https://research.nvidia.com/labs/cosmos-lab/cosmos3/technical-report.pdf)** | **[Website](https://research.nvidia.com/labs/cosmos-lab/cosmos3/)**
42
+
43
+ [NVIDIA Cosmos™](https://github.com/nvidia/cosmos) is a world foundation model platform designed to accelerate the development of Physical AI by enabling machines to understand, simulate, and interact with the physical world across robotics, autonomous driving, and smart space environments, including industrial and factory-scale applications.
44
+
45
+ # Model Overview: Cosmos3-Nano-Policy-DROID
46
+
47
+ ## Description
48
+
49
+ Cosmos3 is a collection of Omnimodal world models capable of generating dynamic, high-quality video, image, audio, and action commands from combinations of text, image, video, and action trajectory inputs. It serves as a foundational building block for a broad range of Physical AI applications and research spanning world understanding, world generation, simulation, and embodied policy learning.
50
+
51
+ This model is ready for commercial and non-commercial use.
52
+
53
+ **Model Developer:** NVIDIA
54
+
55
+ ### Model Versions
56
+ - Cosmos3-Nano:
57
+ - Given multimodal inputs including text, images, video, audio, and action trajectories, generate coherent text, images, video, audio, and action outputs for multimodal understanding, world simulation, future prediction, action reasoning, and Physical AI applications.
58
+
59
+ - Cosmos3-Super:
60
+ - Given multimodal inputs including text, images, video, audio, and action trajectories, generate coherent text, images, video, audio, and action outputs for multimodal understanding, world simulation, future prediction, action reasoning, and Physical AI applications.
61
+
62
+ - Cosmos3-Nano-Policy-DROID:
63
+ - Given language instructions and visual observations from the DROID robot platform, generate robot action trajectories for manipulation and control tasks.
64
+
65
+ - Cosmos3-Super-Image2Video:
66
+ - Given one input image and text instructions, generate temporally coherent video sequences that are consistent with the provided visual content.
67
+
68
+ - Cosmos3-Super-Text2Image:
69
+ - Given text input, generate high-fidelity images that are consistent with the provided description.
70
+
71
+ ### License
72
+
73
+ This model is released under the [OpenMDW1.1](https://openmdw.ai/license/1-1/)
74
+
75
+ ### Deployment Geography
76
+
77
+ Global
78
+
79
+ ### Use Case
80
+
81
+ Physical AI: Encompassing robotics, autonomous vehicles (AV), and smart space environments, including industrial and factory-scale applications.
82
+
83
+ ### Release Date
84
+
85
+ Hugging Face 05/31/2026 via [https://huggingface.co/collections/nvidia/cosmos3](https://huggingface.co/collections/nvidia/cosmos3)
86
+ GitHub 05/31/2026 via [https://github.com/nvidia/cosmos](https://github.com/nvidia/cosmos)
87
+
88
+ ## Model Architecture
89
+
90
+ **Architecture Type:** Transformer
91
+
92
+ **Network Architecture:** Mixture-of-Transformers (MoT)
93
+
94
+ Cosmos3 is an Omni-modal foundation model built on a Mixture-of-Transformers (MoT) architecture consisting of two complementary transformer towers: an autoregressive transformer for discrete token generation and a diffusion transformer for continuous multimodal generation. During inference, text is generated through standard next-token autoregressive decoding, while non-text modalities, such as images, video, audio, and actions, are synthesized through iterative denoising. This unified architecture enables Cosmos3 to model heterogeneous modalities within a single framework while preserving generation mechanisms best suited to each modality.
95
+
96
+ **This model was developed based on:** [Cosmos Framework](https://github.com/nvidia/cosmos-framework)
97
+
98
+ **Number of trainable model parameters:**
99
+
100
+ - Cosmos3-Nano: 16B
101
+ - Cosmos3-Super: 64B
102
+ - Cosmos3-Nano-Policy-DROID: 16B
103
+ - Cosmos3-Super-Image2Video: 64B
104
+ - Cosmos3-Super-Text2Image: 64B
105
+
106
+ ## Input/Output Specifications
107
+
108
+ - **Generator Input**
109
+ - **Input Type(s)**: Text, Image, Video (with audio or without audio), Action Trajectory
110
+ - **Input Format(s)**:
111
+ - Text: String
112
+ - Image: jpg, png, jpeg, webp
113
+ - Video (with or without audio): mp4
114
+ - Action: json (1D list)
115
+ - **Input Parameters**:
116
+ - Text: One-dimensional (1D)
117
+ - Image: Two-dimensional (2D)
118
+ - Video: Three-dimensional (3D)
119
+ - Audio: One-dimensional (1D)
120
+ - Action trajectory: One-dimensional (1D)
121
+ - **Other Properties Related to Input**:
122
+ - For video inputs, we accept various resolutions, including 720p, 480p, and 256p.
123
+ - When using input video with audio muxed into the video MP4 file, the audio should have 2 channels (stereo) and a 48 kHz sample rate.
124
+ - Image and video inputs are RGB color (8 bits per channel, sRGB color space); grayscale inputs are not supported.
125
+ - Action input is a per-frame sequence of robot/agent state or control values (e.g., joint positions, gripper state, camera pose). The full input is a 2D array shaped (T, D), where T is the number of frames and D is the embodiment-specific dimensionality listed below.
126
+ - Input action is only supported for compatible embodiments, including general camera motion (9D), autonomous vehicle (9D), egocentric motion (57D), single Franka Panda arm with RobotiQ gripper (10D), dual Franka Panda arm with RobotiQ gripper (20D), Agibot (29D), UR (10D), Google robot (10D), WidowX 250 (10D), UMI (9D).
127
+ - **Input Size and Length limits:**
128
+ - **Text:** 4096 tokens
129
+ - **Image:** 256p, 480p, and 720p resolution at one of these aspect ratios (16:9, 4:3, 1:1, 3:4, 9:16)
130
+ - **Video:** 256p, 480p, and 720p resolution at one of these aspect ratios (16:9, 4:3, 1:1, 3:4, 9:16). Max number of frames = 5.
131
+ - **Audio:** Max 0.5 second
132
+ - **Action:** 16 – 400 video frames
133
+ - **Generator Output**
134
+ - **Output Type(s)**: Image, video, audio, action, text
135
+ - **Output Format(s)**:
136
+ - Image: JPG
137
+ - Video: MP4
138
+ - Audio: Advanced Audio Coding (AAC) stream (muxed within the MP4)
139
+ - Action: 1D list (.json)
140
+ - Text: string
141
+ - **Output Parameters**:
142
+ - Image: Two-dimensional (2D)
143
+ - Video: Three-dimensional (3D)
144
+ - Audio: One-dimensional (1D)
145
+ - Action: One-dimensional (1D)
146
+ - Text: One-dimensional (1D)
147
+ - **Other Properties Related to Output**:
148
+ - The generated video is an MP4 file, with the resolution, frame rate, and duration specified in the input. The generated audio is encoded in AAC format, muxed into the video MP4 file with 2 channels (stereo) and a 48 kHz sample rate.
149
+ - Video generation supports durations from 5 to 400 frames, with 189 frames as the default generation duration.
150
+ - The generated action is only supported for compatible embodiments, including general camera motion (9D), autonomous vehicle (9D), egocentric motion (57D), single Franka Panda arm with RobotiQ gripper (10D), dual Franka Panda arm with RobotiQ gripper (20D), Agibot (29D), UR (10D), Google robot (10D), WidowX 250 (10D), UMI (9D).
151
+ - Audio: 48 kHz stereo AAC stream muxed into video mp4
152
+ - Video: mp4 at the FPS specified in input
153
+ - Image: JPEG
154
+ - **Reasoner Input**
155
+ - **Input Type(s)**: Text, Text+Image, Text+Video
156
+ - **Input Format(s)**:
157
+ - Text: String
158
+ - Image: jpg, png, jpeg, webp
159
+ - Video: mp4
160
+ - **Input Parameters**:
161
+ - Text: One-dimensional (1D)
162
+ - Image: Two-dimensional (2D)
163
+ - Video: Three-dimensional (3D)
164
+ - **Other Properties Related to Input**:
165
+ - Video inputs are recommended at a frame rate of 4 fps.
166
+ - Long-context inputs supported up to 256K tokens.
167
+ - **Input Size and Length limits:**
168
+ - **Text:** Up to 256K tokens (context window).
169
+ - **Image:** Standard input image formats; passed as file or URL.
170
+ - **Video:** mp4 at the recommended 4 fps.
171
+ - **Reasoner Output**
172
+ - **Output Type(s)**: Text
173
+ - **Output Format(s)**:
174
+ - Text: string
175
+ - **Output Parameters**:
176
+ - Text: One-dimensional (1D)
177
+ - **Other Properties Related to Output**:
178
+ - Default `max_tokens=4096+` is recommended for reasoning outputs; longer outputs may be requested.
179
+ - Reasoning outputs may include structured chain-of-thought, 2D/3D point localization, and bounding-box coordinates for vision-based tasks.
180
+
181
+ The video content visualizes the input text description as a short animated scene, capturing key elements within the specified time constraints.
182
+
183
+ Our AI models are designed and/or optimized to run on NVIDIA GPU-accelerated systems. By leveraging NVIDIA's hardware (e.g., GPU cores) and software frameworks (e.g., CUDA libraries), the model achieves faster training and inference times compared to CPU-only solutions.
184
+
185
+ ## Software Integration
186
+
187
+ **Runtime Engine(s):**
188
+
189
+ - [PyTorch](https://github.com/nvidia/cosmos3)
190
+ - [vLLM-Omni](https://github.com/vllm-project/vllm-omni)
191
+ - [Hugging Face Diffusers](https://huggingface.co/docs/diffusers/en/index)
192
+
193
+ **Supported Hardware Microarchitecture Compatibility:**
194
+
195
+ - NVIDIA Ampere
196
+ - NVIDIA Blackwell
197
+ - NVIDIA Hopper
198
+
199
+ **Operating System(s):**
200
+
201
+ - Linux (We have not tested on other operating systems.)
202
+
203
+ **Note:** Only BF16 precision is tested. Other precisions like FP4, FP8, and FP16 are not officially supported.
204
+
205
+ The integration of foundation and fine-tuned models into AI systems requires additional testing using use-case-specific data to ensure safe and effective deployment. Following the V-model methodology, iterative testing and validation at both unit and system levels are essential to mitigate risks, meet technical and functional requirements, and ensure compliance with safety and ethical standards before deployment.
206
+
207
+ ## Training, Testing, and Evaluation Datasets
208
+
209
+ ### Dataset Overview
210
+
211
+ - **Total Size:** 1.3B data points
212
+ - **Total Number of Datasets:** 393 dataset entries
213
+ - **Dataset partition:** Training [100%], Testing [N/A — evaluation benchmarks used separately], Validation [N/A — evaluation benchmarks used separately]
214
+ - **Time period for training data collection:** 2024–2026
215
+ - **Time period for testing data collection:** N/A (standard public benchmarks)
216
+ - **Time period for validation data collection:** N/A (standard public benchmarks)
217
+
218
+ Raw data from internal and external sources is transformed into training-ready data through multiple stages of curation, filtering, and quality review. Data acquisition spans diverse multimodal sources — robotics, autonomous driving, industrial environments, indoor and outdoor scenes, varied lighting and weather conditions, camera viewpoints, object categories, and human activities — to broaden coverage across Physical AI operating environments. Automated filtering pipelines remove corrupted, duplicate, low-quality, and restricted content. Metadata analysis, heuristic rules, and model-assisted classifiers are applied during preprocessing to flag anomalous distributions and low-diversity subsets. Human review supplements automated filtering for selected datasets, benchmark construction, and targeted quality analysis. Datasets are balanced across modalities and task categories — visual reasoning, text-to-image, text-to-video, image-to-video, audio generation, video transfer, action-conditioned generation, and action command generation — to reduce overrepresentation of narrow domains. Synthetic and simulation-based augmentation supplements coverage of rare physical interactions and edge-case scenarios. Deduplication and provenance tracking are applied across the corpus. The resulting processed data is converted into model-ready tokenized or encoded representations through modality-specific preprocessors before training begins.
219
+
220
+ Training datasets passed through multiple layers of automated and manual safeguards designed to reduce the presence of harmful or policy-violating content across categories including weapons and weapons-related instructional content, criminal planning, child sexual abuse material (CSAM), non-consensual intimate imagery (NCII), sexual content involving minors, harassment, hate speech, profanity, threats and incitement to violence, self-harm or suicide-related content, and graphic violence. Data sources are reviewed for licensing compatibility, provenance, and alignment with internal data governance and safety policies before admission into training corpora. Automated filtering pipelines combine multiple detection strategies: hash-matching against known CSAM and NCII reference databases; classifier-based moderation models trained for explicit sexual content, hate speech, violence, weapons imagery, and other restricted categories; keyword and regex-based screening for criminal-planning, threats, and self-harm phrases in text data; metadata and provenance heuristics for source-level risk signals; and embedding-based anomaly detection to surface samples that fall outside expected distributions. Human review and targeted audits supplement automated filtering for selected datasets, benchmark construction, and safety-sensitive evaluation. For multimodal Physical AI data (robotics, autonomous driving, industrial scenes), additional filtering targets invalid action trajectories, physically implausible interactions, and unsafe control sequences. Synthetic and simulation-generated data are evaluated through internal validation before inclusion. Benchmark evaluations and red-team testing are applied post-training to surface remaining safety gaps across world generation, reasoning, audio, and action tasks. No large-scale data-filtering process can guarantee complete removal of all harmful content; residual risks may remain, particularly in rare edge cases or open-world deployment settings. Ongoing monitoring and dataset review continue post-release.
221
+
222
+ **Data Modality and Training Data Size**
223
+
224
+ | Modality | Reasoning Data Sample Count | Generation Data Sample Count |
225
+ | -------- | ------------------- | -------------------- |
226
+ | Text | 22M | Not Applicable |
227
+ | Image | 19M | 767M |
228
+ | Video | 1M | 348M |
229
+ | Audio | Not Applicable | 139M |
230
+ | Action | Not Applicable | 8M |
231
+
232
+ **Data Collection Method by dataset**
233
+
234
+ - Hybrid: Automatic/Sensors, Synthetic, Automated
235
+
236
+ **Labeling Method by dataset**
237
+
238
+ - Hybrid: Human, Automated
239
+
240
+ **Properties:** The training, testing, and evaluation datasets consist of diverse multimodal video, image, audio, action, synthetic, and sensor-conditioned data sourced from NVIDIA-owned data and publicly available, commercially permissive datasets. These datasets are curated to exclude known restricted content and to support building an Omni model that learns to generate and reason about dynamic physical environments across world reasoning and generation tasks.
241
+
242
+ ### Public Datasets
243
+
244
+ | Dataset&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp; | Samples&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp; |
245
+ |---|---|
246
+ | OpenImage | 1.2M |
247
+ | Coyo700M | 100M |
248
+ | YouTube Video | 340M |
249
+ | UMI | 4.5M |
250
+
251
+ ### Private Datasets
252
+
253
+ | Dataset&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp; | Samples&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp; |
254
+ |---|---|
255
+ | Egocentric | 7M |
256
+ | Nexar | 0.6M |
257
+ | AgiBot | 0.2M |
258
+ | HOI | 0.3M |
259
+
260
+ ### Synthetic Datasets
261
+
262
+ | Dataset | Samples |
263
+ |---|---|
264
+ | synthetic images generated using HiDream-I1 | 15M |
265
+ | synthetic images generated using Qwen-Image-2512 | 14M |
266
+ | synthetic captions generated using Qwen3-VL | 1115M |
267
+
268
+ ## Evaluation Datasets
269
+
270
+ **Data Collection Method by dataset**
271
+
272
+ - Hybrid: Automatic/Sensors, Synthetic, Automated
273
+
274
+ **Labeling Method by dataset**
275
+
276
+ - Hybrid: Human, Automated
277
+
278
+ **Properties:** The training, testing, and evaluation datasets consist of diverse multimodal video, image, audio, action, synthetic, and sensor-conditioned data sourced from NVIDIA-owned data and publicly available, commercially permissive datasets. These datasets are curated to exclude known restricted content and to support building an Omni model that learns to generate and reason about dynamic physical environments across world reasoning and generation tasks.
279
+
280
+ ## Benchmarks
281
+
282
+ Please see our [technical paper](https://research.nvidia.com/labs/cosmos-lab/cosmos3/technical-report.pdf) for detailed evaluations of the base model.
283
+
284
+ ### Action
285
+
286
+ #### RoboLab
287
+
288
+ ![RoboLab — task success rates across language specificity and difficulty levels](images/benchmark-action-2.png)
289
+
290
+ #### RoboArena
291
+
292
+ <p align="center"><img src="images/benchmark-roboarena.png" alt="RoboArena Policy Leaderboard — Cosmos3-Nano-Policy ranked #1"></p>
293
+
294
+ ## Usage
295
+
296
+ - See [Cosmos](https://github.com/nvidia/cosmos) for details.
297
+
298
+ ### Quickstart
299
+
300
+ Cosmos3-Nano-Policy-DROID is served by a policy **Server** that streams actions to a **Client** driving a simulated or real robot. This example uses [`RoboLab`](https://github.com/NVlabs/RoboLab), a simulation benchmark for task-generalist policies, as the client. Start the server first, then connect the client.
301
+
302
+ #### Server
303
+
304
+ First, clone [`cosmos-framework`](https://github.com/NVIDIA/cosmos-framework):
305
+
306
+ ```bash
307
+ git clone https://github.com/NVIDIA/cosmos-framework.git
308
+ cd cosmos-framework
309
+ ```
310
+
311
+ Build the Docker image:
312
+
313
+ ```bash
314
+ docker build \
315
+ -t cosmos-framework:latest \
316
+ .
317
+ ```
318
+
319
+ Set your Hugging Face token and launch the container, which installs the dependencies:
320
+
321
+ ```bash
322
+ # Set your Hugging Face token (https://huggingface.co/settings/tokens):
323
+ export HF_TOKEN=<your_hf_token>
324
+
325
+ docker run \
326
+ -it \
327
+ -e HF_HOME=/workspace/.cache/huggingface \
328
+ -e HF_TOKEN=$HF_TOKEN \
329
+ --net host \
330
+ --rm \
331
+ --runtime nvidia \
332
+ -v .:/workspace \
333
+ -v /workspace/.venv \
334
+ -v $HOME/.cache/huggingface:/root/.cache/huggingface \
335
+ cosmos-framework:latest \
336
+ bash -c '\
337
+ uv sync \
338
+ --all-extras \
339
+ --group=cu130-train \
340
+ --group=policy-server && \
341
+ exec bash; \
342
+ '
343
+ ```
344
+
345
+ Inside the container, start the policy server:
346
+
347
+ ```
348
+ python -m cosmos_framework.scripts.action_policy_server_robolab \
349
+ --port 8000
350
+ ```
351
+
352
+ #### Client
353
+
354
+ Clone [`RoboLab`](https://github.com/NVlabs/RoboLab):
355
+
356
+ ```bash
357
+ git clone https://github.com/NVlabs/RoboLab.git
358
+ cd RoboLab
359
+ ```
360
+
361
+ Build the Docker image:
362
+
363
+ ```bash
364
+ ./docker/build_docker.sh latest
365
+ ```
366
+
367
+ Launch the container:
368
+
369
+ ```bash
370
+ ./docker/run_docker.sh latest
371
+ ```
372
+
373
+ Run a task against the policy server. This opens a viewer window for real-time visualization of the simulation:
374
+
375
+ ```bash
376
+ python policies/cosmos3/run.py \
377
+ --task BananaInBowlTask
378
+ ```
379
+
380
+ To evaluate across multiple sub-environments in parallel in headless mode:
381
+
382
+ ```bash
383
+ python policies/cosmos3/run.py \
384
+ --task BananaInBowlTask \
385
+ --num-envs 10 \
386
+ --headless
387
+ ```
388
+
389
+ Example output:
390
+
391
+ <video controls width="864" height="480" src="
392
+ https://huggingface.co/nvidia/Cosmos3-Nano-Policy-DROID/resolve/main/assets/Pick_up_the_banana_and_place_it_in_the_bowl_0_env0_viewport.mp4"></video>
393
+
394
+ ## Limitations
395
+
396
+ Cosmos3 may produce imperfect outputs in challenging scenarios. Generation artifacts include temporal inconsistency, unstable camera or object motion, imprecise physical interactions, inaccurate audio-video synchronization, and action-state drift — especially in long-horizon or high-resolution outputs. Reasoning may also be incorrect: object states, causal relationships, spatial geometry, temporal ordering, agent intent, and future outcomes can be misinferred, and complex or long-context inputs may yield hallucinated entities, inconsistent interpretations, or implausible predictions. Because the model lacks an explicit physics simulator, 3D geometry, 4D space-time evolution, object permanence, contact dynamics, and physical laws are only approximated — producing artifacts such as disappearing or morphing objects, unrealistic collisions, and physically implausible motions. Quality further degrades in out-of-distribution environments, safety-critical edge cases, and domains underrepresented in training.
397
+
398
+ Cosmos3 outputs should not be treated as physically accurate simulation, reliable ground-truth reasoning, or safety-certified decision making. Applications involving robotics control, autonomous systems, scientific simulation, or safety-critical planning require additional validation, external constraints, system-level safety analysis, and domain-specific guardrails before deployment.
399
+
400
+ ## Inference
401
+
402
+ **Acceleration Engine:** [PyTorch](https://pytorch.org/), [vLLM](https://github.com/vllm-project/vllm), [vLLM-Omni](https://github.com/vllm-project/vllm-omni), [Hugging Face Diffusers](https://github.com/huggingface/diffusers)
403
+
404
+ **Test Hardware:** H100
405
+
406
+ ## Ethical Considerations
407
+
408
+ NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. Developers should work with their internal model team to ensure this model meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
409
+
410
+ Please make sure you have proper rights and permissions for all input image and video content; if image or video includes people, personal health information, or intellectual property, the image or video generated will not blur or maintain proportions of image subjects included.
411
+
412
+ Users are responsible for model inputs and outputs. Users are responsible for ensuring safe integration of this model, including implementing guardrails as well as other safety mechanisms, prior to deployment.
413
+
414
+ For more detailed information on ethical considerations for this model, please see the Model Card++ [Explainability](EXPLAINABILITY.md), [Bias](BIAS.md), [Safety & Security](SAFETY.md), and [Privacy](PRIVACY.md) subcards. Please report model quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
SAFETY.md ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ## Safety & Security
2
+
3
+ | Field | Response |
4
+ | :---- | :---- |
5
+ | Model Application(s) | World reasoning and generation for Physical AI. |
6
+ | Describe the life critical impact: | This model is not a safety-certified component and must not be used as the sole basis for life-critical decisions or control without additional system-level validation, safety analysis, and safeguards. The model is not designed or tested by NVIDIA for use in any system or application where the use of or failure of such system or application developed with the model could result in injury, death, or catastrophic damage. NVIDIA is not liable to any party, in whole or in part, for any claims or damages arising from those uses. Any system or application developed with the model must include sufficient safety and redundancy features and comply with applicable legal and regulatory standards and requirements. |
7
+ | Description of methods implemented in data acquisition or processing, if any, to address other types of potentially harmful data in the training, testing, and validation data: | Training, evaluation, and validation datasets pass through multi-stage automated and manual filtering to reduce harmful, unsafe, restricted, or policy-violating content. Pipelines include source-licensing review, deduplication, metadata-based and classifier-based moderation, embedding-based anomaly detection, and human audits on selected datasets. For Physical AI data (robotics, autonomous driving, industrial scenes), filtering also targets invalid action trajectories, physically implausible interactions, and unsafe control sequences. Synthetic and simulation-generated data are evaluated through internal validation before inclusion. Benchmark and red-team testing surface remaining safety gaps across world generation, reasoning, audio, and action tasks. No data-filtering process can guarantee complete removal; developers are responsible for application-specific safeguards and validation before deployment. |
8
+ | Description of any methods implemented in data acquisition or processing, if any, to address illegal or harmful content in the training data, including, but not limited to, child sexual abuse material (CSAM) and non-consensual intimate imagery (NCII) | In addition to the general unsafe-content filtering described above, training data acquisition and preprocessing apply CSAM- and NCII-specific safeguards: hash-matching systems against known CSAM databases, classifier-based moderation models trained specifically for explicit content and NCII detection, and provenance and licensing review for sources containing human imagery. Identified content is removed at ingest, with human review and targeted audits supplementing automated filtering for selected datasets. Despite these safeguards, no large-scale data-filtering system can guarantee complete detection. Ongoing monitoring and dataset review continue post-release. |
9
+ | Use Case Restrictions | Use is governed by the [OpenMDW1.1](https://openmdw.ai/) |
10
+ | Model and dataset restrictions | The Principle of least privilege (PoLP) is applied limiting access for dataset generation and model development. Restrictions enforce dataset access during training, and dataset license constraints adhered to. |
11
+ | Responsible Data Handling | This AI model was developed based on our policies to ensure responsible data handling and risk mitigation. The datasets used for training have been scanned for harmful content and illegal content, consistent with our policies including scanning for Child Sexual Abuse Material (CSAM). Ongoing review and monitoring mechanisms are in place based on our policies and to maintain data integrity. |
assets/Pick_up_the_banana_and_place_it_in_the_bowl_0_env0_viewport.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e4586cbcbdf7fea30facc1b04a48924f1505234000b04a0877287c812913ebeb
3
+ size 864271
chat_template.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {
2
+ "chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0].role == 'system' %}\n {%- if messages[0].content is string %}\n {{- messages[0].content }}\n {%- else %}\n {%- for content in messages[0].content %}\n {%- if 'text' in content %}\n {{- content.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {{- '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within <tools></tools> XML tags:\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\\n\\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\\n<tool_call>\\n{\\\"name\\\": <function-name>, \\\"arguments\\\": <args-json-object>}\\n</tool_call><|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0].content is string %}\n {{- messages[0].content }}\n {%- else %}\n {%- for content in messages[0].content %}\n {%- if 'text' in content %}\n {{- content.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- for message in messages %}\n {%- if message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' }}\n {%- if message.content is string %}\n {{- message.content }}\n {%- else %}\n {%- for content in message.content %}\n {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}\n <|vision_start|><|image_pad|><|vision_end|>\n {%- elif content.type == 'video' or 'video' in content %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}\n <|vision_start|><|video_pad|><|vision_end|>\n {%- elif 'text' in content %}\n {{- content.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"assistant\" %}\n {{- '<|im_start|>' + message.role + '\\n' }}\n {%- if message.content is string %}\n {{- message.content }}\n {%- else %}\n {%- for content_item in message.content %}\n {%- if 'text' in content_item %}\n {{- content_item.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {%- if message.tool_calls %}\n {%- for tool_call in message.tool_calls %}\n {%- if (loop.first and message.content) or (not loop.first) %}\n {{- '\\n' }}\n {%- endif %}\n {%- if tool_call.function %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '<tool_call>\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {%- if tool_call.arguments is string %}\n {{- tool_call.arguments }}\n {%- else %}\n {{- tool_call.arguments | tojson }}\n {%- endif %}\n {{- '}\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.first or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {%- if message.content is string %}\n {{- message.content }}\n {%- else %}\n {%- for content in message.content %}\n {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}\n <|vision_start|><|image_pad|><|vision_end|>\n {%- elif content.type == 'video' or 'video' in content %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}\n <|vision_start|><|video_pad|><|vision_end|>\n {%- elif 'text' in content %}\n {{- content.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {{- '\\n</tool_response>' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n"
3
+ }
checkpoint.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {}
config.json ADDED
@@ -0,0 +1,244 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "allow_patterns_overrides": [
3
+ "*/*.safetensors"
4
+ ],
5
+ "architectures": [
6
+ "Cosmos3ForConditionalGeneration"
7
+ ],
8
+ "image_token_id": 151655,
9
+ "model": {
10
+ "_recursive_": false,
11
+ "_target": "omni_mot_model",
12
+ "config": {
13
+ "_type": "omni_mot_model_config",
14
+ "action_gen": true,
15
+ "activation_checkpointing": {
16
+ "_type": "activation_checkpointing_config",
17
+ "determinism_check": "default",
18
+ "mode": "full",
19
+ "preserve_rng_state": true,
20
+ "save_ops_regex": [
21
+ "fmha"
22
+ ]
23
+ },
24
+ "causal_training_strategy": "none",
25
+ "compile": {
26
+ "_type": "compile_config",
27
+ "compile_dynamic": true,
28
+ "compiled_region": "language",
29
+ "coordinate_descent_tuning": false,
30
+ "enabled": true,
31
+ "max_autotune_pointwise": false,
32
+ "use_cuda_graphs": false
33
+ },
34
+ "diffusion_expert_config": {
35
+ "_type": "diffusion_expert_config",
36
+ "base_fps": 24,
37
+ "enable_fps_modulation": true,
38
+ "load_weights_from_pretrained": false,
39
+ "max_vae_latent_side_after_patchify": 20,
40
+ "patch_spatial": 2,
41
+ "position_embedding_type": "unified_3d_mrope",
42
+ "rope_h_extrapolation_ratio": 1.0,
43
+ "rope_t_extrapolation_ratio": 1.0,
44
+ "rope_w_extrapolation_ratio": 1.0,
45
+ "timestep_range": 1.0,
46
+ "unified_3d_mrope_reset_spatial_ids": true,
47
+ "unified_3d_mrope_temporal_modality_margin": 15000
48
+ },
49
+ "ema": {
50
+ "_type": "ema_config",
51
+ "enabled": false,
52
+ "iteration_shift": 0,
53
+ "rate": 0.1
54
+ },
55
+ "fixed_step_sampler_config": null,
56
+ "input_caption_key": "ai_caption",
57
+ "input_image_key": "images",
58
+ "input_video_key": "video",
59
+ "joint_attn_implementation": "two_way",
60
+ "latent_downsample_factor": 16,
61
+ "lbl": {
62
+ "_type": "lbl_config",
63
+ "coeff_gen": null,
64
+ "coeff_und": null,
65
+ "method": "local"
66
+ },
67
+ "log_enc_time_every_n": 100,
68
+ "lora_alpha": 32,
69
+ "lora_enabled": false,
70
+ "lora_rank": 16,
71
+ "lora_target_modules": "q_proj_moe_gen,k_proj_moe_gen,v_proj_moe_gen,o_proj_moe_gen",
72
+ "max_action_dim": 64,
73
+ "max_num_tokens_after_packing": -1,
74
+ "natten_parameter_list": null,
75
+ "net": null,
76
+ "num_embodiment_domains": 32,
77
+ "parallelism": {
78
+ "_type": "parallelism_config",
79
+ "cfg_parallel_shard_degree": 1,
80
+ "context_parallel_shard_degree": 1,
81
+ "data_parallel_replicate_degree": 1,
82
+ "data_parallel_shard_degree": 8,
83
+ "enable_inference_mode": false,
84
+ "fsdp_master_dtype": "float32"
85
+ },
86
+ "precision": "bfloat16",
87
+ "rectified_flow_inference_config": {
88
+ "_type": "rectified_flow_inference_config",
89
+ "num_train_timesteps": 1000,
90
+ "scheduler_type": "unipc",
91
+ "shift": 1,
92
+ "use_dynamic_shifting": false
93
+ },
94
+ "rectified_flow_training_config": {
95
+ "_type": "rectified_flow_training_config",
96
+ "action_loss_weight": 10.0,
97
+ "high_sigma_ratio": 0.05,
98
+ "high_sigma_timesteps_max": 1000,
99
+ "high_sigma_timesteps_min": 995,
100
+ "image_loss_scale": null,
101
+ "independent_action_schedule": false,
102
+ "independent_sound_schedule": false,
103
+ "loss_scale": 10.0,
104
+ "normalize_loss_by_active": false,
105
+ "shift": {
106
+ "256": 3,
107
+ "480": 5,
108
+ "720": 10
109
+ },
110
+ "shift_action": null,
111
+ "shift_sound": null,
112
+ "sound_loss_scale": null,
113
+ "train_time_action_distribution": "logitnormal",
114
+ "train_time_image_distribution": "logitnormal",
115
+ "train_time_sound_distribution": "logitnormal",
116
+ "train_time_video_distribution": "waver",
117
+ "train_time_weight": "uniform",
118
+ "use_discrete_rf": false,
119
+ "use_dynamic_shift": false,
120
+ "use_high_sigma_strategy": false,
121
+ "use_high_sigma_strategy_action": false,
122
+ "use_high_sigma_strategy_sound": false
123
+ },
124
+ "resolution": "720",
125
+ "sound_dim": null,
126
+ "sound_gen": false,
127
+ "sound_latent_fps": 25,
128
+ "sound_tokenizer": null,
129
+ "state_ch": 48,
130
+ "state_t": 300,
131
+ "tokenizer": {
132
+ "_target": "wan2pt2_vae_interface",
133
+ "bucket_name": "bucket",
134
+ "chunk_duration": 93,
135
+ "encode_bucket_multiple": null,
136
+ "encode_chunk_frames": {
137
+ "256": 68,
138
+ "480": 24,
139
+ "720": 12
140
+ },
141
+ "encode_exact_durations": [
142
+ 17
143
+ ],
144
+ "keep_decoder_cache": false,
145
+ "object_store_credential_path_pretrained": "/lustre/fsw/portfolios/cosmos/projects/cosmos_base_misc/users/yuzhud/imaginaire4-robolab/credentials/gcp_training.secret",
146
+ "spatial_compression_factor": 16,
147
+ "temporal_compression_factor": 4,
148
+ "temporal_window": null,
149
+ "use_streaming_encode": false,
150
+ "vae_path": "pretrained/tokenizers/video/wan2pt2/Wan2.2_VAE.pth"
151
+ },
152
+ "video_temporal_causal": false,
153
+ "vision_gen": true,
154
+ "vlm_config": {
155
+ "_type": "vlm_config",
156
+ "layer_module": null,
157
+ "model_instance": {
158
+ "_target": "qwen3_vl_text_for_causal_lm",
159
+ "config": {
160
+ "_target": "create_vlm_config",
161
+ "base_config": {
162
+ "_target": "qwen3_vl_mot_config_from_json_file",
163
+ "json_file": "cosmos3://vfm/models/vlm/qwen3_vl/configs/Qwen3-VL-8B-Instruct.json"
164
+ },
165
+ "qk_norm_for_text": true
166
+ }
167
+ },
168
+ "model_name": "Qwen/Qwen3-VL-8B-Instruct",
169
+ "pretrained_weights": {
170
+ "_type": "pretrained_weights_config",
171
+ "backbone_path": "s3://bucket/cosmos3/pretrained/huggingface/Qwen/Qwen3-VL-8B-Instruct/",
172
+ "checkpoint_format": null,
173
+ "credentials_path": "/lustre/fsw/portfolios/cosmos/projects/cosmos_base_misc/users/yuzhud/imaginaire4-robolab/credentials/gcp_checkpoint.secret",
174
+ "enable_gcs_patch_in_boto3": true,
175
+ "enabled": false
176
+ },
177
+ "qk_norm": false,
178
+ "tie_word_embeddings": false,
179
+ "tokenizer": {
180
+ "_target": "build_processor_lazy",
181
+ "config_variant": "hf",
182
+ "tokenizer_type": "Qwen/Qwen3-VL-8B-Instruct"
183
+ },
184
+ "use_system_prompt": false
185
+ }
186
+ }
187
+ },
188
+ "model_type": "cosmos3_omni",
189
+ "text_config": {
190
+ "attention_bias": false,
191
+ "attention_dropout": 0.0,
192
+ "bos_token_id": 151643,
193
+ "dtype": "bfloat16",
194
+ "eos_token_id": 151645,
195
+ "head_dim": 128,
196
+ "hidden_act": "silu",
197
+ "hidden_size": 4096,
198
+ "initializer_range": 0.02,
199
+ "intermediate_size": 12288,
200
+ "max_position_embeddings": 262144,
201
+ "model_type": "qwen3_vl_text",
202
+ "num_attention_heads": 32,
203
+ "num_hidden_layers": 36,
204
+ "num_key_value_heads": 8,
205
+ "rms_norm_eps": 1e-06,
206
+ "rope_scaling": {
207
+ "mrope_interleaved": true,
208
+ "mrope_section": [
209
+ 24,
210
+ 20,
211
+ 20
212
+ ],
213
+ "rope_type": "default"
214
+ },
215
+ "rope_theta": 5000000,
216
+ "use_cache": true,
217
+ "vocab_size": 151936
218
+ },
219
+ "tie_word_embeddings": false,
220
+ "transformers_version": "4.57.0.dev0",
221
+ "video_token_id": 151656,
222
+ "vision_config": {
223
+ "deepstack_visual_indexes": [
224
+ 8,
225
+ 16,
226
+ 24
227
+ ],
228
+ "depth": 27,
229
+ "hidden_act": "gelu_pytorch_tanh",
230
+ "hidden_size": 1152,
231
+ "in_channels": 3,
232
+ "initializer_range": 0.02,
233
+ "intermediate_size": 4304,
234
+ "model_type": "qwen3_vl",
235
+ "num_heads": 16,
236
+ "num_position_embeddings": 2304,
237
+ "out_hidden_size": 4096,
238
+ "patch_size": 16,
239
+ "spatial_merge_size": 2,
240
+ "temporal_patch_size": 2
241
+ },
242
+ "vision_end_token_id": 151653,
243
+ "vision_start_token_id": 151652
244
+ }
generation_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 151643,
3
+ "pad_token_id": 151643,
4
+ "do_sample": true,
5
+ "eos_token_id": [
6
+ 151645,
7
+ 151643
8
+ ],
9
+ "top_k": 20,
10
+ "top_p": 0.8,
11
+ "repetition_penalty": 1.0,
12
+ "temperature": 0.7,
13
+ "transformers_version": "4.56.0"
14
+ }
images/benchmark-action-2.png ADDED
images/benchmark-roboarena.png ADDED

Git LFS Details

  • SHA256: 099861aa042783769dab30e6a28e52cfd40224e739eddc68f9ceef124d6b2012
  • Pointer size: 131 Bytes
  • Size of remote file: 282 kB
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
model_index.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "Cosmos3OmniDiffusersPipeline",
3
+ "_diffusers_version": "0.37.1",
4
+ "scheduler": [
5
+ "diffusers",
6
+ "UniPCMultistepScheduler"
7
+ ],
8
+ "text_tokenizer": [
9
+ "transformers",
10
+ "Qwen2TokenizerFast"
11
+ ],
12
+ "transformer": [
13
+ "diffusers",
14
+ "Cosmos3OmniTransformer"
15
+ ],
16
+ "vae": [
17
+ "diffusers",
18
+ "AutoencoderKLWan"
19
+ ],
20
+ "vision_encoder": [
21
+ "transformers",
22
+ "Qwen3VLVisionModel"
23
+ ]
24
+ }
preprocessor_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "size": {
3
+ "longest_edge": 16777216,
4
+ "shortest_edge": 65536
5
+ },
6
+ "patch_size": 16,
7
+ "temporal_patch_size": 2,
8
+ "merge_size": 2,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_std": [
15
+ 0.5,
16
+ 0.5,
17
+ 0.5
18
+ ],
19
+ "processor_class": "Qwen3VLProcessor",
20
+ "image_processor_type": "Qwen2VLImageProcessorFast"
21
+ }
scheduler/scheduler_config.json ADDED
@@ -0,0 +1,33 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "UniPCMultistepScheduler",
3
+ "_diffusers_version": "0.37.1",
4
+ "beta_end": 0.02,
5
+ "beta_schedule": "linear",
6
+ "beta_start": 0.0001,
7
+ "disable_corrector": [],
8
+ "dynamic_thresholding_ratio": 0.995,
9
+ "final_sigmas_type": "zero",
10
+ "flow_shift": 1.0,
11
+ "lower_order_final": true,
12
+ "num_train_timesteps": 1000,
13
+ "predict_x0": true,
14
+ "prediction_type": "flow_prediction",
15
+ "rescale_betas_zero_snr": false,
16
+ "sample_max_value": 1.0,
17
+ "shift_terminal": null,
18
+ "sigma_max": 200.0,
19
+ "sigma_min": 0.147,
20
+ "solver_order": 2,
21
+ "solver_p": null,
22
+ "solver_type": "bh2",
23
+ "steps_offset": 0,
24
+ "thresholding": false,
25
+ "time_shift_type": "exponential",
26
+ "timestep_spacing": "linspace",
27
+ "trained_betas": null,
28
+ "use_beta_sigmas": false,
29
+ "use_dynamic_shifting": false,
30
+ "use_exponential_sigmas": false,
31
+ "use_flow_sigmas": true,
32
+ "use_karras_sigmas": true
33
+ }
text_tokenizer/added_tokens.json ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</think>": 151668,
3
+ "</tool_call>": 151658,
4
+ "</tool_response>": 151666,
5
+ "<think>": 151667,
6
+ "<tool_call>": 151657,
7
+ "<tool_response>": 151665,
8
+ "<|box_end|>": 151649,
9
+ "<|box_start|>": 151648,
10
+ "<|endoftext|>": 151643,
11
+ "<|file_sep|>": 151664,
12
+ "<|fim_middle|>": 151660,
13
+ "<|fim_pad|>": 151662,
14
+ "<|fim_prefix|>": 151659,
15
+ "<|fim_suffix|>": 151661,
16
+ "<|im_end|>": 151645,
17
+ "<|im_start|>": 151644,
18
+ "<|image_pad|>": 151655,
19
+ "<|object_ref_end|>": 151647,
20
+ "<|object_ref_start|>": 151646,
21
+ "<|quad_end|>": 151651,
22
+ "<|quad_start|>": 151650,
23
+ "<|repo_name|>": 151663,
24
+ "<|video_pad|>": 151656,
25
+ "<|vision_end|>": 151653,
26
+ "<|vision_pad|>": 151654,
27
+ "<|vision_start|>": 151652
28
+ }
text_tokenizer/chat_template.jinja ADDED
@@ -0,0 +1,120 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0].role == 'system' %}
4
+ {%- if messages[0].content is string %}
5
+ {{- messages[0].content }}
6
+ {%- else %}
7
+ {%- for content in messages[0].content %}
8
+ {%- if 'text' in content %}
9
+ {{- content.text }}
10
+ {%- endif %}
11
+ {%- endfor %}
12
+ {%- endif %}
13
+ {{- '\n\n' }}
14
+ {%- endif %}
15
+ {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
16
+ {%- for tool in tools %}
17
+ {{- "\n" }}
18
+ {{- tool | tojson }}
19
+ {%- endfor %}
20
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
21
+ {%- else %}
22
+ {%- if messages[0].role == 'system' %}
23
+ {{- '<|im_start|>system\n' }}
24
+ {%- if messages[0].content is string %}
25
+ {{- messages[0].content }}
26
+ {%- else %}
27
+ {%- for content in messages[0].content %}
28
+ {%- if 'text' in content %}
29
+ {{- content.text }}
30
+ {%- endif %}
31
+ {%- endfor %}
32
+ {%- endif %}
33
+ {{- '<|im_end|>\n' }}
34
+ {%- endif %}
35
+ {%- endif %}
36
+ {%- set image_count = namespace(value=0) %}
37
+ {%- set video_count = namespace(value=0) %}
38
+ {%- for message in messages %}
39
+ {%- if message.role == "user" %}
40
+ {{- '<|im_start|>' + message.role + '\n' }}
41
+ {%- if message.content is string %}
42
+ {{- message.content }}
43
+ {%- else %}
44
+ {%- for content in message.content %}
45
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
46
+ {%- set image_count.value = image_count.value + 1 %}
47
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
48
+ <|vision_start|><|image_pad|><|vision_end|>
49
+ {%- elif content.type == 'video' or 'video' in content %}
50
+ {%- set video_count.value = video_count.value + 1 %}
51
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
52
+ <|vision_start|><|video_pad|><|vision_end|>
53
+ {%- elif 'text' in content %}
54
+ {{- content.text }}
55
+ {%- endif %}
56
+ {%- endfor %}
57
+ {%- endif %}
58
+ {{- '<|im_end|>\n' }}
59
+ {%- elif message.role == "assistant" %}
60
+ {{- '<|im_start|>' + message.role + '\n' }}
61
+ {%- if message.content is string %}
62
+ {{- message.content }}
63
+ {%- else %}
64
+ {%- for content_item in message.content %}
65
+ {%- if 'text' in content_item %}
66
+ {{- content_item.text }}
67
+ {%- endif %}
68
+ {%- endfor %}
69
+ {%- endif %}
70
+ {%- if message.tool_calls %}
71
+ {%- for tool_call in message.tool_calls %}
72
+ {%- if (loop.first and message.content) or (not loop.first) %}
73
+ {{- '\n' }}
74
+ {%- endif %}
75
+ {%- if tool_call.function %}
76
+ {%- set tool_call = tool_call.function %}
77
+ {%- endif %}
78
+ {{- '<tool_call>\n{"name": "' }}
79
+ {{- tool_call.name }}
80
+ {{- '", "arguments": ' }}
81
+ {%- if tool_call.arguments is string %}
82
+ {{- tool_call.arguments }}
83
+ {%- else %}
84
+ {{- tool_call.arguments | tojson }}
85
+ {%- endif %}
86
+ {{- '}\n</tool_call>' }}
87
+ {%- endfor %}
88
+ {%- endif %}
89
+ {{- '<|im_end|>\n' }}
90
+ {%- elif message.role == "tool" %}
91
+ {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
92
+ {{- '<|im_start|>user' }}
93
+ {%- endif %}
94
+ {{- '\n<tool_response>\n' }}
95
+ {%- if message.content is string %}
96
+ {{- message.content }}
97
+ {%- else %}
98
+ {%- for content in message.content %}
99
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
100
+ {%- set image_count.value = image_count.value + 1 %}
101
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
102
+ <|vision_start|><|image_pad|><|vision_end|>
103
+ {%- elif content.type == 'video' or 'video' in content %}
104
+ {%- set video_count.value = video_count.value + 1 %}
105
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
106
+ <|vision_start|><|video_pad|><|vision_end|>
107
+ {%- elif 'text' in content %}
108
+ {{- content.text }}
109
+ {%- endif %}
110
+ {%- endfor %}
111
+ {%- endif %}
112
+ {{- '\n</tool_response>' }}
113
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
114
+ {{- '<|im_end|>\n' }}
115
+ {%- endif %}
116
+ {%- endif %}
117
+ {%- endfor %}
118
+ {%- if add_generation_prompt %}
119
+ {{- '<|im_start|>assistant\n' }}
120
+ {%- endif %}
text_tokenizer/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
text_tokenizer/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|im_end|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
text_tokenizer/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:aeb13307a71acd8fe81861d94ad54ab689df773318809eed3cbe794b4492dae4
3
+ size 11422654
text_tokenizer/tokenizer_config.json ADDED
@@ -0,0 +1,239 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ },
181
+ "151665": {
182
+ "content": "<tool_response>",
183
+ "lstrip": false,
184
+ "normalized": false,
185
+ "rstrip": false,
186
+ "single_word": false,
187
+ "special": false
188
+ },
189
+ "151666": {
190
+ "content": "</tool_response>",
191
+ "lstrip": false,
192
+ "normalized": false,
193
+ "rstrip": false,
194
+ "single_word": false,
195
+ "special": false
196
+ },
197
+ "151667": {
198
+ "content": "<think>",
199
+ "lstrip": false,
200
+ "normalized": false,
201
+ "rstrip": false,
202
+ "single_word": false,
203
+ "special": false
204
+ },
205
+ "151668": {
206
+ "content": "</think>",
207
+ "lstrip": false,
208
+ "normalized": false,
209
+ "rstrip": false,
210
+ "single_word": false,
211
+ "special": false
212
+ }
213
+ },
214
+ "additional_special_tokens": [
215
+ "<|im_start|>",
216
+ "<|im_end|>",
217
+ "<|object_ref_start|>",
218
+ "<|object_ref_end|>",
219
+ "<|box_start|>",
220
+ "<|box_end|>",
221
+ "<|quad_start|>",
222
+ "<|quad_end|>",
223
+ "<|vision_start|>",
224
+ "<|vision_end|>",
225
+ "<|vision_pad|>",
226
+ "<|image_pad|>",
227
+ "<|video_pad|>"
228
+ ],
229
+ "bos_token": null,
230
+ "clean_up_tokenization_spaces": false,
231
+ "eos_token": "<|im_end|>",
232
+ "errors": "replace",
233
+ "extra_special_tokens": {},
234
+ "model_max_length": 262144,
235
+ "pad_token": "<|endoftext|>",
236
+ "split_special_tokens": false,
237
+ "tokenizer_class": "Qwen2Tokenizer",
238
+ "unk_token": null
239
+ }
text_tokenizer/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,239 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ },
181
+ "151665": {
182
+ "content": "<tool_response>",
183
+ "lstrip": false,
184
+ "normalized": false,
185
+ "rstrip": false,
186
+ "single_word": false,
187
+ "special": false
188
+ },
189
+ "151666": {
190
+ "content": "</tool_response>",
191
+ "lstrip": false,
192
+ "normalized": false,
193
+ "rstrip": false,
194
+ "single_word": false,
195
+ "special": false
196
+ },
197
+ "151667": {
198
+ "content": "<think>",
199
+ "lstrip": false,
200
+ "normalized": false,
201
+ "rstrip": false,
202
+ "single_word": false,
203
+ "special": false
204
+ },
205
+ "151668": {
206
+ "content": "</think>",
207
+ "lstrip": false,
208
+ "normalized": false,
209
+ "rstrip": false,
210
+ "single_word": false,
211
+ "special": false
212
+ }
213
+ },
214
+ "additional_special_tokens": [
215
+ "<|im_start|>",
216
+ "<|im_end|>",
217
+ "<|object_ref_start|>",
218
+ "<|object_ref_end|>",
219
+ "<|box_start|>",
220
+ "<|box_end|>",
221
+ "<|quad_start|>",
222
+ "<|quad_end|>",
223
+ "<|vision_start|>",
224
+ "<|vision_end|>",
225
+ "<|vision_pad|>",
226
+ "<|image_pad|>",
227
+ "<|video_pad|>"
228
+ ],
229
+ "bos_token": null,
230
+ "chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0].role == 'system' %}\n {%- if messages[0].content is string %}\n {{- messages[0].content }}\n {%- else %}\n {%- for content in messages[0].content %}\n {%- if 'text' in content %}\n {{- content.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {{- '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within <tools></tools> XML tags:\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\\n\\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\\n<tool_call>\\n{\\\"name\\\": <function-name>, \\\"arguments\\\": <args-json-object>}\\n</tool_call><|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0].content is string %}\n {{- messages[0].content }}\n {%- else %}\n {%- for content in messages[0].content %}\n {%- if 'text' in content %}\n {{- content.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- for message in messages %}\n {%- if message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' }}\n {%- if message.content is string %}\n {{- message.content }}\n {%- else %}\n {%- for content in message.content %}\n {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}\n <|vision_start|><|image_pad|><|vision_end|>\n {%- elif content.type == 'video' or 'video' in content %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}\n <|vision_start|><|video_pad|><|vision_end|>\n {%- elif 'text' in content %}\n {{- content.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"assistant\" %}\n {{- '<|im_start|>' + message.role + '\\n' }}\n {%- if message.content is string %}\n {{- message.content }}\n {%- else %}\n {%- for content_item in message.content %}\n {%- if 'text' in content_item %}\n {{- content_item.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {%- if message.tool_calls %}\n {%- for tool_call in message.tool_calls %}\n {%- if (loop.first and message.content) or (not loop.first) %}\n {{- '\\n' }}\n {%- endif %}\n {%- if tool_call.function %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '<tool_call>\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {%- if tool_call.arguments is string %}\n {{- tool_call.arguments }}\n {%- else %}\n {{- tool_call.arguments | tojson }}\n {%- endif %}\n {{- '}\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.first or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {%- if message.content is string %}\n {{- message.content }}\n {%- else %}\n {%- for content in message.content %}\n {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}\n <|vision_start|><|image_pad|><|vision_end|>\n {%- elif content.type == 'video' or 'video' in content %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}\n <|vision_start|><|video_pad|><|vision_end|>\n {%- elif 'text' in content %}\n {{- content.text }}\n {%- endif %}\n {%- endfor %}\n {%- endif %}\n {{- '\\n</tool_response>' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n",
231
+ "clean_up_tokenization_spaces": false,
232
+ "eos_token": "<|im_end|>",
233
+ "errors": "replace",
234
+ "model_max_length": 262144,
235
+ "pad_token": "<|endoftext|>",
236
+ "split_special_tokens": false,
237
+ "tokenizer_class": "Qwen2Tokenizer",
238
+ "unk_token": null
239
+ }
transformer/config.json ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "Cosmos3OmniTransformer",
3
+ "_diffusers_version": "0.37.1",
4
+ "action_dim": 64,
5
+ "action_gen": true,
6
+ "attention_bias": false,
7
+ "attention_dropout": 0.0,
8
+ "base_fps": 24,
9
+ "dtype": "bfloat16",
10
+ "enable_fps_modulation": true,
11
+ "freeze_und": false,
12
+ "head_dim": 128,
13
+ "hidden_act": "silu",
14
+ "hidden_size": 4096,
15
+ "initializer_range": 0.02,
16
+ "intermediate_size": 12288,
17
+ "joint_attn_implementation": "two_way",
18
+ "latent_channel": 48,
19
+ "latent_patch_size": 2,
20
+ "max_action_dim": 64,
21
+ "max_position_embeddings": 262144,
22
+ "model_type": "qwen3_vl_text",
23
+ "num_attention_heads": 32,
24
+ "num_embodiment_domains": 32,
25
+ "num_hidden_layers": 36,
26
+ "num_key_value_heads": 8,
27
+ "patch_latent_dim": 192,
28
+ "position_embedding_type": "unified_3d_mrope",
29
+ "qk_norm": false,
30
+ "qk_norm_for_diffusion": true,
31
+ "qk_norm_for_text": true,
32
+ "rms_norm_eps": 1e-06,
33
+ "rope_scaling": {
34
+ "mrope_interleaved": true,
35
+ "mrope_section": [
36
+ 24,
37
+ 20,
38
+ 20
39
+ ],
40
+ "rope_type": "default"
41
+ },
42
+ "rope_theta": 5000000,
43
+ "sound_dim": null,
44
+ "sound_gen": false,
45
+ "sound_latent_fps": 25,
46
+ "temporal_compression_factor_sound": 1,
47
+ "timestep_scale": 0.001,
48
+ "unified_3d_mrope_reset_spatial_ids": true,
49
+ "unified_3d_mrope_temporal_modality_margin": 15000,
50
+ "use_cache": true,
51
+ "use_moe": true,
52
+ "video_temporal_causal": false,
53
+ "vocab_size": 151936
54
+ }
transformer/diffusion_pytorch_model-00001-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f260af4e9c877c22ba9a44246767ff4a0bc9cd8f62fd6d683bedfcac75378ddd
3
+ size 4902249008
transformer/diffusion_pytorch_model-00002-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bce7ee18cc05ad291f57095d2705cb5ed26d6eca4469a44a0c40f8d507d724e9
3
+ size 4999863656
transformer/diffusion_pytorch_model-00003-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:885c1fc451c28ca34809df3b975b63f58adbaad29da53d6c4ad33613135526c2
3
+ size 4932719448
transformer/diffusion_pytorch_model-00004-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3a2639189fb96cdf40fba25be3ecee60a42a27431daccdd40f049bbd37ff7244
3
+ size 4983084656
transformer/diffusion_pytorch_model-00005-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:067fca5fbee766cd6828c96478a7ab93ac4bd9595662cae71787420ba4f27f14
3
+ size 4949498576
transformer/diffusion_pytorch_model-00006-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c3e56f83740a353e44f8342c997fe3cffdcb5cc801d7e54c9bd17c7c016f8a7c
3
+ size 4261644464
transformer/diffusion_pytorch_model-00007-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5e23d7dbafab591bce1543b55f3ade34231085261d8a1754dfc26822bc691566
3
+ size 1317303952
transformer/diffusion_pytorch_model.safetensors.index.json ADDED
@@ -0,0 +1,816 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_size": 30346273152
4
+ },
5
+ "weight_map": {
6
+ "action_modality_embed": "diffusion_pytorch_model-00001-of-00007.safetensors",
7
+ "action_proj_in.bias.weight": "diffusion_pytorch_model-00007-of-00007.safetensors",
8
+ "action_proj_in.fc.weight": "diffusion_pytorch_model-00007-of-00007.safetensors",
9
+ "action_proj_out.bias.weight": "diffusion_pytorch_model-00007-of-00007.safetensors",
10
+ "action_proj_out.fc.weight": "diffusion_pytorch_model-00007-of-00007.safetensors",
11
+ "embed_tokens.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
12
+ "layers.0.input_layernorm.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
13
+ "layers.0.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
14
+ "layers.0.mlp.down_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
15
+ "layers.0.mlp.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
16
+ "layers.0.mlp.up_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
17
+ "layers.0.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
18
+ "layers.0.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
19
+ "layers.0.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
20
+ "layers.0.post_attention_layernorm.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
21
+ "layers.0.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
22
+ "layers.0.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
23
+ "layers.0.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
24
+ "layers.0.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
25
+ "layers.0.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
26
+ "layers.0.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
27
+ "layers.0.self_attn.norm_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
28
+ "layers.0.self_attn.norm_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
29
+ "layers.0.self_attn.to_add_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
30
+ "layers.0.self_attn.to_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
31
+ "layers.0.self_attn.to_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
32
+ "layers.0.self_attn.to_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
33
+ "layers.0.self_attn.to_v.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
34
+ "layers.1.input_layernorm.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
35
+ "layers.1.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
36
+ "layers.1.mlp.down_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
37
+ "layers.1.mlp.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
38
+ "layers.1.mlp.up_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
39
+ "layers.1.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
40
+ "layers.1.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
41
+ "layers.1.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
42
+ "layers.1.post_attention_layernorm.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
43
+ "layers.1.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
44
+ "layers.1.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
45
+ "layers.1.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
46
+ "layers.1.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
47
+ "layers.1.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
48
+ "layers.1.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
49
+ "layers.1.self_attn.norm_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
50
+ "layers.1.self_attn.norm_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
51
+ "layers.1.self_attn.to_add_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
52
+ "layers.1.self_attn.to_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
53
+ "layers.1.self_attn.to_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
54
+ "layers.1.self_attn.to_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
55
+ "layers.1.self_attn.to_v.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
56
+ "layers.10.input_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
57
+ "layers.10.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
58
+ "layers.10.mlp.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
59
+ "layers.10.mlp.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
60
+ "layers.10.mlp.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
61
+ "layers.10.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
62
+ "layers.10.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
63
+ "layers.10.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
64
+ "layers.10.post_attention_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
65
+ "layers.10.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
66
+ "layers.10.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
67
+ "layers.10.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
68
+ "layers.10.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
69
+ "layers.10.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
70
+ "layers.10.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
71
+ "layers.10.self_attn.norm_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
72
+ "layers.10.self_attn.norm_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
73
+ "layers.10.self_attn.to_add_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
74
+ "layers.10.self_attn.to_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
75
+ "layers.10.self_attn.to_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
76
+ "layers.10.self_attn.to_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
77
+ "layers.10.self_attn.to_v.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
78
+ "layers.11.input_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
79
+ "layers.11.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
80
+ "layers.11.mlp.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
81
+ "layers.11.mlp.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
82
+ "layers.11.mlp.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
83
+ "layers.11.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
84
+ "layers.11.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
85
+ "layers.11.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
86
+ "layers.11.post_attention_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
87
+ "layers.11.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
88
+ "layers.11.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
89
+ "layers.11.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
90
+ "layers.11.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
91
+ "layers.11.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
92
+ "layers.11.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
93
+ "layers.11.self_attn.norm_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
94
+ "layers.11.self_attn.norm_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
95
+ "layers.11.self_attn.to_add_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
96
+ "layers.11.self_attn.to_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
97
+ "layers.11.self_attn.to_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
98
+ "layers.11.self_attn.to_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
99
+ "layers.11.self_attn.to_v.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
100
+ "layers.12.input_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
101
+ "layers.12.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
102
+ "layers.12.mlp.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
103
+ "layers.12.mlp.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
104
+ "layers.12.mlp.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
105
+ "layers.12.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
106
+ "layers.12.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
107
+ "layers.12.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
108
+ "layers.12.post_attention_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
109
+ "layers.12.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
110
+ "layers.12.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
111
+ "layers.12.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
112
+ "layers.12.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
113
+ "layers.12.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
114
+ "layers.12.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
115
+ "layers.12.self_attn.norm_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
116
+ "layers.12.self_attn.norm_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
117
+ "layers.12.self_attn.to_add_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
118
+ "layers.12.self_attn.to_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
119
+ "layers.12.self_attn.to_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
120
+ "layers.12.self_attn.to_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
121
+ "layers.12.self_attn.to_v.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
122
+ "layers.13.input_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
123
+ "layers.13.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
124
+ "layers.13.mlp.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
125
+ "layers.13.mlp.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
126
+ "layers.13.mlp.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
127
+ "layers.13.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
128
+ "layers.13.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
129
+ "layers.13.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
130
+ "layers.13.post_attention_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
131
+ "layers.13.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
132
+ "layers.13.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
133
+ "layers.13.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
134
+ "layers.13.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
135
+ "layers.13.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
136
+ "layers.13.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
137
+ "layers.13.self_attn.norm_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
138
+ "layers.13.self_attn.norm_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
139
+ "layers.13.self_attn.to_add_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
140
+ "layers.13.self_attn.to_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
141
+ "layers.13.self_attn.to_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
142
+ "layers.13.self_attn.to_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
143
+ "layers.13.self_attn.to_v.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
144
+ "layers.14.input_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
145
+ "layers.14.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
146
+ "layers.14.mlp.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
147
+ "layers.14.mlp.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
148
+ "layers.14.mlp.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
149
+ "layers.14.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
150
+ "layers.14.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
151
+ "layers.14.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
152
+ "layers.14.post_attention_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
153
+ "layers.14.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
154
+ "layers.14.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
155
+ "layers.14.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
156
+ "layers.14.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
157
+ "layers.14.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
158
+ "layers.14.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
159
+ "layers.14.self_attn.norm_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
160
+ "layers.14.self_attn.norm_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
161
+ "layers.14.self_attn.to_add_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
162
+ "layers.14.self_attn.to_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
163
+ "layers.14.self_attn.to_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
164
+ "layers.14.self_attn.to_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
165
+ "layers.14.self_attn.to_v.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
166
+ "layers.15.input_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
167
+ "layers.15.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
168
+ "layers.15.mlp.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
169
+ "layers.15.mlp.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
170
+ "layers.15.mlp.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
171
+ "layers.15.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
172
+ "layers.15.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
173
+ "layers.15.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
174
+ "layers.15.post_attention_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
175
+ "layers.15.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
176
+ "layers.15.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
177
+ "layers.15.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
178
+ "layers.15.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
179
+ "layers.15.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
180
+ "layers.15.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
181
+ "layers.15.self_attn.norm_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
182
+ "layers.15.self_attn.norm_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
183
+ "layers.15.self_attn.to_add_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
184
+ "layers.15.self_attn.to_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
185
+ "layers.15.self_attn.to_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
186
+ "layers.15.self_attn.to_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
187
+ "layers.15.self_attn.to_v.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
188
+ "layers.16.input_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
189
+ "layers.16.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
190
+ "layers.16.mlp.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
191
+ "layers.16.mlp.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
192
+ "layers.16.mlp.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
193
+ "layers.16.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
194
+ "layers.16.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
195
+ "layers.16.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
196
+ "layers.16.post_attention_layernorm.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
197
+ "layers.16.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
198
+ "layers.16.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
199
+ "layers.16.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
200
+ "layers.16.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
201
+ "layers.16.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
202
+ "layers.16.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
203
+ "layers.16.self_attn.norm_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
204
+ "layers.16.self_attn.norm_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
205
+ "layers.16.self_attn.to_add_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
206
+ "layers.16.self_attn.to_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
207
+ "layers.16.self_attn.to_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
208
+ "layers.16.self_attn.to_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
209
+ "layers.16.self_attn.to_v.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
210
+ "layers.17.input_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
211
+ "layers.17.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
212
+ "layers.17.mlp.down_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
213
+ "layers.17.mlp.gate_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
214
+ "layers.17.mlp.up_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
215
+ "layers.17.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
216
+ "layers.17.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
217
+ "layers.17.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
218
+ "layers.17.post_attention_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
219
+ "layers.17.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
220
+ "layers.17.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
221
+ "layers.17.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
222
+ "layers.17.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
223
+ "layers.17.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
224
+ "layers.17.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
225
+ "layers.17.self_attn.norm_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
226
+ "layers.17.self_attn.norm_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
227
+ "layers.17.self_attn.to_add_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
228
+ "layers.17.self_attn.to_k.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
229
+ "layers.17.self_attn.to_out.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
230
+ "layers.17.self_attn.to_q.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
231
+ "layers.17.self_attn.to_v.weight": "diffusion_pytorch_model-00003-of-00007.safetensors",
232
+ "layers.18.input_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
233
+ "layers.18.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
234
+ "layers.18.mlp.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
235
+ "layers.18.mlp.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
236
+ "layers.18.mlp.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
237
+ "layers.18.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
238
+ "layers.18.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
239
+ "layers.18.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
240
+ "layers.18.post_attention_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
241
+ "layers.18.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
242
+ "layers.18.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
243
+ "layers.18.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
244
+ "layers.18.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
245
+ "layers.18.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
246
+ "layers.18.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
247
+ "layers.18.self_attn.norm_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
248
+ "layers.18.self_attn.norm_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
249
+ "layers.18.self_attn.to_add_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
250
+ "layers.18.self_attn.to_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
251
+ "layers.18.self_attn.to_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
252
+ "layers.18.self_attn.to_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
253
+ "layers.18.self_attn.to_v.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
254
+ "layers.19.input_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
255
+ "layers.19.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
256
+ "layers.19.mlp.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
257
+ "layers.19.mlp.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
258
+ "layers.19.mlp.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
259
+ "layers.19.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
260
+ "layers.19.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
261
+ "layers.19.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
262
+ "layers.19.post_attention_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
263
+ "layers.19.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
264
+ "layers.19.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
265
+ "layers.19.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
266
+ "layers.19.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
267
+ "layers.19.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
268
+ "layers.19.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
269
+ "layers.19.self_attn.norm_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
270
+ "layers.19.self_attn.norm_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
271
+ "layers.19.self_attn.to_add_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
272
+ "layers.19.self_attn.to_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
273
+ "layers.19.self_attn.to_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
274
+ "layers.19.self_attn.to_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
275
+ "layers.19.self_attn.to_v.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
276
+ "layers.2.input_layernorm.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
277
+ "layers.2.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
278
+ "layers.2.mlp.down_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
279
+ "layers.2.mlp.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
280
+ "layers.2.mlp.up_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
281
+ "layers.2.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
282
+ "layers.2.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
283
+ "layers.2.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
284
+ "layers.2.post_attention_layernorm.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
285
+ "layers.2.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
286
+ "layers.2.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
287
+ "layers.2.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
288
+ "layers.2.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
289
+ "layers.2.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
290
+ "layers.2.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
291
+ "layers.2.self_attn.norm_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
292
+ "layers.2.self_attn.norm_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
293
+ "layers.2.self_attn.to_add_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
294
+ "layers.2.self_attn.to_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
295
+ "layers.2.self_attn.to_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
296
+ "layers.2.self_attn.to_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
297
+ "layers.2.self_attn.to_v.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
298
+ "layers.20.input_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
299
+ "layers.20.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
300
+ "layers.20.mlp.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
301
+ "layers.20.mlp.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
302
+ "layers.20.mlp.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
303
+ "layers.20.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
304
+ "layers.20.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
305
+ "layers.20.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
306
+ "layers.20.post_attention_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
307
+ "layers.20.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
308
+ "layers.20.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
309
+ "layers.20.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
310
+ "layers.20.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
311
+ "layers.20.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
312
+ "layers.20.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
313
+ "layers.20.self_attn.norm_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
314
+ "layers.20.self_attn.norm_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
315
+ "layers.20.self_attn.to_add_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
316
+ "layers.20.self_attn.to_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
317
+ "layers.20.self_attn.to_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
318
+ "layers.20.self_attn.to_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
319
+ "layers.20.self_attn.to_v.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
320
+ "layers.21.input_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
321
+ "layers.21.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
322
+ "layers.21.mlp.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
323
+ "layers.21.mlp.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
324
+ "layers.21.mlp.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
325
+ "layers.21.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
326
+ "layers.21.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
327
+ "layers.21.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
328
+ "layers.21.post_attention_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
329
+ "layers.21.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
330
+ "layers.21.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
331
+ "layers.21.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
332
+ "layers.21.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
333
+ "layers.21.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
334
+ "layers.21.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
335
+ "layers.21.self_attn.norm_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
336
+ "layers.21.self_attn.norm_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
337
+ "layers.21.self_attn.to_add_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
338
+ "layers.21.self_attn.to_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
339
+ "layers.21.self_attn.to_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
340
+ "layers.21.self_attn.to_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
341
+ "layers.21.self_attn.to_v.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
342
+ "layers.22.input_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
343
+ "layers.22.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
344
+ "layers.22.mlp.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
345
+ "layers.22.mlp.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
346
+ "layers.22.mlp.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
347
+ "layers.22.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
348
+ "layers.22.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
349
+ "layers.22.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
350
+ "layers.22.post_attention_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
351
+ "layers.22.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
352
+ "layers.22.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
353
+ "layers.22.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
354
+ "layers.22.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
355
+ "layers.22.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
356
+ "layers.22.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
357
+ "layers.22.self_attn.norm_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
358
+ "layers.22.self_attn.norm_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
359
+ "layers.22.self_attn.to_add_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
360
+ "layers.22.self_attn.to_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
361
+ "layers.22.self_attn.to_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
362
+ "layers.22.self_attn.to_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
363
+ "layers.22.self_attn.to_v.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
364
+ "layers.23.input_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
365
+ "layers.23.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
366
+ "layers.23.mlp.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
367
+ "layers.23.mlp.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
368
+ "layers.23.mlp.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
369
+ "layers.23.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
370
+ "layers.23.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
371
+ "layers.23.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
372
+ "layers.23.post_attention_layernorm.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
373
+ "layers.23.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
374
+ "layers.23.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
375
+ "layers.23.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
376
+ "layers.23.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
377
+ "layers.23.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
378
+ "layers.23.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
379
+ "layers.23.self_attn.norm_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
380
+ "layers.23.self_attn.norm_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
381
+ "layers.23.self_attn.to_add_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
382
+ "layers.23.self_attn.to_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
383
+ "layers.23.self_attn.to_out.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
384
+ "layers.23.self_attn.to_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
385
+ "layers.23.self_attn.to_v.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
386
+ "layers.24.input_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
387
+ "layers.24.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
388
+ "layers.24.mlp.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
389
+ "layers.24.mlp.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
390
+ "layers.24.mlp.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
391
+ "layers.24.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
392
+ "layers.24.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
393
+ "layers.24.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
394
+ "layers.24.post_attention_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
395
+ "layers.24.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
396
+ "layers.24.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
397
+ "layers.24.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
398
+ "layers.24.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
399
+ "layers.24.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
400
+ "layers.24.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
401
+ "layers.24.self_attn.norm_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
402
+ "layers.24.self_attn.norm_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
403
+ "layers.24.self_attn.to_add_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
404
+ "layers.24.self_attn.to_k.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
405
+ "layers.24.self_attn.to_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
406
+ "layers.24.self_attn.to_q.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
407
+ "layers.24.self_attn.to_v.weight": "diffusion_pytorch_model-00004-of-00007.safetensors",
408
+ "layers.25.input_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
409
+ "layers.25.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
410
+ "layers.25.mlp.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
411
+ "layers.25.mlp.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
412
+ "layers.25.mlp.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
413
+ "layers.25.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
414
+ "layers.25.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
415
+ "layers.25.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
416
+ "layers.25.post_attention_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
417
+ "layers.25.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
418
+ "layers.25.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
419
+ "layers.25.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
420
+ "layers.25.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
421
+ "layers.25.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
422
+ "layers.25.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
423
+ "layers.25.self_attn.norm_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
424
+ "layers.25.self_attn.norm_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
425
+ "layers.25.self_attn.to_add_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
426
+ "layers.25.self_attn.to_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
427
+ "layers.25.self_attn.to_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
428
+ "layers.25.self_attn.to_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
429
+ "layers.25.self_attn.to_v.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
430
+ "layers.26.input_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
431
+ "layers.26.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
432
+ "layers.26.mlp.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
433
+ "layers.26.mlp.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
434
+ "layers.26.mlp.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
435
+ "layers.26.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
436
+ "layers.26.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
437
+ "layers.26.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
438
+ "layers.26.post_attention_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
439
+ "layers.26.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
440
+ "layers.26.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
441
+ "layers.26.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
442
+ "layers.26.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
443
+ "layers.26.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
444
+ "layers.26.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
445
+ "layers.26.self_attn.norm_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
446
+ "layers.26.self_attn.norm_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
447
+ "layers.26.self_attn.to_add_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
448
+ "layers.26.self_attn.to_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
449
+ "layers.26.self_attn.to_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
450
+ "layers.26.self_attn.to_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
451
+ "layers.26.self_attn.to_v.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
452
+ "layers.27.input_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
453
+ "layers.27.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
454
+ "layers.27.mlp.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
455
+ "layers.27.mlp.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
456
+ "layers.27.mlp.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
457
+ "layers.27.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
458
+ "layers.27.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
459
+ "layers.27.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
460
+ "layers.27.post_attention_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
461
+ "layers.27.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
462
+ "layers.27.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
463
+ "layers.27.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
464
+ "layers.27.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
465
+ "layers.27.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
466
+ "layers.27.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
467
+ "layers.27.self_attn.norm_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
468
+ "layers.27.self_attn.norm_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
469
+ "layers.27.self_attn.to_add_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
470
+ "layers.27.self_attn.to_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
471
+ "layers.27.self_attn.to_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
472
+ "layers.27.self_attn.to_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
473
+ "layers.27.self_attn.to_v.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
474
+ "layers.28.input_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
475
+ "layers.28.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
476
+ "layers.28.mlp.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
477
+ "layers.28.mlp.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
478
+ "layers.28.mlp.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
479
+ "layers.28.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
480
+ "layers.28.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
481
+ "layers.28.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
482
+ "layers.28.post_attention_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
483
+ "layers.28.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
484
+ "layers.28.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
485
+ "layers.28.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
486
+ "layers.28.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
487
+ "layers.28.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
488
+ "layers.28.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
489
+ "layers.28.self_attn.norm_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
490
+ "layers.28.self_attn.norm_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
491
+ "layers.28.self_attn.to_add_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
492
+ "layers.28.self_attn.to_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
493
+ "layers.28.self_attn.to_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
494
+ "layers.28.self_attn.to_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
495
+ "layers.28.self_attn.to_v.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
496
+ "layers.29.input_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
497
+ "layers.29.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
498
+ "layers.29.mlp.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
499
+ "layers.29.mlp.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
500
+ "layers.29.mlp.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
501
+ "layers.29.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
502
+ "layers.29.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
503
+ "layers.29.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
504
+ "layers.29.post_attention_layernorm.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
505
+ "layers.29.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
506
+ "layers.29.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
507
+ "layers.29.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
508
+ "layers.29.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
509
+ "layers.29.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
510
+ "layers.29.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
511
+ "layers.29.self_attn.norm_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
512
+ "layers.29.self_attn.norm_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
513
+ "layers.29.self_attn.to_add_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
514
+ "layers.29.self_attn.to_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
515
+ "layers.29.self_attn.to_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
516
+ "layers.29.self_attn.to_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
517
+ "layers.29.self_attn.to_v.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
518
+ "layers.3.input_layernorm.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
519
+ "layers.3.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
520
+ "layers.3.mlp.down_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
521
+ "layers.3.mlp.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
522
+ "layers.3.mlp.up_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
523
+ "layers.3.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
524
+ "layers.3.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
525
+ "layers.3.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
526
+ "layers.3.post_attention_layernorm.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
527
+ "layers.3.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
528
+ "layers.3.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
529
+ "layers.3.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
530
+ "layers.3.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
531
+ "layers.3.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
532
+ "layers.3.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
533
+ "layers.3.self_attn.norm_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
534
+ "layers.3.self_attn.norm_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
535
+ "layers.3.self_attn.to_add_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
536
+ "layers.3.self_attn.to_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
537
+ "layers.3.self_attn.to_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
538
+ "layers.3.self_attn.to_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
539
+ "layers.3.self_attn.to_v.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
540
+ "layers.30.input_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
541
+ "layers.30.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
542
+ "layers.30.mlp.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
543
+ "layers.30.mlp.gate_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
544
+ "layers.30.mlp.up_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
545
+ "layers.30.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
546
+ "layers.30.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
547
+ "layers.30.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
548
+ "layers.30.post_attention_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
549
+ "layers.30.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
550
+ "layers.30.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
551
+ "layers.30.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
552
+ "layers.30.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
553
+ "layers.30.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
554
+ "layers.30.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
555
+ "layers.30.self_attn.norm_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
556
+ "layers.30.self_attn.norm_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
557
+ "layers.30.self_attn.to_add_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
558
+ "layers.30.self_attn.to_k.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
559
+ "layers.30.self_attn.to_out.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
560
+ "layers.30.self_attn.to_q.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
561
+ "layers.30.self_attn.to_v.weight": "diffusion_pytorch_model-00005-of-00007.safetensors",
562
+ "layers.31.input_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
563
+ "layers.31.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
564
+ "layers.31.mlp.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
565
+ "layers.31.mlp.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
566
+ "layers.31.mlp.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
567
+ "layers.31.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
568
+ "layers.31.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
569
+ "layers.31.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
570
+ "layers.31.post_attention_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
571
+ "layers.31.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
572
+ "layers.31.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
573
+ "layers.31.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
574
+ "layers.31.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
575
+ "layers.31.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
576
+ "layers.31.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
577
+ "layers.31.self_attn.norm_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
578
+ "layers.31.self_attn.norm_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
579
+ "layers.31.self_attn.to_add_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
580
+ "layers.31.self_attn.to_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
581
+ "layers.31.self_attn.to_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
582
+ "layers.31.self_attn.to_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
583
+ "layers.31.self_attn.to_v.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
584
+ "layers.32.input_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
585
+ "layers.32.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
586
+ "layers.32.mlp.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
587
+ "layers.32.mlp.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
588
+ "layers.32.mlp.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
589
+ "layers.32.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
590
+ "layers.32.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
591
+ "layers.32.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
592
+ "layers.32.post_attention_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
593
+ "layers.32.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
594
+ "layers.32.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
595
+ "layers.32.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
596
+ "layers.32.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
597
+ "layers.32.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
598
+ "layers.32.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
599
+ "layers.32.self_attn.norm_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
600
+ "layers.32.self_attn.norm_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
601
+ "layers.32.self_attn.to_add_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
602
+ "layers.32.self_attn.to_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
603
+ "layers.32.self_attn.to_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
604
+ "layers.32.self_attn.to_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
605
+ "layers.32.self_attn.to_v.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
606
+ "layers.33.input_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
607
+ "layers.33.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
608
+ "layers.33.mlp.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
609
+ "layers.33.mlp.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
610
+ "layers.33.mlp.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
611
+ "layers.33.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
612
+ "layers.33.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
613
+ "layers.33.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
614
+ "layers.33.post_attention_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
615
+ "layers.33.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
616
+ "layers.33.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
617
+ "layers.33.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
618
+ "layers.33.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
619
+ "layers.33.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
620
+ "layers.33.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
621
+ "layers.33.self_attn.norm_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
622
+ "layers.33.self_attn.norm_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
623
+ "layers.33.self_attn.to_add_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
624
+ "layers.33.self_attn.to_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
625
+ "layers.33.self_attn.to_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
626
+ "layers.33.self_attn.to_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
627
+ "layers.33.self_attn.to_v.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
628
+ "layers.34.input_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
629
+ "layers.34.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
630
+ "layers.34.mlp.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
631
+ "layers.34.mlp.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
632
+ "layers.34.mlp.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
633
+ "layers.34.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
634
+ "layers.34.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
635
+ "layers.34.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
636
+ "layers.34.post_attention_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
637
+ "layers.34.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
638
+ "layers.34.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
639
+ "layers.34.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
640
+ "layers.34.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
641
+ "layers.34.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
642
+ "layers.34.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
643
+ "layers.34.self_attn.norm_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
644
+ "layers.34.self_attn.norm_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
645
+ "layers.34.self_attn.to_add_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
646
+ "layers.34.self_attn.to_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
647
+ "layers.34.self_attn.to_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
648
+ "layers.34.self_attn.to_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
649
+ "layers.34.self_attn.to_v.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
650
+ "layers.35.input_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
651
+ "layers.35.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
652
+ "layers.35.mlp.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
653
+ "layers.35.mlp.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
654
+ "layers.35.mlp.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
655
+ "layers.35.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
656
+ "layers.35.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
657
+ "layers.35.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
658
+ "layers.35.post_attention_layernorm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
659
+ "layers.35.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
660
+ "layers.35.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
661
+ "layers.35.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
662
+ "layers.35.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
663
+ "layers.35.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
664
+ "layers.35.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
665
+ "layers.35.self_attn.norm_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
666
+ "layers.35.self_attn.norm_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
667
+ "layers.35.self_attn.to_add_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
668
+ "layers.35.self_attn.to_k.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
669
+ "layers.35.self_attn.to_out.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
670
+ "layers.35.self_attn.to_q.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
671
+ "layers.35.self_attn.to_v.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
672
+ "layers.4.input_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
673
+ "layers.4.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
674
+ "layers.4.mlp.down_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
675
+ "layers.4.mlp.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
676
+ "layers.4.mlp.up_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
677
+ "layers.4.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
678
+ "layers.4.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
679
+ "layers.4.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
680
+ "layers.4.post_attention_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
681
+ "layers.4.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
682
+ "layers.4.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
683
+ "layers.4.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
684
+ "layers.4.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
685
+ "layers.4.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
686
+ "layers.4.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
687
+ "layers.4.self_attn.norm_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
688
+ "layers.4.self_attn.norm_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
689
+ "layers.4.self_attn.to_add_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
690
+ "layers.4.self_attn.to_k.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
691
+ "layers.4.self_attn.to_out.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
692
+ "layers.4.self_attn.to_q.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
693
+ "layers.4.self_attn.to_v.weight": "diffusion_pytorch_model-00001-of-00007.safetensors",
694
+ "layers.5.input_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
695
+ "layers.5.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
696
+ "layers.5.mlp.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
697
+ "layers.5.mlp.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
698
+ "layers.5.mlp.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
699
+ "layers.5.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
700
+ "layers.5.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
701
+ "layers.5.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
702
+ "layers.5.post_attention_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
703
+ "layers.5.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
704
+ "layers.5.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
705
+ "layers.5.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
706
+ "layers.5.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
707
+ "layers.5.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
708
+ "layers.5.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
709
+ "layers.5.self_attn.norm_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
710
+ "layers.5.self_attn.norm_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
711
+ "layers.5.self_attn.to_add_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
712
+ "layers.5.self_attn.to_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
713
+ "layers.5.self_attn.to_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
714
+ "layers.5.self_attn.to_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
715
+ "layers.5.self_attn.to_v.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
716
+ "layers.6.input_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
717
+ "layers.6.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
718
+ "layers.6.mlp.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
719
+ "layers.6.mlp.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
720
+ "layers.6.mlp.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
721
+ "layers.6.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
722
+ "layers.6.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
723
+ "layers.6.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
724
+ "layers.6.post_attention_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
725
+ "layers.6.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
726
+ "layers.6.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
727
+ "layers.6.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
728
+ "layers.6.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
729
+ "layers.6.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
730
+ "layers.6.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
731
+ "layers.6.self_attn.norm_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
732
+ "layers.6.self_attn.norm_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
733
+ "layers.6.self_attn.to_add_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
734
+ "layers.6.self_attn.to_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
735
+ "layers.6.self_attn.to_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
736
+ "layers.6.self_attn.to_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
737
+ "layers.6.self_attn.to_v.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
738
+ "layers.7.input_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
739
+ "layers.7.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
740
+ "layers.7.mlp.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
741
+ "layers.7.mlp.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
742
+ "layers.7.mlp.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
743
+ "layers.7.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
744
+ "layers.7.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
745
+ "layers.7.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
746
+ "layers.7.post_attention_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
747
+ "layers.7.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
748
+ "layers.7.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
749
+ "layers.7.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
750
+ "layers.7.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
751
+ "layers.7.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
752
+ "layers.7.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
753
+ "layers.7.self_attn.norm_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
754
+ "layers.7.self_attn.norm_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
755
+ "layers.7.self_attn.to_add_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
756
+ "layers.7.self_attn.to_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
757
+ "layers.7.self_attn.to_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
758
+ "layers.7.self_attn.to_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
759
+ "layers.7.self_attn.to_v.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
760
+ "layers.8.input_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
761
+ "layers.8.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
762
+ "layers.8.mlp.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
763
+ "layers.8.mlp.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
764
+ "layers.8.mlp.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
765
+ "layers.8.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
766
+ "layers.8.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
767
+ "layers.8.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
768
+ "layers.8.post_attention_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
769
+ "layers.8.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
770
+ "layers.8.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
771
+ "layers.8.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
772
+ "layers.8.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
773
+ "layers.8.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
774
+ "layers.8.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
775
+ "layers.8.self_attn.norm_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
776
+ "layers.8.self_attn.norm_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
777
+ "layers.8.self_attn.to_add_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
778
+ "layers.8.self_attn.to_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
779
+ "layers.8.self_attn.to_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
780
+ "layers.8.self_attn.to_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
781
+ "layers.8.self_attn.to_v.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
782
+ "layers.9.input_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
783
+ "layers.9.input_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
784
+ "layers.9.mlp.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
785
+ "layers.9.mlp.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
786
+ "layers.9.mlp.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
787
+ "layers.9.mlp_moe_gen.down_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
788
+ "layers.9.mlp_moe_gen.gate_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
789
+ "layers.9.mlp_moe_gen.up_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
790
+ "layers.9.post_attention_layernorm.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
791
+ "layers.9.post_attention_layernorm_moe_gen.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
792
+ "layers.9.self_attn.add_k_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
793
+ "layers.9.self_attn.add_q_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
794
+ "layers.9.self_attn.add_v_proj.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
795
+ "layers.9.self_attn.norm_added_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
796
+ "layers.9.self_attn.norm_added_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
797
+ "layers.9.self_attn.norm_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
798
+ "layers.9.self_attn.norm_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
799
+ "layers.9.self_attn.to_add_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
800
+ "layers.9.self_attn.to_k.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
801
+ "layers.9.self_attn.to_out.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
802
+ "layers.9.self_attn.to_q.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
803
+ "layers.9.self_attn.to_v.weight": "diffusion_pytorch_model-00002-of-00007.safetensors",
804
+ "lm_head.weight": "diffusion_pytorch_model-00007-of-00007.safetensors",
805
+ "norm.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
806
+ "norm_moe_gen.weight": "diffusion_pytorch_model-00006-of-00007.safetensors",
807
+ "proj_in.bias": "diffusion_pytorch_model-00007-of-00007.safetensors",
808
+ "proj_in.weight": "diffusion_pytorch_model-00007-of-00007.safetensors",
809
+ "proj_out.bias": "diffusion_pytorch_model-00007-of-00007.safetensors",
810
+ "proj_out.weight": "diffusion_pytorch_model-00007-of-00007.safetensors",
811
+ "time_embedder.linear_1.bias": "diffusion_pytorch_model-00007-of-00007.safetensors",
812
+ "time_embedder.linear_1.weight": "diffusion_pytorch_model-00007-of-00007.safetensors",
813
+ "time_embedder.linear_2.bias": "diffusion_pytorch_model-00007-of-00007.safetensors",
814
+ "time_embedder.linear_2.weight": "diffusion_pytorch_model-00007-of-00007.safetensors"
815
+ }
816
+ }
vae/config.json ADDED
@@ -0,0 +1,129 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "AutoencoderKLWan",
3
+ "_diffusers_version": "0.37.1",
4
+ "_name_or_path": "Wan-AI/Wan2.2-TI2V-5B-Diffusers",
5
+ "attn_scales": [],
6
+ "base_dim": 160,
7
+ "clip_output": false,
8
+ "decoder_base_dim": 256,
9
+ "dim_mult": [
10
+ 1,
11
+ 2,
12
+ 4,
13
+ 4
14
+ ],
15
+ "dropout": 0.0,
16
+ "in_channels": 12,
17
+ "is_residual": true,
18
+ "latents_mean": [
19
+ -0.2289,
20
+ -0.0052,
21
+ -0.1323,
22
+ -0.2339,
23
+ -0.2799,
24
+ 0.0174,
25
+ 0.1838,
26
+ 0.1557,
27
+ -0.1382,
28
+ 0.0542,
29
+ 0.2813,
30
+ 0.0891,
31
+ 0.157,
32
+ -0.0098,
33
+ 0.0375,
34
+ -0.1825,
35
+ -0.2246,
36
+ -0.1207,
37
+ -0.0698,
38
+ 0.5109,
39
+ 0.2665,
40
+ -0.2108,
41
+ -0.2158,
42
+ 0.2502,
43
+ -0.2055,
44
+ -0.0322,
45
+ 0.1109,
46
+ 0.1567,
47
+ -0.0729,
48
+ 0.0899,
49
+ -0.2799,
50
+ -0.123,
51
+ -0.0313,
52
+ -0.1649,
53
+ 0.0117,
54
+ 0.0723,
55
+ -0.2839,
56
+ -0.2083,
57
+ -0.052,
58
+ 0.3748,
59
+ 0.0152,
60
+ 0.1957,
61
+ 0.1433,
62
+ -0.2944,
63
+ 0.3573,
64
+ -0.0548,
65
+ -0.1681,
66
+ -0.0667
67
+ ],
68
+ "latents_std": [
69
+ 0.4765,
70
+ 1.0364,
71
+ 0.4514,
72
+ 1.1677,
73
+ 0.5313,
74
+ 0.499,
75
+ 0.4818,
76
+ 0.5013,
77
+ 0.8158,
78
+ 1.0344,
79
+ 0.5894,
80
+ 1.0901,
81
+ 0.6885,
82
+ 0.6165,
83
+ 0.8454,
84
+ 0.4978,
85
+ 0.5759,
86
+ 0.3523,
87
+ 0.7135,
88
+ 0.6804,
89
+ 0.5833,
90
+ 1.4146,
91
+ 0.8986,
92
+ 0.5659,
93
+ 0.7069,
94
+ 0.5338,
95
+ 0.4889,
96
+ 0.4917,
97
+ 0.4069,
98
+ 0.4999,
99
+ 0.6866,
100
+ 0.4093,
101
+ 0.5709,
102
+ 0.6065,
103
+ 0.6415,
104
+ 0.4944,
105
+ 0.5726,
106
+ 1.2042,
107
+ 0.5458,
108
+ 1.6887,
109
+ 0.3971,
110
+ 1.06,
111
+ 0.3943,
112
+ 0.5537,
113
+ 0.5444,
114
+ 0.4089,
115
+ 0.7468,
116
+ 0.7744
117
+ ],
118
+ "num_res_blocks": 2,
119
+ "out_channels": 12,
120
+ "patch_size": 2,
121
+ "scale_factor_spatial": 16,
122
+ "scale_factor_temporal": 4,
123
+ "temperal_downsample": [
124
+ false,
125
+ true,
126
+ true
127
+ ],
128
+ "z_dim": 48
129
+ }
vae/diffusion_pytorch_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:230496cb59ff85bc9c040487737c4062480cb61c71e697b197b4c30142f2a0da
3
+ size 1409400600
video_preprocessor_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "size": {
3
+ "longest_edge": 25165824,
4
+ "shortest_edge": 4096
5
+ },
6
+ "patch_size": 16,
7
+ "temporal_patch_size": 2,
8
+ "merge_size": 2,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_std": [
15
+ 0.5,
16
+ 0.5,
17
+ 0.5
18
+ ],
19
+ "processor_class": "Qwen3VLProcessor",
20
+ "video_processor_type": "Qwen3VLVideoProcessor"
21
+ }
vision_encoder/config.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3VLVisionModel"
4
+ ],
5
+ "deepstack_visual_indexes": [
6
+ 8,
7
+ 16,
8
+ 24
9
+ ],
10
+ "depth": 27,
11
+ "dtype": "bfloat16",
12
+ "hidden_act": "gelu_pytorch_tanh",
13
+ "hidden_size": 1152,
14
+ "in_channels": 3,
15
+ "initializer_range": 0.02,
16
+ "intermediate_size": 4304,
17
+ "model_type": "qwen3_vl",
18
+ "num_heads": 16,
19
+ "num_position_embeddings": 2304,
20
+ "out_hidden_size": 4096,
21
+ "patch_size": 16,
22
+ "spatial_merge_size": 2,
23
+ "temporal_patch_size": 2,
24
+ "transformers_version": "4.57.6"
25
+ }
vision_encoder/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b76056ebf9067cdb36e77554b2c1855aa1b662a9265c8ed26468b490e40b4adb
3
+ size 1152811416
vocab.json ADDED
The diff for this file is too large to render. See raw diff