kelseye commited on Aug 25

Commit

fac48c4

verified ·

1 Parent(s): 85d11c2

Upload folder using huggingface_hub

Browse files

Files changed (36) hide show

.gitattributes +8 -0
README.md +354 -3
assets/comp_effic.png +3 -0
assets/logo.png +0 -0
assets/moe_2.png +3 -0
assets/moe_arch.png +0 -0
assets/performance.png +3 -0
assets/vae.png +3 -0
config.json +43 -42
wav2vec2-large-xlsr-53-english/.msc +0 -0
wav2vec2-large-xlsr-53-english/.mv +1 -0
wav2vec2-large-xlsr-53-english/README.md +165 -0
wav2vec2-large-xlsr-53-english/alphabet.json +1 -0
wav2vec2-large-xlsr-53-english/config.json +75 -0
wav2vec2-large-xlsr-53-english/configuration.json +1 -0
wav2vec2-large-xlsr-53-english/eval.py +164 -0
wav2vec2-large-xlsr-53-english/flax_model.msgpack +3 -0
wav2vec2-large-xlsr-53-english/full_eval.sh +15 -0
wav2vec2-large-xlsr-53-english/language_model/attrs.json +1 -0
wav2vec2-large-xlsr-53-english/language_model/lm.binary +3 -0
wav2vec2-large-xlsr-53-english/language_model/unigrams.txt +0 -0
wav2vec2-large-xlsr-53-english/log_mozilla-foundation_common_voice_6_0_en_test_predictions.txt +0 -0
wav2vec2-large-xlsr-53-english/log_mozilla-foundation_common_voice_6_0_en_test_predictions_greedy.txt +0 -0
wav2vec2-large-xlsr-53-english/log_mozilla-foundation_common_voice_6_0_en_test_targets.txt +0 -0
wav2vec2-large-xlsr-53-english/log_speech-recognition-community-v2_dev_data_en_validation_predictions.txt +0 -0
wav2vec2-large-xlsr-53-english/log_speech-recognition-community-v2_dev_data_en_validation_predictions_greedy.txt +0 -0
wav2vec2-large-xlsr-53-english/log_speech-recognition-community-v2_dev_data_en_validation_targets.txt +0 -0
wav2vec2-large-xlsr-53-english/model.safetensors +3 -0
wav2vec2-large-xlsr-53-english/mozilla-foundation_common_voice_6_0_en_test_eval_results.txt +2 -0
wav2vec2-large-xlsr-53-english/mozilla-foundation_common_voice_6_0_en_test_eval_results_greedy.txt +2 -0
wav2vec2-large-xlsr-53-english/preprocessor_config.json +10 -0
wav2vec2-large-xlsr-53-english/pytorch_model.bin +3 -0
wav2vec2-large-xlsr-53-english/special_tokens_map.json +1 -0
wav2vec2-large-xlsr-53-english/speech-recognition-community-v2_dev_data_en_validation_eval_results.txt +2 -0
wav2vec2-large-xlsr-53-english/speech-recognition-community-v2_dev_data_en_validation_eval_results_greedy.txt +2 -0
wav2vec2-large-xlsr-53-english/vocab.json +1 -0

.gitattributes CHANGED Viewed

@@ -7,3 +7,11 @@ diffusion_pytorch_model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs
 google/umt5-xxl/spiece.model filter=lfs diff=lfs merge=lfs -text
 google/umt5-xxl/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 models_t5_umt5-xxl-enc-bf16.pth filter=lfs diff=lfs merge=lfs -text

 google/umt5-xxl/spiece.model filter=lfs diff=lfs merge=lfs -text
 google/umt5-xxl/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 models_t5_umt5-xxl-enc-bf16.pth filter=lfs diff=lfs merge=lfs -text
+assets/comp_effic.png filter=lfs diff=lfs merge=lfs -text
+assets/moe_2.png filter=lfs diff=lfs merge=lfs -text
+assets/performance.png filter=lfs diff=lfs merge=lfs -text
+assets/vae.png filter=lfs diff=lfs merge=lfs -text
+wav2vec2-large-xlsr-53-english/flax_model.msgpack filter=lfs diff=lfs merge=lfs -text
+wav2vec2-large-xlsr-53-english/language_model/lm.binary filter=lfs diff=lfs merge=lfs -text
+wav2vec2-large-xlsr-53-english/model.safetensors filter=lfs diff=lfs merge=lfs -text
+wav2vec2-large-xlsr-53-english/pytorch_model.bin filter=lfs diff=lfs merge=lfs -text

README.md CHANGED Viewed

@@ -1,3 +1,354 @@
----
-license: apache-2.0
----

+# Wan2.2
+<p align="center">
+    <img src="assets/logo.png" width="400"/>
+<p>
+<p align="center">
+    💜 <a href="https://wan.video"><b>Wan</b></a> &nbsp&nbsp ｜ &nbsp&nbsp 🖥️ <a href="https://github.com/Wan-Video/Wan2.2">GitHub</a> &nbsp&nbsp  | &nbsp&nbsp🤗 <a href="https://huggingface.co/Wan-AI/">Hugging Face</a>&nbsp&nbsp | &nbsp&nbsp🤖 <a href="https://modelscope.cn/organization/Wan-AI">ModelScope</a>&nbsp&nbsp | &nbsp&nbsp 📑 <a href="https://arxiv.org/abs/2503.20314">Paper</a> &nbsp&nbsp | &nbsp&nbsp 📑 <a href="https://wan.video/welcome?spm=a2ty_o02.30011076.0.0.6c9ee41eCcluqg">Blog</a> &nbsp&nbsp |  &nbsp&nbsp 💬  <a href="https://discord.gg/AKNgpMK4Yj">Discord</a>&nbsp&nbsp
+    <br>
+    📕 <a href="https://alidocs.dingtalk.com/i/nodes/jb9Y4gmKWrx9eo4dCql9LlbYJGXn6lpz">使用指南(中文)</a>&nbsp&nbsp | &nbsp&nbsp 📘 <a href="https://alidocs.dingtalk.com/i/nodes/EpGBa2Lm8aZxe5myC99MelA2WgN7R35y">User Guide(English)</a>&nbsp&nbsp | &nbsp&nbsp💬 <a href="https://gw.alicdn.com/imgextra/i2/O1CN01tqjWFi1ByuyehkTSB_!!6000000000015-0-tps-611-1279.jpg">WeChat(微信)</a>&nbsp&nbsp
+<br>
+-----
+[**Wan: Open and Advanced Large-Scale Video Generative Models**](https://arxiv.org/abs/2503.20314) <be>
+We are excited to introduce **Wan2.2**, a major upgrade to our foundational video models. With **Wan2.2**, we have focused on incorporating the following innovations:
+- 👍 **Effective MoE Architecture**: Wan2.2 introduces a Mixture-of-Experts (MoE) architecture into video diffusion models. By separating the denoising process cross timesteps with specialized powerful expert models, this enlarges the overall model capacity while maintaining the same computational cost.
+- 👍 **Cinematic-level Aesthetics**: Wan2.2 incorporates meticulously curated aesthetic data, complete with detailed labels for lighting, composition, contrast, color tone, and more. This allows for more precise and controllable cinematic style generation, facilitating the creation of videos with customizable aesthetic preferences.
+- 👍 **Complex Motion Generation**: Compared to Wan2.1, Wan2.2 is trained on a significantly larger data, with +65.6% more images and +83.2% more videos. This expansion notably enhances the model's generalization across multiple dimensions such as motions,  semantics, and aesthetics, achieving TOP performance among all open-sourced and closed-sourced models.
+- 👍 **Efficient High-Definition Hybrid TI2V**:  Wan2.2 open-sources a 5B model built with our advanced Wan2.2-VAE that achieves a compression ratio of **16×16×4**. This model supports both text-to-video and image-to-video generation at 720P resolution with 24fps and can also run on consumer-grade graphics cards like 4090. It is one of the fastest **720P@24fps** models currently available, capable of serving both the industrial and academic sectors simultaneously.
+## Video Demos
+<div align="center">
+  <video src="https://github.com/user-attachments/assets/b63bfa58-d5d7-4de6-a1a2-98970b06d9a7" width="70%" poster=""> </video>
+</div>
+## 🔥 Latest News!!
+* Aug 26, 2025: 🎵 We introduce **Wan2.2-S2V-14B**, an audio-driven cinematic video generation model, including [inference code](#run-speech-to-video-generation), [model weights](#model-download), and blog! Try our [ModelScope Gradio](https://www.modelscope.cn/studios/Wan-AI/Wan2.2-S2V) and [HuggingFace Gradio](https://huggingface.co/spaces/Wan-AI/Wan2.2-S2V) demos!
+* Jul 28, 2025: 👋 We have open a [HF space](https://huggingface.co/spaces/Wan-AI/Wan-2.2-5B) using the TI2V-5B model. Enjoy!
+* Jul 28, 2025: 👋 Wan2.2 has been integrated into ComfyUI ([CN](https://docs.comfy.org/zh-CN/tutorials/video/wan/wan2_2) | [EN](https://docs.comfy.org/tutorials/video/wan/wan2_2)). Enjoy!
+* Jul 28, 2025: 👋 Wan2.2's T2V, I2V and TI2V have been integrated into Diffusers ([T2V-A14B](https://huggingface.co/Wan-AI/Wan2.2-T2V-A14B-Diffusers) | [I2V-A14B](https://huggingface.co/Wan-AI/Wan2.2-I2V-A14B-Diffusers) | [TI2V-5B](https://huggingface.co/Wan-AI/Wan2.2-TI2V-5B-Diffusers)). Feel free to give it a try!
+* Jul 28, 2025: 👋 We've released the inference code and model weights of **Wan2.2**.
+## Community Works
+If your research or project builds upon [**Wan2.1**](https://github.com/Wan-Video/Wan2.1) or [**Wan2.2**](https://github.com/Wan-Video/Wan2.2), and you would like more people to see it, please inform us.
+- [DiffSynth-Studio](https://github.com/modelscope/DiffSynth-Studio) provides comprehensive support for Wan 2.2, including low-GPU-memory layer-by-layer offload, FP8 quantization, sequence parallelism, LoRA training, full training.
+- [Kijai's ComfyUI WanVideoWrapper](https://github.com/kijai/ComfyUI-WanVideoWrapper) is an alternative implementation of Wan models for ComfyUI. Thanks to its Wan-only focus, it's on the frontline of getting cutting edge optimizations and hot research features, which are often hard to integrate into ComfyUI quickly due to its more rigid structure.
+## 📑 Todo List
+- Wan2.2 Text-to-Video
+    - [x] Multi-GPU Inference code of the A14B and 14B models
+    - [x] Checkpoints of the A14B and 14B models
+    - [x] ComfyUI integration
+    - [x] Diffusers integration
+- Wan2.2 Image-to-Video
+    - [x] Multi-GPU Inference code of the A14B model
+    - [x] Checkpoints of the A14B model
+    - [x] ComfyUI integration
+    - [x] Diffusers integration
+- Wan2.2 Text-Image-to-Video
+    - [x] Multi-GPU Inference code of the 5B model
+    - [x] Checkpoints of the 5B model
+    - [x] ComfyUI integration
+    - [x] Diffusers integration
+- Wan2.2-S2V Speech-to-Video
+    - [x] Inference code of Wan2.2-S2V
+    - [x] Checkpoints of Wan2.2-S2V-14B
+    - [ ] ComfyUI integration
+    - [ ] Diffusers integration
+## Run Wan2.2
+#### Installation
+Clone the repo:
+```sh
+git clone https://github.com/Wan-Video/Wan2.2.git
+cd Wan2.2
+```
+Install dependencies:
+```sh
+# Ensure torch >= 2.4.0
+# If the installation of `flash_attn` fails, try installing the other packages first and install `flash_attn` last
+pip install -r requirements.txt
+```
+#### Model Download
+| Models              | Download Links                                                                                                                              | Description |
+|--------------------|---------------------------------------------------------------------------------------------------------------------------------------------|-------------|
+| T2V-A14B    | 🤗 [Huggingface](https://huggingface.co/Wan-AI/Wan2.2-T2V-A14B)    🤖 [ModelScope](https://modelscope.cn/models/Wan-AI/Wan2.2-T2V-A14B)    | Text-to-Video MoE model, supports 480P & 720P |
+| I2V-A14B    | 🤗 [Huggingface](https://huggingface.co/Wan-AI/Wan2.2-I2V-A14B)    🤖 [ModelScope](https://modelscope.cn/models/Wan-AI/Wan2.2-I2V-A14B)    | Image-to-Video MoE model, supports 480P & 720P |
+| TI2V-5B     | 🤗 [Huggingface](https://huggingface.co/Wan-AI/Wan2.2-TI2V-5B)     🤖 [ModelScope](https://modelscope.cn/models/Wan-AI/Wan2.2-TI2V-5B)     | High-compression VAE, T2V+I2V, supports 720P |
+| S2V-14B     | 🤗 [Huggingface](https://huggingface.co/Wan-AI/Wan2.2-S2V-14B)     🤖 [ModelScope](https://modelscope.cn/models/Wan-AI/Wan2.2-S2V-14B)     | Speech-to-Video model, supports 480P & 720P |
+> 💡Note:
+> The TI2V-5B model supports 720P video generation at **24 FPS**.
+Download models using huggingface-cli:
+``` sh
+pip install "huggingface_hub[cli]"
+huggingface-cli download Wan-AI/Wan2.2-T2V-A14B --local-dir ./Wan2.2-T2V-A14B
+```
+Download models using modelscope-cli:
+``` sh
+pip install modelscope
+modelscope download Wan-AI/Wan2.2-T2V-A14B --local_dir ./Wan2.2-T2V-A14B
+```
+#### Run Text-to-Video Generation
+This repository supports the `Wan2.2-T2V-A14B` Text-to-Video model and can simultaneously support video generation at 480P and 720P resolutions.
+##### (1) Without Prompt Extension
+To facilitate implementation, we will start with a basic version of the inference process that skips the [prompt extension](#2-using-prompt-extention) step.
+- Single-GPU inference
+``` sh
+python generate.py  --task t2v-A14B --size 1280*720 --ckpt_dir ./Wan2.2-T2V-A14B --offload_model True --convert_model_dtype --prompt "Two anthropomorphic cats in comfy boxing gear and bright gloves fight intensely on a spotlighted stage."
+```
+> 💡 This command can run on a GPU with at least 80GB VRAM.
+> 💡If you encounter OOM (Out-of-Memory) issues, you can use the `--offload_model True`, `--convert_model_dtype` and `--t5_cpu` options to reduce GPU memory usage.
+- Multi-GPU inference using FSDP + DeepSpeed Ulysses
+  We use [PyTorch FSDP](https://docs.pytorch.org/docs/stable/fsdp.html) and [DeepSpeed Ulysses](https://arxiv.org/abs/2309.14509) to accelerate inference.
+``` sh
+torchrun --nproc_per_node=8 generate.py --task t2v-A14B --size 1280*720 --ckpt_dir ./Wan2.2-T2V-A14B --dit_fsdp --t5_fsdp --ulysses_size 8 --prompt "Two anthropomorphic cats in comfy boxing gear and bright gloves fight intensely on a spotlighted stage."
+```
+##### (2) Using Prompt Extension
+Extending the prompts can effectively enrich the details in the generated videos, further enhancing the video quality. Therefore, we recommend enabling prompt extension. We provide the following two methods for prompt extension:
+- Use the Dashscope API for extension.
+  - Apply for a `dashscope.api_key` in advance ([EN](https://www.alibabacloud.com/help/en/model-studio/getting-started/first-api-call-to-qwen) | [CN](https://help.aliyun.com/zh/model-studio/getting-started/first-api-call-to-qwen)).
+  - Configure the environment variable `DASH_API_KEY` to specify the Dashscope API key. For users of Alibaba Cloud's international site, you also need to set the environment variable `DASH_API_URL` to 'https://dashscope-intl.aliyuncs.com/api/v1'. For more detailed instructions, please refer to the [dashscope document](https://www.alibabacloud.com/help/en/model-studio/developer-reference/use-qwen-by-calling-api?spm=a2c63.p38356.0.i1).
+  - Use the `qwen-plus` model for text-to-video tasks and `qwen-vl-max` for image-to-video tasks.
+  - You can modify the model used for extension with the parameter `--prompt_extend_model`. For example:
+```sh
+DASH_API_KEY=your_key torchrun --nproc_per_node=8 generate.py  --task t2v-A14B --size 1280*720 --ckpt_dir ./Wan2.2-T2V-A14B --dit_fsdp --t5_fsdp --ulysses_size 8 --prompt "Two anthropomorphic cats in comfy boxing gear and bright gloves fight intensely on a spotlighted stage" --use_prompt_extend --prompt_extend_method 'dashscope' --prompt_extend_target_lang 'zh'
+```
+- Using a local model for extension.
+  - By default, the Qwen model on HuggingFace is used for this extension. Users can choose Qwen models or other models based on the available GPU memory size.
+  - For text-to-video tasks, you can use models like `Qwen/Qwen2.5-14B-Instruct`, `Qwen/Qwen2.5-7B-Instruct` and `Qwen/Qwen2.5-3B-Instruct`.
+  - For image-to-video tasks, you can use models like `Qwen/Qwen2.5-VL-7B-Instruct` and `Qwen/Qwen2.5-VL-3B-Instruct`.
+  - Larger models generally provide better extension results but require more GPU memory.
+  - You can modify the model used for extension with the parameter `--prompt_extend_model` , allowing you to specify either a local model path or a Hugging Face model. For example:
+``` sh
+torchrun --nproc_per_node=8 generate.py  --task t2v-A14B --size 1280*720 --ckpt_dir ./Wan2.2-T2V-A14B --dit_fsdp --t5_fsdp --ulysses_size 8 --prompt "Two anthropomorphic cats in comfy boxing gear and bright gloves fight intensely on a spotlighted stage" --use_prompt_extend --prompt_extend_method 'local_qwen' --prompt_extend_target_lang 'zh'
+```
+#### Run Image-to-Video Generation
+This repository supports the `Wan2.2-I2V-A14B` Image-to-Video model and can simultaneously support video generation at 480P and 720P resolutions.
+- Single-GPU inference
+```sh
+python generate.py --task i2v-A14B --size 1280*720 --ckpt_dir ./Wan2.2-I2V-A14B --offload_model True --convert_model_dtype --image examples/i2v_input.JPG --prompt "Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline's intricate details and the refreshing atmosphere of the seaside."
+```
+> This command can run on a GPU with at least 80GB VRAM.
+> 💡For the Image-to-Video task, the `size` parameter represents the area of the generated video, with the aspect ratio following that of the original input image.
+- Multi-GPU inference using FSDP + DeepSpeed Ulysses
+```sh
+torchrun --nproc_per_node=8 generate.py --task i2v-A14B --size 1280*720 --ckpt_dir ./Wan2.2-I2V-A14B --image examples/i2v_input.JPG --dit_fsdp --t5_fsdp --ulysses_size 8 --prompt "Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline's intricate details and the refreshing atmosphere of the seaside."
+```
+- Image-to-Video Generation without prompt
+```sh
+DASH_API_KEY=your_key torchrun --nproc_per_node=8 generate.py --task i2v-A14B --size 1280*720 --ckpt_dir ./Wan2.2-I2V-A14B --prompt '' --image examples/i2v_input.JPG --dit_fsdp --t5_fsdp --ulysses_size 8 --use_prompt_extend --prompt_extend_method 'dashscope'
+```
+> 💡The model can generate videos solely from the input image. You can use prompt extension to generate prompt from the image.
+> The process of prompt extension can be referenced [here](#2-using-prompt-extention).
+#### Run Text-Image-to-Video Generation
+This repository supports the `Wan2.2-TI2V-5B` Text-Image-to-Video model and can support video generation at 720P resolutions.
+- Single-GPU Text-to-Video inference
+```sh
+python generate.py --task ti2v-5B --size 1280*704 --ckpt_dir ./Wan2.2-TI2V-5B --offload_model True --convert_model_dtype --t5_cpu --prompt "Two anthropomorphic cats in comfy boxing gear and bright gloves fight intensely on a spotlighted stage"
+```
+> 💡Unlike other tasks, the 720P resolution of the Text-Image-to-Video task is `1280*704` or `704*1280`.
+> This command can run on a GPU with at least 24GB VRAM (e.g, RTX 4090 GPU).
+> 💡If you are running on a GPU with at least 80GB VRAM, you can remove the `--offload_model True`, `--convert_model_dtype` and `--t5_cpu` options to speed up execution.
+- Single-GPU Image-to-Video inference
+```sh
+python generate.py --task ti2v-5B --size 1280*704 --ckpt_dir ./Wan2.2-TI2V-5B --offload_model True --convert_model_dtype --t5_cpu --image examples/i2v_input.JPG --prompt "Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline's intricate details and the refreshing atmosphere of the seaside."
+```
+> 💡If the image parameter is configured, it is an Image-to-Video generation; otherwise, it defaults to a Text-to-Video generation.
+> 💡Similar to Image-to-Video, the `size` parameter represents the area of the generated video, with the aspect ratio following that of the original input image.
+- Multi-GPU inference using FSDP + DeepSpeed Ulysses
+```sh
+torchrun --nproc_per_node=8 generate.py --task ti2v-5B --size 1280*704 --ckpt_dir ./Wan2.2-TI2V-5B --dit_fsdp --t5_fsdp --ulysses_size 8 --image examples/i2v_input.JPG --prompt "Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline's intricate details and the refreshing atmosphere of the seaside."
+```
+> The process of prompt extension can be referenced [here](#2-using-prompt-extention).
+#### Run Speech-to-Video Generation
+This repository supports the `Wan2.2-S2V-14B` Speech-to-Video model and can simultaneously support video generation at 480P and 720P resolutions.
+- Single-GPU Speech-to-Video inference
+```sh
+python generate.py  --task s2v-14B --size 1024*704 --ckpt_dir ./Wan2.2-S2V-14B/ --offload_model True --convert_model_dtype --prompt "Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard."  --image "examples/i2v_input.JPG" --audio "examples/talk.wav"
+# Without setting --num_clip, the generated video length will automatically adjust based on the input audio length
+```
+> 💡 This command can run on a GPU with at least 80GB VRAM.
+- Multi-GPU inference using FSDP + DeepSpeed Ulysses
+```sh
+torchrun --nproc_per_node=8 generate.py --task s2v-14B --size 1024*704 --ckpt_dir ./Wan2.2-S2V-14B/ --dit_fsdp --t5_fsdp --ulysses_size 8 --prompt "Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard." --image "examples/i2v_input.JPG" --audio "examples/talk.wav"
+```
+- Pose + Audio driven generation
+```sh
+torchrun --nproc_per_node=8 generate.py --task s2v-14B --size 1024*704 --ckpt_dir ./Wan2.2-S2V-14B/ --dit_fsdp --t5_fsdp --ulysses_size 8 --prompt "a person is singing" --image "examples/pose.png" --audio "examples/sing.MP3" --pose_video "./examples/pose.mp4"
+```
+> 💡For the Speech-to-Video task, the `size` parameter represents the area of the generated video, with the aspect ratio following that of the original input image.
+> 💡The model can generate videos from audio input combined with reference image and optional text prompt.
+> 💡The `--pose_video` parameter enables pose-driven generation, allowing the model to follow specific pose sequences while generating videos synchronized with audio input.
+> 💡The `--num_clip` parameter controls the number of video clips generated, useful for quick preview with shorter generation time.
+## Computational Efficiency on Different GPUs
+We test the computational efficiency of different **Wan2.2** models on different GPUs in the following table. The results are presented in the format: **Total time (s) / peak GPU memory (GB)**.
+<div align="center">
+    <img src="assets/comp_effic.png" alt="" style="width: 80%;" />
+</div>
+> The parameter settings for the tests presented in this table are as follows:
+> (1) Multi-GPU: 14B: `--ulysses_size 4/8 --dit_fsdp --t5_fsdp`, 5B: `--ulysses_size 4/8 --offload_model True --convert_model_dtype --t5_cpu`; Single-GPU: 14B: `--offload_model True --convert_model_dtype`, 5B: `--offload_model True --convert_model_dtype --t5_cpu`
+(--convert_model_dtype converts model parameter types to config.param_dtype);
+> (2) The distributed testing utilizes the built-in FSDP and Ulysses implementations, with FlashAttention3 deployed on Hopper architecture GPUs;
+> (3) Tests were run without the `--use_prompt_extend` flag;
+> (4) Reported results are the average of multiple samples taken after the warm-up phase.
+-------
+## Introduction of Wan2.2
+**Wan2.2** builds on the foundation of Wan2.1 with notable improvements in generation quality and model capability. This upgrade is driven by a series of key technical innovations, mainly including the Mixture-of-Experts (MoE) architecture, upgraded training data, and high-compression video generation.
+##### (1) Mixture-of-Experts (MoE) Architecture
+Wan2.2 introduces Mixture-of-Experts (MoE) architecture into the video generation diffusion model. MoE has been widely validated in large language models as an efficient approach to increase total model parameters while keeping inference cost nearly unchanged. In Wan2.2, the A14B model series adopts a two-expert design tailored to the denoising process of diffusion models: a high-noise expert for the early stages, focusing on overall layout; and a low-noise expert for the later stages, refining video details. Each expert model has about 14B parameters, resulting in a total of 27B parameters but only 14B active parameters per step, keeping inference computation and GPU memory nearly unchanged.
+<div align="center">
+    <img src="assets/moe_arch.png" alt="" style="width: 90%;" />
+</div>
+The transition point between the two experts is determined by the signal-to-noise ratio (SNR), a metric that decreases monotonically as the denoising step $t$ increases. At the beginning of the denoising process, $t$ is large and the noise level is high, so the SNR is at its minimum, denoted as ${SNR}_{min}$. In this stage, the high-noise expert is activated. We define a threshold step ${t}_{moe}$ corresponding to half of the ${SNR}_{min}$, and switch to the low-noise expert when $t<{t}_{moe}$.
+<div align="center">
+    <img src="assets/moe_2.png" alt="" style="width: 90%;" />
+</div>
+To validate the effectiveness of the MoE architecture, four settings are compared based on their validation loss curves. The baseline **Wan2.1** model does not employ the MoE architecture. Among the MoE-based variants, the **Wan2.1 & High-Noise Expert** reuses the Wan2.1 model as the low-noise expert while uses the  Wan2.2's high-noise expert, while the **Wan2.1 & Low-Noise Expert** uses Wan2.1 as the high-noise expert and employ the Wan2.2's low-noise expert. The **Wan2.2 (MoE)** (our final version) achieves the lowest validation loss, indicating that its generated video distribution is closest to ground-truth and exhibits superior convergence.
+##### (2) Efficient High-Definition Hybrid TI2V
+To enable more efficient deployment, Wan2.2 also explores a high-compression design. In addition to the 27B MoE models, a 5B dense model, i.e., TI2V-5B, is released. It is supported by a high-compression Wan2.2-VAE, which achieves a $T\times H\times W$ compression ratio of $4\times16\times16$, increasing the overall compression rate to 64 while maintaining high-quality video reconstruction. With an additional patchification layer, the total compression ratio of TI2V-5B reaches $4\times32\times32$. Without specific optimization, TI2V-5B can generate a 5-second 720P video in under 9 minutes on a single consumer-grade GPU, ranking among the fastest 720P@24fps video generation models. This model also natively supports both text-to-video and image-to-video tasks within a single unified framework, covering both academic research and practical applications.
+<div align="center">
+    <img src="assets/vae.png" alt="" style="width: 80%;" />
+</div>
+##### Comparisons to SOTAs
+We compared Wan2.2 with leading closed-source commercial models on our new Wan-Bench 2.0, evaluating performance across multiple crucial dimensions. The results demonstrate that Wan2.2 achieves superior performance compared to these leading models.
+<div align="center">
+    <img src="assets/performance.png" alt="" style="width: 90%;" />
+</div>
+## Citation
+If you find our work helpful, please cite us.
+```
+@article{wan2025,
+      title={Wan: Open and Advanced Large-Scale Video Generative Models},
+      author={Team Wan and Ang Wang and Baole Ai and Bin Wen and Chaojie Mao and Chen-Wei Xie and Di Chen and Feiwu Yu and Haiming Zhao and Jianxiao Yang and Jianyuan Zeng and Jiayu Wang and Jingfeng Zhang and Jingren Zhou and Jinkai Wang and Jixuan Chen and Kai Zhu and Kang Zhao and Keyu Yan and Lianghua Huang and Mengyang Feng and Ningyi Zhang and Pandeng Li and Pingyu Wu and Ruihang Chu and Ruili Feng and Shiwei Zhang and Siyang Sun and Tao Fang and Tianxing Wang and Tianyi Gui and Tingyu Weng and Tong Shen and Wei Lin and Wei Wang and Wei Wang and Wenmeng Zhou and Wente Wang and Wenting Shen and Wenyuan Yu and Xianzhong Shi and Xiaoming Huang and Xin Xu and Yan Kou and Yangyu Lv and Yifei Li and Yijing Liu and Yiming Wang and Yingya Zhang and Yitong Huang and Yong Li and You Wu and Yu Liu and Yulin Pan and Yun Zheng and Yuntao Hong and Yupeng Shi and Yutong Feng and Zeyinzi Jiang and Zhen Han and Zhi-Fan Wu and Ziyu Liu},
+      journal = {arXiv preprint arXiv:2503.20314},
+      year={2025}
+}
+```
+## License Agreement
+The models in this repository are licensed under the Apache 2.0 License. We claim no rights over the your generated contents, granting you the freedom to use them while ensuring that your usage complies with the provisions of this license. You are fully accountable for your use of the models, which must not involve sharing any content that violates applicable laws, causes harm to individuals or groups, disseminates personal information intended for harm, spreads misinformation, or targets vulnerable populations. For a complete list of restrictions and details regarding your rights, please refer to the full text of the [license](LICENSE.txt).
+## Acknowledgements
+We would like to thank the contributors to the [SD3](https://huggingface.co/stabilityai/stable-diffusion-3-medium), [Qwen](https://huggingface.co/Qwen), [umt5-xxl](https://huggingface.co/google/umt5-xxl), [diffusers](https://github.com/huggingface/diffusers) and [HuggingFace](https://huggingface.co) repositories, for their open research.
+## Contact Us
+If you would like to leave a message to our research or product teams, feel free to join our [Discord](https://discord.gg/AKNgpMK4Yj) or [WeChat groups](https://gw.alicdn.com/imgextra/i2/O1CN01tqjWFi1ByuyehkTSB_!!6000000000015-0-tps-611-1279.jpg)!

assets/comp_effic.png ADDED Viewed

Git LFS Details

SHA256: 75ee012dcfb08365bec67a3ec7afc126fc2817f79b9f80e38711792d4770e32b
Pointer size: 131 Bytes
Size of remote file: 202 kB

assets/logo.png ADDED Viewed

assets/moe_2.png ADDED Viewed

Git LFS Details

SHA256: 4ea471ccb64349bd08bc9a78f336ae000e9ca3b40da9a652b8028b214a8c6093
Pointer size: 131 Bytes
Size of remote file: 528 kB

assets/moe_arch.png ADDED Viewed

assets/performance.png ADDED Viewed

Git LFS Details

SHA256: 97ef99c13c8ae717a8a11c8d8ec927b69077c647cc6689755d08fc38e7fbb830
Pointer size: 131 Bytes
Size of remote file: 307 kB

assets/vae.png ADDED Viewed

Git LFS Details

SHA256: 4aaea5e187f1c5908e15ade5bef24c9fb59882986bc3d2ad75f7fe820f3d772f
Pointer size: 131 Bytes
Size of remote file: 165 kB

config.json CHANGED Viewed

@@ -1,43 +1,44 @@
 {
-  "__name__": "Config: Transformer config for WanModel_S2V",
-  "_class_name": "WanModel_S2V",
-  "_diffusers_version": "0.34.0",
-  "adain_mode": "attn_norm",
-  "add_last_motion": true,
-  "audio_dim": 1024,
-  "audio_inject_layers": [
-    0,
-    4,
-    8,
-    12,
-    16,
-    20,
-    24,
-    27,
-    30,
-    33,
-    36,
-    39
-  ],
-  "cond_dim": 16,
-  "dim": 5120,
-  "enable_adain": true,
-  "enable_framepack": true,
-  "enable_motioner": false,
-  "enable_tsm": false,
-  "eps": 1e-06,
-  "ffn_dim": 13824,
-  "framepack_drop_mode": "padd",
-  "freq_dim": 256,
-  "in_dim": 16,
-  "model_type": "s2v",
-  "motion_token_num": 1024,
-  "num_audio_token": 4,
-  "num_heads": 40,
-  "num_layers": 40,
-  "out_dim": 16,
-  "text_len": 512,
-  "trainable_token_pos_emb": false,
-  "zero_init": true,
-  "zero_timestep": true
-}

 {
+    "__name__": "Config: Transformer config for WanModel_S2V",
+    "_class_name": "WanModel_S2V",
+    "_diffusers_version": "0.34.0",
+    "adain_mode": "attn_norm",
+    "add_last_motion": true,
+    "audio_dim": 1024,
+    "audio_inject_layers": [
+      0,
+      4,
+      8,
+      12,
+      16,
+      20,
+      24,
+      27,
+      30,
+      33,
+      36,
+      39
+    ],
+    "cond_dim": 16,
+    "dim": 5120,
+    "enable_adain": true,
+    "enable_framepack": true,
+    "enable_motioner": false,
+    "enable_tsm": false,
+    "eps": 1e-06,
+    "ffn_dim": 13824,
+    "framepack_drop_mode": "padd",
+    "freq_dim": 256,
+    "in_dim": 16,
+    "model_type": "s2v",
+    "motion_token_num": 1024,
+    "num_audio_token": 4,
+    "num_heads": 40,
+    "num_layers": 40,
+    "out_dim": 16,
+    "text_len": 512,
+    "trainable_token_pos_emb": false,
+    "zero_init": true,
+    "zero_timestep": true
+  }

wav2vec2-large-xlsr-53-english/.msc ADDED Viewed

Binary file (2.33 kB). View file

wav2vec2-large-xlsr-53-english/.mv ADDED Viewed

	@@ -0,0 +1 @@


1	+ Revision:master,CreatedAt:1730986758

wav2vec2-large-xlsr-53-english/README.md ADDED Viewed

	@@ -0,0 +1,165 @@

+---
+language: en
+datasets:
+- common_voice
+- mozilla-foundation/common_voice_6_0
+metrics:
+- wer
+- cer
+tags:
+- audio
+- automatic-speech-recognition
+- en
+- hf-asr-leaderboard
+- mozilla-foundation/common_voice_6_0
+- robust-speech-event
+- speech
+- xlsr-fine-tuning-week
+license: apache-2.0
+model-index:
+- name: XLSR Wav2Vec2 English by Jonatas Grosman
+  results:
+  - task:
+      name: Automatic Speech Recognition
+      type: automatic-speech-recognition
+    dataset:
+      name: Common Voice en
+      type: common_voice
+      args: en
+    metrics:
+    - name: Test WER
+      type: wer
+      value: 19.06
+    - name: Test CER
+      type: cer
+      value: 7.69
+    - name: Test WER (+LM)
+      type: wer
+      value: 14.81
+    - name: Test CER (+LM)
+      type: cer
+      value: 6.84
+  - task:
+      name: Automatic Speech Recognition
+      type: automatic-speech-recognition
+    dataset:
+      name: Robust Speech Event - Dev Data
+      type: speech-recognition-community-v2/dev_data
+      args: en
+    metrics:
+    - name: Dev WER
+      type: wer
+      value: 27.72
+    - name: Dev CER
+      type: cer
+      value: 11.65
+    - name: Dev WER (+LM)
+      type: wer
+      value: 20.85
+    - name: Dev CER (+LM)
+      type: cer
+      value: 11.01
+---
+# Fine-tuned XLSR-53 large model for speech recognition in English
+Fine-tuned [facebook/wav2vec2-large-xlsr-53](https://huggingface.co/facebook/wav2vec2-large-xlsr-53) on English using the train and validation splits of [Common Voice 6.1](https://huggingface.co/datasets/common_voice).
+When using this model, make sure that your speech input is sampled at 16kHz.
+This model has been fine-tuned thanks to the GPU credits generously given by the [OVHcloud](https://www.ovhcloud.com/en/public-cloud/ai-training/) :)
+The script used for training can be found here: https://github.com/jonatasgrosman/wav2vec2-sprint
+## Usage
+The model can be used directly (without a language model) as follows...
+Using the [HuggingSound](https://github.com/jonatasgrosman/huggingsound) library:
+```python
+from huggingsound import SpeechRecognitionModel
+model = SpeechRecognitionModel("jonatasgrosman/wav2vec2-large-xlsr-53-english")
+audio_paths = ["/path/to/file.mp3", "/path/to/another_file.wav"]
+transcriptions = model.transcribe(audio_paths)
+```
+Writing your own inference script:
+```python
+import torch
+import librosa
+from datasets import load_dataset
+from transformers import Wav2Vec2ForCTC, Wav2Vec2Processor
+LANG_ID = "en"
+MODEL_ID = "jonatasgrosman/wav2vec2-large-xlsr-53-english"
+SAMPLES = 10
+test_dataset = load_dataset("common_voice", LANG_ID, split=f"test[:{SAMPLES}]")
+processor = Wav2Vec2Processor.from_pretrained(MODEL_ID)
+model = Wav2Vec2ForCTC.from_pretrained(MODEL_ID)
+# Preprocessing the datasets.
+# We need to read the audio files as arrays
+def speech_file_to_array_fn(batch):
+    speech_array, sampling_rate = librosa.load(batch["path"], sr=16_000)
+    batch["speech"] = speech_array
+    batch["sentence"] = batch["sentence"].upper()
+    return batch
+test_dataset = test_dataset.map(speech_file_to_array_fn)
+inputs = processor(test_dataset["speech"], sampling_rate=16_000, return_tensors="pt", padding=True)
+with torch.no_grad():
+    logits = model(inputs.input_values, attention_mask=inputs.attention_mask).logits
+predicted_ids = torch.argmax(logits, dim=-1)
+predicted_sentences = processor.batch_decode(predicted_ids)
+for i, predicted_sentence in enumerate(predicted_sentences):
+    print("-" * 100)
+    print("Reference:", test_dataset[i]["sentence"])
+    print("Prediction:", predicted_sentence)
+```
+| Reference  | Prediction |
+| ------------- | ------------- |
+| "SHE'LL BE ALL RIGHT." | SHE'LL BE ALL RIGHT |
+| SIX | SIX |
+| "ALL'S WELL THAT ENDS WELL." | ALL AS WELL THAT ENDS WELL |
+| DO YOU MEAN IT? | DO YOU MEAN IT |
+| THE NEW PATCH IS LESS INVASIVE THAN THE OLD ONE, BUT STILL CAUSES REGRESSIONS. | THE NEW PATCH IS LESS INVASIVE THAN THE OLD ONE BUT STILL CAUSES REGRESSION |
+| HOW IS MOZILLA GOING TO HANDLE AMBIGUITIES LIKE QUEUE AND CUE? | HOW IS MOSLILLAR GOING TO HANDLE ANDBEWOOTH HIS LIKE Q AND Q |
+| "I GUESS YOU MUST THINK I'M KINDA BATTY." | RUSTIAN WASTIN PAN ONTE BATTLY |
+| NO ONE NEAR THE REMOTE MACHINE YOU COULD RING? | NO ONE NEAR THE REMOTE MACHINE YOU COULD RING |
+| SAUCE FOR THE GOOSE IS SAUCE FOR THE GANDER. | SAUCE FOR THE GUICE IS SAUCE FOR THE GONDER |
+| GROVES STARTED WRITING SONGS WHEN SHE WAS FOUR YEARS OLD. | GRAFS STARTED WRITING SONGS WHEN SHE WAS FOUR YEARS OLD |
+## Evaluation
+1. To evaluate on `mozilla-foundation/common_voice_6_0` with split `test`
+```bash
+python eval.py --model_id jonatasgrosman/wav2vec2-large-xlsr-53-english --dataset mozilla-foundation/common_voice_6_0 --config en --split test
+```
+2. To evaluate on `speech-recognition-community-v2/dev_data`
+```bash
+python eval.py --model_id jonatasgrosman/wav2vec2-large-xlsr-53-english --dataset speech-recognition-community-v2/dev_data --config en --split validation --chunk_length_s 5.0 --stride_length_s 1.0
+```
+## Citation
+If you want to cite this model you can use this:
+```bibtex
+@misc{grosman2021xlsr53-large-english,
+  title={Fine-tuned {XLSR}-53 large model for speech recognition in {E}nglish},
+  author={Grosman, Jonatas},
+  howpublished={\url{https://huggingface.co/jonatasgrosman/wav2vec2-large-xlsr-53-english}},
+  year={2021}
+}
+```

wav2vec2-large-xlsr-53-english/alphabet.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"labels": ["", "<s>", "</s>", "⁇", " ", "'", "-", "a", "b", "c", "d", "e", "f", "g", "h", "i", "j", "k", "l", "m", "n", "o", "p", "q", "r", "s", "t", "u", "v", "w", "x", "y", "z"], "is_bpe": false}

wav2vec2-large-xlsr-53-english/config.json ADDED Viewed

	@@ -0,0 +1,75 @@

+{
+  "_name_or_path": "facebook/wav2vec2-large-xlsr-53",
+  "activation_dropout": 0.05,
+  "apply_spec_augment": true,
+  "architectures": [
+    "Wav2Vec2ForCTC"
+  ],
+  "attention_dropout": 0.1,
+  "bos_token_id": 1,
+  "conv_bias": true,
+  "conv_dim": [
+    512,
+    512,
+    512,
+    512,
+    512,
+    512,
+    512
+  ],
+  "conv_kernel": [
+    10,
+    3,
+    3,
+    3,
+    3,
+    2,
+    2
+  ],
+  "conv_stride": [
+    5,
+    2,
+    2,
+    2,
+    2,
+    2,
+    2
+  ],
+  "ctc_loss_reduction": "mean",
+  "ctc_zero_infinity": true,
+  "do_stable_layer_norm": true,
+  "eos_token_id": 2,
+  "feat_extract_activation": "gelu",
+  "feat_extract_dropout": 0.0,
+  "feat_extract_norm": "layer",
+  "feat_proj_dropout": 0.05,
+  "final_dropout": 0.0,
+  "hidden_act": "gelu",
+  "hidden_dropout": 0.05,
+  "hidden_size": 1024,
+  "initializer_range": 0.02,
+  "intermediate_size": 4096,
+  "layer_norm_eps": 1e-05,
+  "layerdrop": 0.05,
+  "mask_channel_length": 10,
+  "mask_channel_min_space": 1,
+  "mask_channel_other": 0.0,
+  "mask_channel_prob": 0.0,
+  "mask_channel_selection": "static",
+  "mask_feature_length": 10,
+  "mask_feature_prob": 0.0,
+  "mask_time_length": 10,
+  "mask_time_min_space": 1,
+  "mask_time_other": 0.0,
+  "mask_time_prob": 0.05,
+  "mask_time_selection": "static",
+  "model_type": "wav2vec2",
+  "num_attention_heads": 16,
+  "num_conv_pos_embedding_groups": 16,
+  "num_conv_pos_embeddings": 128,
+  "num_feat_extract_layers": 7,
+  "num_hidden_layers": 24,
+  "pad_token_id": 0,
+  "transformers_version": "4.7.0.dev0",
+  "vocab_size": 33
+}

wav2vec2-large-xlsr-53-english/configuration.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"framework": "pytorch", "task": "automatic-speech-recognition", "allow_remote": true}

wav2vec2-large-xlsr-53-english/eval.py ADDED Viewed

	@@ -0,0 +1,164 @@

+#!/usr/bin/env python3
+from datasets import load_dataset, load_metric, Audio, Dataset
+from transformers import pipeline, AutoFeatureExtractor, AutoTokenizer, AutoConfig, AutoModelForCTC, Wav2Vec2Processor, Wav2Vec2ProcessorWithLM
+import re
+import torch
+import argparse
+from typing import Dict
+def log_results(result: Dataset, args: Dict[str, str]):
+    """ DO NOT CHANGE. This function computes and logs the result metrics. """
+    log_outputs = args.log_outputs
+    dataset_id = "_".join(args.dataset.split("/") + [args.config, args.split])
+    # load metric
+    wer = load_metric("wer")
+    cer = load_metric("cer")
+    # compute metrics
+    wer_result = wer.compute(references=result["target"], predictions=result["prediction"])
+    cer_result = cer.compute(references=result["target"], predictions=result["prediction"])
+    # print & log results
+    result_str = (
+        f"WER: {wer_result}\n"
+        f"CER: {cer_result}"
+    )
+    print(result_str)
+    with open(f"{dataset_id}_eval_results.txt", "w") as f:
+        f.write(result_str)
+    # log all results in text file. Possibly interesting for analysis
+    if log_outputs is not None:
+        pred_file = f"log_{dataset_id}_predictions.txt"
+        target_file = f"log_{dataset_id}_targets.txt"
+        with open(pred_file, "w") as p, open(target_file, "w") as t:
+            # mapping function to write output
+            def write_to_file(batch, i):
+                p.write(f"{i}" + "\n")
+                p.write(batch["prediction"] + "\n")
+                t.write(f"{i}" + "\n")
+                t.write(batch["target"] + "\n")
+            result.map(write_to_file, with_indices=True)
+def normalize_text(text: str, invalid_chars_regex: str, to_lower: bool) -> str:
+    """ DO ADAPT FOR YOUR USE CASE. this function normalizes the target text. """
+    text = text.lower() if to_lower else text.upper()
+    text = re.sub(invalid_chars_regex, " ", text)
+    text = re.sub("\s+", " ", text).strip()
+    return text
+def main(args):
+    # load dataset
+    dataset = load_dataset(args.dataset, args.config, split=args.split, use_auth_token=True)
+    # for testing: only process the first two examples as a test
+    # dataset = dataset.select(range(10))
+    # load processor
+    if args.greedy:
+        processor = Wav2Vec2Processor.from_pretrained(args.model_id)
+        decoder = None
+    else:
+        processor = Wav2Vec2ProcessorWithLM.from_pretrained(args.model_id)
+        decoder = processor.decoder
+    feature_extractor = processor.feature_extractor
+    tokenizer = processor.tokenizer
+    # resample audio
+    dataset = dataset.cast_column("audio", Audio(sampling_rate=feature_extractor.sampling_rate))
+    # load eval pipeline
+    if args.device is None:
+        args.device = 0 if torch.cuda.is_available() else -1
+    config = AutoConfig.from_pretrained(args.model_id)
+    model = AutoModelForCTC.from_pretrained(args.model_id)
+    #asr = pipeline("automatic-speech-recognition", model=args.model_id, device=args.device)
+    asr = pipeline("automatic-speech-recognition", config=config, model=model, tokenizer=tokenizer,
+                   feature_extractor=feature_extractor, decoder=decoder, device=args.device)
+    # build normalizer config
+    tokenizer = AutoTokenizer.from_pretrained(args.model_id)
+    tokens = [x for x in tokenizer.convert_ids_to_tokens(range(0, tokenizer.vocab_size))]
+    special_tokens = [
+        tokenizer.pad_token, tokenizer.word_delimiter_token,
+        tokenizer.unk_token, tokenizer.bos_token,
+        tokenizer.eos_token,
+    ]
+    non_special_tokens = [x for x in tokens if x not in special_tokens]
+    invalid_chars_regex = f"[^\s{re.escape(''.join(set(non_special_tokens)))}]"
+    normalize_to_lower = False
+    for token in non_special_tokens:
+        if token.isalpha() and token.islower():
+            normalize_to_lower = True
+            break
+    # map function to decode audio
+    def map_to_pred(batch, args=args, asr=asr, invalid_chars_regex=invalid_chars_regex, normalize_to_lower=normalize_to_lower):
+        prediction = asr(batch["audio"]["array"], chunk_length_s=args.chunk_length_s, stride_length_s=args.stride_length_s)
+        batch["prediction"] = prediction["text"]
+        batch["target"] = normalize_text(batch["sentence"], invalid_chars_regex, normalize_to_lower)
+        return batch
+    # run inference on all examples
+    result = dataset.map(map_to_pred, remove_columns=dataset.column_names)
+    # filtering out empty targets
+    result = result.filter(lambda example: example["target"] != "")
+    # compute and log_results
+    # do not change function below
+    log_results(result, args)
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser()
+    parser.add_argument(
+        "--model_id", type=str, required=True, help="Model identifier. Should be loadable with 🤗 Transformers"
+    )
+    parser.add_argument(
+        "--dataset", type=str, required=True, help="Dataset name to evaluate the `model_id`. Should be loadable with 🤗 Datasets"
+    )
+    parser.add_argument(
+        "--config", type=str, required=True, help="Config of the dataset. *E.g.* `'en'`  for Common Voice"
+    )
+    parser.add_argument(
+        "--split", type=str, required=True, help="Split of the dataset. *E.g.* `'test'`"
+    )
+    parser.add_argument(
+        "--chunk_length_s", type=float, default=None, help="Chunk length in seconds. Defaults to None. For long audio files a good value would be 5.0 seconds."
+    )
+    parser.add_argument(
+        "--stride_length_s", type=float, default=None, help="Stride of the audio chunks. Defaults to None. For long audio files a good value would be 1.0 seconds."
+    )
+    parser.add_argument(
+        "--log_outputs", action='store_true', help="If defined, write outputs to log file for analysis."
+    )
+    parser.add_argument(
+        "--greedy", action='store_true', help="If defined, the LM will be ignored during inference."
+    )
+    parser.add_argument(
+        "--device",
+        type=int,
+        default=None,
+        help="The device to run the pipeline on. -1 for CPU (default), 0 for the first GPU and so on.",
+    )
+    args = parser.parse_args()
+    main(args)

wav2vec2-large-xlsr-53-english/flax_model.msgpack ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0d3842440388a575e19f19cdb05714afd7018392bd1a7e247c601530a653aa40
+size 1261905572

wav2vec2-large-xlsr-53-english/full_eval.sh ADDED Viewed

	@@ -0,0 +1,15 @@

+# CV - TEST
+python eval.py --model_id jonatasgrosman/wav2vec2-large-xlsr-53-english --dataset mozilla-foundation/common_voice_6_0 --config en --split test --log_outputs --greedy
+mv log_mozilla-foundation_common_voice_6_0_en_test_predictions.txt log_mozilla-foundation_common_voice_6_0_en_test_predictions_greedy.txt
+mv mozilla-foundation_common_voice_6_0_en_test_eval_results.txt mozilla-foundation_common_voice_6_0_en_test_eval_results_greedy.txt
+python eval.py --model_id jonatasgrosman/wav2vec2-large-xlsr-53-english --dataset mozilla-foundation/common_voice_6_0 --config en --split test --log_outputs
+# HF EVENT - DEV
+python eval.py --model_id jonatasgrosman/wav2vec2-large-xlsr-53-english --dataset speech-recognition-community-v2/dev_data --config en --split validation --chunk_length_s 5.0 --stride_length_s 1.0 --log_outputs --greedy
+mv log_speech-recognition-community-v2_dev_data_en_validation_predictions.txt log_speech-recognition-community-v2_dev_data_en_validation_predictions_greedy.txt
+mv speech-recognition-community-v2_dev_data_en_validation_eval_results.txt speech-recognition-community-v2_dev_data_en_validation_eval_results_greedy.txt
+python eval.py --model_id jonatasgrosman/wav2vec2-large-xlsr-53-english --dataset speech-recognition-community-v2/dev_data --config en --split validation --chunk_length_s 5.0 --stride_length_s 1.0 --log_outputs

wav2vec2-large-xlsr-53-english/language_model/attrs.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"alpha": 0.5, "beta": 1.5, "unk_score_offset": -10.0, "score_boundary": true}

wav2vec2-large-xlsr-53-english/language_model/lm.binary ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:47e16abf6384ebd1b3395144330b60710dc43f3d16c4b2b4794071cd117230e5
+size 862913451

wav2vec2-large-xlsr-53-english/language_model/unigrams.txt ADDED Viewed