Instructions to use developerjeremylive/Viggle-Animate-etheroi with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Diffusers
How to use developerjeremylive/Viggle-Animate-etheroi with Diffusers:
pip install -U diffusers transformers accelerate
import torch from diffusers import DiffusionPipeline # switch to "mps" for apple devices pipe = DiffusionPipeline.from_pretrained("developerjeremylive/Viggle-Animate-etheroi", dtype=torch.bfloat16, device_map="cuda") prompt = "Astronaut in a jungle, cold color palette, muted colors, detailed, 8k" image = pipe(prompt).images[0] - Notebooks
- Google Colab
- Kaggle
Commit ·
b78f054
0
Parent(s):
Duplicate from Viggle/Viggle-Animate
Browse filesCo-authored-by: Yun Chen <yycc@users.noreply.huggingface.co>
This view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +68 -0
- LICENSE +84 -0
- LICENSE-CODE +201 -0
- MODIFICATIONS.md +90 -0
- MiniMaxH3TokenRefinerBlock/package.pt2 +3 -0
- MiniMaxH3TransformerBlock/package.pt2 +3 -0
- NOTICE +13 -0
- README.md +347 -0
- assets/fixed_embed_fwd_anyframe.pt +3 -0
- assets/fixed_prompt.txt +16 -0
- examples/demo.sh +43 -0
- examples/media/audio-pin-driving.mp4 +3 -0
- examples/media/audio-pin-out.mp4 +3 -0
- examples/media/audio-pin-ref.png +3 -0
- examples/media/before-after.png +3 -0
- examples/media/community-tweet.png +3 -0
- examples/media/compare-cosplay.mp4 +3 -0
- examples/media/compare-costume.mp4 +3 -0
- examples/media/compare-fastmotion.mp4 +3 -0
- examples/media/compare-gamechar.mp4 +3 -0
- examples/media/compare-highkick.mp4 +3 -0
- examples/media/compare-prop.mp4 +3 -0
- examples/media/compare-style.mp4 +3 -0
- examples/media/hero-corgi.mp4 +3 -0
- examples/media/hero-duo.mp4 +3 -0
- examples/media/output.mp4 +3 -0
- examples/media/pipeline.png +3 -0
- examples/media/pipeline.svg +81 -0
- examples/media/reference.png +3 -0
- examples/media/swap-airliner.mp4 +3 -0
- examples/media/swap-anime-duo.mp4 +3 -0
- examples/media/swap-claymation.mp4 +3 -0
- examples/media/swap-corgi.mp4 +3 -0
- examples/media/swap-panda.mp4 +3 -0
- examples/media/swap-penguin.mp4 +3 -0
- examples/media/swap-photoreal-duo.mp4 +3 -0
- examples/media/swap-robot.mp4 +3 -0
- examples/media/swap-wushu-animals.mp4 +3 -0
- examples/media/swap-wushu-robots.mp4 +3 -0
- examples/media/teaser.mp4 +3 -0
- inference/sample.py +205 -0
- lora/pytorch_lora_weights.safetensors +3 -0
- requirements.txt +7 -0
- transformer/config.json +26 -0
- transformer/diffusion_pytorch_model-00001-of-00014.safetensors +3 -0
- transformer/diffusion_pytorch_model-00002-of-00014.safetensors +3 -0
- transformer/diffusion_pytorch_model-00003-of-00014.safetensors +3 -0
- transformer/diffusion_pytorch_model-00004-of-00014.safetensors +3 -0
- transformer/diffusion_pytorch_model-00005-of-00014.safetensors +3 -0
- transformer/diffusion_pytorch_model-00006-of-00014.safetensors +3 -0
.gitattributes
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
+
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
+
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
+
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
+
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
+
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
+
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
+
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
+
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
+
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
+
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
+
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
+
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
+
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
+
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
+
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
+
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
+
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
+
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
+
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
+
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
+
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
+
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
+
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
+
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
+
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
+
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
+
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
+
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
+
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
+
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
+
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
examples/media/before-after.png filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
examples/media/output.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
examples/media/reference.png filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
examples/media/compare-monkey.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
examples/media/compare-parka.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
examples/media/compare-horns.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
examples/media/compare-costume.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
examples/media/compare-prop.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
examples/media/compare-style.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
examples/media/swap-airliner.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
examples/media/swap-anime-duo.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
examples/media/swap-claymation.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 48 |
+
examples/media/swap-corgi.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 49 |
+
examples/media/swap-panda.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 50 |
+
examples/media/swap-penguin.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 51 |
+
examples/media/swap-photoreal-duo.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 52 |
+
examples/media/swap-robot.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 53 |
+
examples/media/swap-wushu-animals.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 54 |
+
examples/media/swap-wushu-robots.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 55 |
+
examples/media/hero-corgi.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 56 |
+
examples/media/hero-duo.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 57 |
+
examples/media/compare-cosplay.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 58 |
+
examples/media/compare-gamechar.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 59 |
+
examples/media/teaser.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 60 |
+
MiniMaxH3TransformerBlock/package.pt2 filter=lfs diff=lfs merge=lfs -text
|
| 61 |
+
MiniMaxH3TokenRefinerBlock/package.pt2 filter=lfs diff=lfs merge=lfs -text
|
| 62 |
+
examples/media/compare-fastmotion.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 63 |
+
examples/media/compare-highkick.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 64 |
+
examples/media/pipeline.png filter=lfs diff=lfs merge=lfs -text
|
| 65 |
+
examples/media/community-tweet.png filter=lfs diff=lfs merge=lfs -text
|
| 66 |
+
examples/media/audio-pin-driving.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 67 |
+
examples/media/audio-pin-out.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 68 |
+
examples/media/audio-pin-ref.png filter=lfs diff=lfs merge=lfs -text
|
LICENSE
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
MiniMax H3 COMMUNITY LICENSE AGREEMENT
|
| 2 |
+
MiniMax H3 release date/License date: August 2, 2026.
|
| 3 |
+
The scope of this License Agreement (this “Agreement”) is expressly limited to the “Applicable Territory” as defined below.
|
| 4 |
+
By clicking to accept, or by using, reproducing, modifying, distributing, running, or displaying any portion or element of the MiniMax H3 Works (including through any Hosted Services) in any manner, you acknowledge and accept the terms of this Agreement, and this Agreement shall take immediate effect upon the occurrence of such act.
|
| 5 |
+
I. Definitions
|
| 6 |
+
1. “Acceptable Use Policy” means the policy published by MiniMax in Exhibit A.
|
| 7 |
+
2. “Agreement” means the terms and conditions set forth herein that govern the use, reproduction, distribution, modification, running, and display of the MiniMax H3 Works or any portion or element thereof.
|
| 8 |
+
3. “Applicable Territory” means worldwide, excluding the Excluded Territories.
|
| 9 |
+
4. “Documentation” means the specifications, manuals, and documentation concerning MiniMax H3 that are publicly released by MiniMax.
|
| 10 |
+
5. “Excluded Territories” means the European Union, the United Kingdom, the Republic of Korea and the United States of America.
|
| 11 |
+
6. “MiniMax H3” means the video generation model, together with its software and algorithms, including trained model weights, parameters (including optimizer states), machine-learning model code, inference-supporting code, and other elements thereof made publicly available by Us, as released at https://huggingface.co/MiniMaxAI/MiniMax-H3.
|
| 12 |
+
7. “MiniMax H3 Works” means (i) the Materials, (ii) the Model Derivatives, and (iii) all derivatives thereof.
|
| 13 |
+
8. “Hosted Services” means hosted services provided via application programming interfaces (APIs), web access, or any other electronic or remote means.
|
| 14 |
+
9. “Licensee,” “you,” or “your” means the natural or legal person exercising rights and/or using the MiniMax H3 Works for any purpose in any field of use under this Agreement.
|
| 15 |
+
10. “Materials” means, collectively, MiniMax H3 and the Documentation (and any portion thereof), in each case as made available by MiniMax under this Agreement and proprietary to MiniMax.
|
| 16 |
+
11. “Model Derivatives” means all of the following: (i) any modification of MiniMax H3 or any Model Derivative thereof; (ii) any work based on MiniMax H3 or any Model Derivative thereof; or (iii) any other machine learning model created by transferring the patterns of the weights, parameters, operational patterns, or Outputs of MiniMax H3 or any Model Derivative thereof to another model, such that the latter model exhibits behavior similar to MiniMax H3 or its Model Derivatives, including by distillation methods, methods using intermediate data representations, or methods based on training using synthetic-data Outputs generated by MiniMax H3 or its Model Derivatives. For the avoidance of doubt, Outputs are not deemed Model Derivatives.
|
| 17 |
+
12. “Output” means any result of operating or otherwise using MiniMax H3 or any Model Derivatives (including through Hosted Services).
|
| 18 |
+
13. “Third Party” means any natural or legal person that is not under common control with us or with you.
|
| 19 |
+
14. “Including” means “including but not limited to.”
|
| 20 |
+
15. “We,” “Us” or “MiniMax” means Nanonoble Pte. Ltd..
|
| 21 |
+
II. Grant of Rights
|
| 22 |
+
Solely within the Applicable Territory, we grant you a non-exclusive, non-transferable, royalty-free, limited license to use, reproduce, distribute, create derivative works (including Model Derivatives), and modify the Materials in accordance with the terms of this Agreement and the Acceptable Use Policy, based on the intellectual property and other rights owned by MiniMax that are embodied in or used by the Materials. You shall not violate (or encourage or permit any person to violate) any term of this Agreement or the Acceptable Use Policy.
|
| 23 |
+
We will continuously evaluate the applicable laws, regulations and compliance requirements for the Excluded Territories. In the meantime, should any person in such Excluded Territories be interested in deploying our models, you are welcome to contact us about obtaining a license, which will be granted based on robust controls and guardrails for purposes of complying with the laws, regulations and compliance requirements of the Excluded Territories.
|
| 24 |
+
III. Distribution and Redistribution
|
| 25 |
+
Subject to and conditioned on your continuing compliance with this Agreement, including its territorial restrictions and the Acceptable Use Policy, and solely within the Applicable Territory, you may distribute or make available the MiniMax H3 Works to Third Parties within the Applicable Territory; provided, that all of the following conditions are met:
|
| 26 |
+
1. You must provide a copy of this Agreement to all such Third Parties who receive the MiniMax H3 Works or use your products or services related thereto;
|
| 27 |
+
2. You must cause any modified files to carry prominent notices stating that you have modified such files;
|
| 28 |
+
3. You are encouraged to:
|
| 29 |
+
a. display a notice on any product or service developed using MiniMax H3 indicating that the product or service is “Powered by MiniMax H3”;
|
| 30 |
+
b. add an AI-generation identifier to files produced using generative AI models including MiniMax H3; and
|
| 31 |
+
c. publish at least one technical blog post or a public statement describing your experience using MiniMax H3 Works;
|
| 32 |
+
4. All distributions to Third Parties (other than through Hosted Services) must be accompanied by a “NOTICE” text file containing the following notice:
|
| 33 |
+
“MiniMax H3 is licensed under the MiniMax H3 Community License Agreement, Copyright © 2026 MiniMax. All Rights Reserved.”
|
| 34 |
+
You may add your own copyright notices on your modifications; except as provided in this Section and in Section V, however, you may not impose additional or different terms and conditions on the use, reproduction, or distribution of your modifications or of any aggregate Model Derivatives, and your use, reproduction, modification, distribution, running, and display of the work must otherwise comply with the terms and conditions of this Agreement (including the provisions concerning the Applicable Territory). If you receive the MiniMax H3 Works from a Licensee as part of an integrated end-user product, the provisions of Section III of this Agreement do not apply to you, but Section V and Exhibit A remain applicable.
|
| 35 |
+
IV. Additional Commercial Terms
|
| 36 |
+
1. You shall obtain a separate, prior written authorization from MiniMax by contacting api@minimax.io with the subject line “MiniMax H3 licensing - authorization request”, if your commercial products and services generate more than 20 million US dollars (or equivalent in other currencies) in yearly revenue.
|
| 37 |
+
2. You shall prominently display “MiniMax H3”on the user interface of commercial product or service that uses MiniMax H3 or MiniMax H3 Works.
|
| 38 |
+
V. Use Restrictions
|
| 39 |
+
1. Your use of the MiniMax H3 Works must comply with applicable laws and regulations (including trade-compliance laws and regulations) and must comply with the Acceptable Use Policy for the MiniMax H3 Works, which is incorporated into this Agreement by reference.
|
| 40 |
+
2. Before providing access to the MiniMax H3 Works or any product, service, or Hosted Service incorporating them, you must bind each recipient or user to enforceable terms at least as protective as the use restrictions in this Section V and Exhibit A, and you must notify each recipient or user that those restrictions apply.
|
| 41 |
+
3. You may not use the MiniMax H3 Works or any of their Outputs or results to improve any other artificial intelligence model (other than MiniMax H3 or its Model Derivatives).
|
| 42 |
+
4. You may not use, reproduce, modify, distribute, or display the MiniMax H3 Works or any of their Outputs or results outside the Applicable Territory. Any such use outside the Applicable Territory is not authorized by this Agreement.
|
| 43 |
+
5. If you provide or make available to any Third Party a product, service, or Hosted Service that permits the generation of Outputs using MiniMax H3 or any Model Derivative, you must, before making that product or service available and throughout its operation, implement, maintain, test, and periodically review reasonable and proportionate technical and organizational safeguards designed to prevent and mitigate access, uses, and Outputs that violate this Section V or Exhibit A, including uses or Outputs that infringe, misappropriate, or otherwise violate any Third Party’s intellectual-property or other rights. You must not knowingly disable, materially weaken, or permit the circumvention of those safeguards. You must maintain a reasonably accessible mechanism for reporting suspected violations. Upon receiving a good-faith report or otherwise obtaining actual knowledge of a violation, you must promptly investigate and take reasonable steps within your control to stop or mitigate the violation, including removing or disabling access to offending content or services and suspending or terminating repeat violators where appropriate. You are responsible for implementing and enforcing these requirements with respect to your products, services, systems, users, and downstream recipients.
|
| 44 |
+
VI. Intellectual Property
|
| 45 |
+
1. Subject to MiniMax’s rights in the MiniMax H3 Works (and the intellectual property therein), and to your compliance with the terms and conditions of this Agreement, as between you and MiniMax, you will own the derivative works and modifications of the Materials that you have created or had created, as well as any Model Derivatives.
|
| 46 |
+
2. Except for the limited license expressly granted in this paragraph, no trademark license is granted under this Agreement; with respect to MiniMax H3 Works, the Licensee may not use any name or mark owned by or associated with MiniMax or any of its affiliates, except as reasonably and customarily necessary to describe and distribute the MiniMax H3 Works. MiniMax hereby grants you a license to use the “MiniMax H3” mark (the “Mark”) within the Applicable Territory solely for the purpose of complying with Section III.3; provided, that you comply with all applicable trademark-protection laws. All goodwill arising from your use of the Mark shall inure to the benefit of MiniMax.
|
| 47 |
+
3. If you bring or assert any suit or other legal proceeding (including a cross-claim or counterclaim in any action) against us or any other natural or legal person alleging that the Materials, any Output, or any portion of the foregoing infringes any intellectual property right or other right owned by you or for which you can obtain a license, all licenses granted to you under this Agreement will terminate as of the date such suit or proceeding is filed. You shall defend, indemnify, and hold us harmless against any Third-Party claim arising out of or related to the use or distribution of the MiniMax H3 Works by you or by any Third Party.
|
| 48 |
+
4. MiniMax claims no rights over the Outputs you generate. You and your users are entirely responsible for the Outputs and any subsequent use thereof.
|
| 49 |
+
VII. Disclaimers and Limitations of Liability
|
| 50 |
+
1. We have no obligation to support, update, provide training for, or develop any further version of the MiniMax H3 Works, or to grant any license with respect thereto.
|
| 51 |
+
2. UNLESS AND ONLY TO THE EXTENT REQUIRED BY APPLICABLE LAW, THE MINIMAX H3 WORKS AND ANY OUTPUT AND RESULTS THEREFROM ARE PROVIDED “AS IS” WITHOUT ANY EXPRESS OR IMPLIED WARRANTIES OF ANY KIND INCLUDING ANY WARRANTIES OF TITLE, MERCHANTABILITY, NONINFRINGEMENT, COURSE OF DEALING, USAGE OF TRADE, OR FITNESS FOR A PARTICULAR PURPOSE. YOU ARE SOLELY RESPONSIBLE FOR DETERMINING THE APPROPRIATENESS OF USING, REPRODUCING, MODIFYING, PERFORMING, DISPLAYING OR DISTRIBUTING ANY OF THE MINIMAX H3 WORKS OR OUTPUTS AND ASSUME ANY AND ALL RISKS ASSOCIATED WITH YOUR OR A THIRD PARTY’S USE OR DISTRIBUTION OF ANY OF THE MINIMAX H3 WORKS OR OUTPUTS AND YOUR EXERCISE OF RIGHTS AND PERMISSIONS UNDER THIS AGREEMENT.
|
| 52 |
+
3. TO THE FULLEST EXTENT PERMITTED BY APPLICABLE LAW, IN NO EVENT SHALL MINIMAX OR ITS AFFILIATES BE LIABLE UNDER ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, TORT, NEGLIGENCE, PRODUCTS LIABILITY, OR OTHERWISE, FOR ANY DAMAGES, INCLUDING ANY DIRECT, INDIRECT, SPECIAL, INCIDENTAL, EXEMPLARY, CONSEQUENTIAL OR PUNITIVE DAMAGES, OR LOST PROFITS OF ANY KIND ARISING FROM THIS AGREEMENT OR RELATED TO ANY OF THE MINIMAX H3 WORKS OR OUTPUTS, EVEN IF MINIMAX OR ITS AFFILIATES HAVE BEEN ADVISED OF THE POSSIBILITY OF ANY OF THE FOREGOING.
|
| 53 |
+
VIII. Term and Termination
|
| 54 |
+
1. This Agreement is effective from the moment you accept this Agreement or begin accessing the Materials, and, subject to your compliance with its terms and conditions, will remain in effect until terminated as provided herein.
|
| 55 |
+
2. If you breach any term or condition of this Agreement, we have the right to terminate this Agreement. Upon termination, you must immediately cease accessing, using, and distributing the MiniMax H3 Works; delete or destroy all copies within your possession or control; and notify each downstream recipient that your authorization has ended. The obligations in the preceding sentence and Sections VI.1, VI.3, VII, and IX survive termination.
|
| 56 |
+
IX. Governing Law and Jurisdiction
|
| 57 |
+
1. This Agreement, and any dispute arising out of or related to this Agreement, shall be governed by the laws of the Hong Kong Special Administrative Region of the People’s Republic of China, without regard to its conflict-of-laws rules. The United Nations Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
|
| 58 |
+
2. Any dispute arising out of or related to this Agreement shall be subject to the exclusive jurisdiction of the courts of the Hong Kong Special Administrative Region of the People’s Republic of China with competent jurisdiction. Both MiniMax and the Licensee hereby consent to the exclusive jurisdiction of such courts for any such dispute.
|
| 59 |
+
Additional Note: Please note that the encoder of MiniMax H3 uses Qwen3-VL-32B, which is licensed under Apache 2.0 License: https://github.com/QwenLM/Qwen3-VL/blob/main/LICENSE.
|
| 60 |
+
|
| 61 |
+
Exhibit A — Acceptable Use Policy
|
| 62 |
+
MiniMax reserves the right to update this Acceptable Use Policy from time to time.
|
| 63 |
+
Last revised: August 2, 2026.
|
| 64 |
+
MiniMax is committed to promoting the safe and fair use of its tools and features, including MiniMax H3. You agree not to use MiniMax H3, any Model Derivatives, or any Output in any of the following ways:
|
| 65 |
+
1. Use outside the Applicable Territory;
|
| 66 |
+
2. Use in any manner that violates any applicable national, federal, state, local, or international law, regulation, or other legal requirement, or that infringes, misappropriates, or otherwise violates any Third Party’s intellectual-property or other proprietary rights, including through unauthorized reproduction, distribution, public display, public performance, or creation of derivative works;
|
| 67 |
+
3. Use in any manner that may harm yourself or others;
|
| 68 |
+
4. Use to repurpose or distribute the Outputs of MiniMax H3 or any Model Derivatives in order to harm yourself or others;
|
| 69 |
+
5. Use to circumvent or bypass any safety guardrails or safeguards we have implemented;
|
| 70 |
+
6. Use in any manner that exploits or harms, or intends to exploit or harm, minors;
|
| 71 |
+
7. Use to generate or disseminate verifiably false information and/or content for the purpose of harming others or influencing elections;
|
| 72 |
+
8. Use to manufacture or facilitate false online engagement, including fake reviews and other means of false online engagement;
|
| 73 |
+
9. Use to intentionally defame, disparage, or otherwise harass others;
|
| 74 |
+
10. Use to generate and/or disseminate malware (including ransomware) or any other content intended to damage electronic systems;
|
| 75 |
+
11. Use to generate or disseminate personally identifiable information for the purpose of harming others;
|
| 76 |
+
12. Use to generate or disseminate information (including images, code, posts, or articles) in or to any public environment (including via bot tweets or similar means) without clearly and prominently disclosing that such information and/or content is machine-generated;
|
| 77 |
+
13. Use to impersonate another person without that person’s consent, authorization, or lawful right to do so;
|
| 78 |
+
14. Use to make high-risk automated decisions in critical domains that affect individual safety, rights, or well-being (such as law enforcement, immigration, healthcare or medical services, critical-infrastructure management, product-safety components, essential services, credit, employment, housing, education, social scoring, or insurance);
|
| 79 |
+
15. Use in any manner that violates or disregards the social, ethical, or moral standards of other countries or regions;
|
| 80 |
+
16. Use to carry out, assist, threaten, incite, plan, advocate for, or encourage violent extremism or terrorism;
|
| 81 |
+
17. Use for any purpose intended to discriminate against, or harm, individuals or groups based on protected characteristics or categories, online or offline social behavior, or known or predicted personality traits;
|
| 82 |
+
18. Use to intentionally exploit the vulnerabilities of specific populations based on age, social, physical, or psychological characteristics, so as to materially distort the behavior of a member of that group in a manner that causes, or is likely to cause, physical or psychological harm to that person or to others;
|
| 83 |
+
19. Use for military purposes;
|
| 84 |
+
20. Use to engage in any unauthorized or unlicensed professional activity, including but not limited to financial, legal, medical or healthcare, or other professional practice.
|
LICENSE-CODE
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright [yyyy] [name of copyright owner]
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
MODIFICATIONS.md
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Modified files
|
| 2 |
+
|
| 3 |
+
Section III.2 of the MiniMax H3 Community License Agreement requires that modified
|
| 4 |
+
files carry a prominent notice saying so. This file is that notice.
|
| 5 |
+
|
| 6 |
+
Everything below is derived from
|
| 7 |
+
[`MiniMaxAI/MiniMax-H3`](https://huggingface.co/MiniMaxAI/MiniMax-H3).
|
| 8 |
+
|
| 9 |
+
## `transformer/` — modified
|
| 10 |
+
|
| 11 |
+
**Every weight file in `transformer/` has been modified.** It started as the base
|
| 12 |
+
model's `transformer_ref/` (the `ref2va` transformer, 33.1 B parameters) and every
|
| 13 |
+
parameter was updated by a full finetune on a character-replacement objective. The
|
| 14 |
+
architecture, `config.json` and tensor names are unchanged, so it is a drop-in
|
| 15 |
+
replacement for the base `transformer_ref/`; the numbers in it are not the base
|
| 16 |
+
model's numbers.
|
| 17 |
+
|
| 18 |
+
The file layout also differs: the finetune was written as one 61.7 GiB safetensors
|
| 19 |
+
file and re-sharded here into 14 parts, because HuggingFace rejects single files
|
| 20 |
+
above 50 GB. The 638 tensors and their contents are unchanged by that re-sharding.
|
| 21 |
+
|
| 22 |
+
## `lora/pytorch_lora_weights.safetensors` — new
|
| 23 |
+
|
| 24 |
+
Not a MiniMax file. A rank-128 LoRA over 302 linear layers of `transformer/`,
|
| 25 |
+
trained by us with DMD2 distillation. It is a delta on the finetuned transformer
|
| 26 |
+
above, not on the base model — loading it onto stock `transformer_ref/` produces
|
| 27 |
+
garbage.
|
| 28 |
+
|
| 29 |
+
## `assets/fixed_embed_fwd_anyframe.pt` — new
|
| 30 |
+
|
| 31 |
+
Not a MiniMax file. A frozen 362 × 5120 text-conditioning tensor we computed once
|
| 32 |
+
with the base model's own text encoder, so that inference never has to load
|
| 33 |
+
Qwen3-VL. It is an *output* of the base model's encoder in the sense of Section
|
| 34 |
+
I.12, computed from the prompt in `assets/fixed_prompt.txt`.
|
| 35 |
+
|
| 36 |
+
## `assets/fixed_prompt.txt` — new
|
| 37 |
+
|
| 38 |
+
Not a MiniMax file. The prompt text the tensor above was computed from, included so
|
| 39 |
+
that what conditions every render is readable rather than opaque.
|
| 40 |
+
|
| 41 |
+
## `inference/sample.py`, `examples/demo.sh` — new
|
| 42 |
+
|
| 43 |
+
Not MiniMax files. Written by us against the public `diffusers` API.
|
| 44 |
+
|
| 45 |
+
## `LICENSE-CODE`, `NOTICE` — new
|
| 46 |
+
|
| 47 |
+
Not MiniMax files. `LICENSE-CODE` is the Apache 2.0 text, and it covers `inference/` and
|
| 48 |
+
`examples/` only — each file there carries an `SPDX-License-Identifier: Apache-2.0` header.
|
| 49 |
+
`NOTICE` records that the weights are *not* Apache 2.0. `LICENSE` is the MiniMax H3
|
| 50 |
+
Community License Agreement itself, included unmodified as Section III.1 requires.
|
| 51 |
+
|
| 52 |
+
## `examples/media/` — new
|
| 53 |
+
|
| 54 |
+
Two kinds of media here, with different provenance.
|
| 55 |
+
|
| 56 |
+
**The demo** — `reference.png`, `output.mp4`, `before-after.png` — derives from
|
| 57 |
+
`assets/ref2va.mp4`, a video MiniMax published with the base model. That clip is itself a
|
| 58 |
+
MiniMax H3 generation rather than camera footage. All three files are a 512 × 768 portrait
|
| 59 |
+
crop of it (`crop=512:768:389:0`, no scaling):
|
| 60 |
+
|
| 61 |
+
- `reference.png` — the crop's first frame with the young man repainted as an invented
|
| 62 |
+
elderly woman. Produced with OpenAI's `gpt-image-2`; the character is fictional and is
|
| 63 |
+
not a real person or an existing property.
|
| 64 |
+
- `output.mp4` — that reference propagated across 124 frames by this model.
|
| 65 |
+
- `before-after.png` — frames from the driving crop above frames from the output.
|
| 66 |
+
|
| 67 |
+
The driving clip itself is **not** bundled. `examples/demo.sh` rebuilds it, with the
|
| 68 |
+
documented crop, from your own copy of the base model.
|
| 69 |
+
|
| 70 |
+
**The comparison clips** — `compare-prop.mp4`, `compare-costume.mp4`, `compare-style.mp4` — are a
|
| 71 |
+
different matter. Their driving videos are
|
| 72 |
+
real filmed footage that we hold the rights to, and they are among the clips this model was
|
| 73 |
+
evaluated against. Each file is a four-panel stack: painted reference, driving video, this model,
|
| 74 |
+
Wan2.2-Animate-14B. The Wan2.2-Animate panels were rendered by us from the official
|
| 75 |
+
[`Wan-AI/Wan2.2-Animate-14B`](https://huggingface.co/Wan-AI/Wan2.2-Animate-14B) weights and code,
|
| 76 |
+
unmodified, at the replacement-mode settings its own README documents (20 steps, `sample_shift 5.0`,
|
| 77 |
+
`--refert_num 1 --replace_flag --use_relighting_lora`, preprocessing at `--w_len 1 --h_len 1`) —
|
| 78 |
+
that panel is Wan's output, not ours.
|
| 79 |
+
|
| 80 |
+
Wan's `generate.py` hardcodes a 30 fps output timebase regardless of the source, so its raw files
|
| 81 |
+
claim 4.10 s for motion that is 24 fps. The panels are retimed (`setpts`), not resampled, so no
|
| 82 |
+
frames are dropped and both models play at the same speed. Every panel is letterboxed into the
|
| 83 |
+
driving clip's own geometry; nothing is stretched.
|
| 84 |
+
|
| 85 |
+
## Not redistributed here
|
| 86 |
+
|
| 87 |
+
The VAE, audio VAE, schedulers, text encoder, tokenizer and processor are **not**
|
| 88 |
+
included in this repository and are not modified. They are loaded at runtime from
|
| 89 |
+
your own copy of `MiniMaxAI/MiniMax-H3`, which you must download separately and
|
| 90 |
+
under its own license terms.
|
MiniMaxH3TokenRefinerBlock/package.pt2
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1fc10961935eb10891d1e2449b18185533d5714a40bb6f6929b8d6b7c1b4d39f
|
| 3 |
+
size 604596
|
MiniMaxH3TransformerBlock/package.pt2
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:58c651aa49a7455aa6a49e17ae7c74b4fd73e9306456a1baaa3300d0aef1ee09
|
| 3 |
+
size 853072
|
NOTICE
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
MiniMax H3 is licensed under the MiniMax H3 Community License Agreement,
|
| 2 |
+
Copyright © 2026 MiniMax. All Rights Reserved.
|
| 3 |
+
|
| 4 |
+
---
|
| 5 |
+
|
| 6 |
+
Viggle-Animate is a Model Derivative of MiniMax H3, as that term is defined in
|
| 7 |
+
Section I.11 of the MiniMax H3 Community License Agreement. It is distributed
|
| 8 |
+
under that same Agreement, a copy of which is included in this repository as
|
| 9 |
+
LICENSE. See MODIFICATIONS.md for the list of files that were modified.
|
| 10 |
+
|
| 11 |
+
Powered by MiniMax H3.
|
| 12 |
+
|
| 13 |
+
Modifications and additions Copyright © 2026 Viggle AI.
|
README.md
ADDED
|
@@ -0,0 +1,347 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: other
|
| 3 |
+
license_name: minimax-h3-community-license
|
| 4 |
+
license_link: LICENSE
|
| 5 |
+
base_model: MiniMaxAI/MiniMax-H3
|
| 6 |
+
pipeline_tag: video-to-video
|
| 7 |
+
tags:
|
| 8 |
+
- video-editing
|
| 9 |
+
- character-replacement
|
| 10 |
+
- video-to-video
|
| 11 |
+
- distillation
|
| 12 |
+
- dmd
|
| 13 |
+
---
|
| 14 |
+
|
| 15 |
+
# Viggle-Animate
|
| 16 |
+
|
| 17 |
+
### Character Replacement in Video from a Single Repainted Frame
|
| 18 |
+
|
| 19 |
+
**[Try the demo](https://huggingface.co/spaces/Viggle/viggle-animate)** ·
|
| 20 |
+
**[Research write-up](https://viggle.ai/research/viggle-animate-character-replacement-from-a-repainted-frame?utm_source=huggingface&utm_medium=social&utm_campaign=viggle-animate&utm_content=viggle/viggle-animate)** ·
|
| 21 |
+
**[ComfyUI nodes](https://github.com/Saganaki22/ComfyUI-Viggle-Animate-H3)** ·
|
| 22 |
+
**[viggle.ai/h3](https://viggle.ai/h3)** ·
|
| 23 |
+
Built on **[MiniMaxAI/MiniMax-H3](https://huggingface.co/MiniMaxAI/MiniMax-H3)**
|
| 24 |
+
|
| 25 |
+
<a href="https://x.com/cocktailpeanut/status/2097332291844399514" style="display:block;max-width:460px;margin:0 0 .5em">
|
| 26 |
+
<img src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/community-tweet.png" width="100%" style="display:block;border-radius:14px" alt="@cocktailpeanut on X: two AI CEOs dropped into a film scene with Viggle-Animate - 1.5M views, 14K likes">
|
| 27 |
+
</a>
|
| 28 |
+
|
| 29 |
+
<sub>Made with Viggle-Animate by <a href="https://x.com/cocktailpeanut/status/2097332291844399514">@cocktailpeanut</a>, not by us. Click through to watch it on X.</sub>
|
| 30 |
+
|
| 31 |
+
<video autoplay controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/teaser.mp4"></video>
|
| 32 |
+
|
| 33 |
+
**Viggle-Animate replaces the character in a video with whatever you paint into one of its own
|
| 34 |
+
frames** — motion, camera and timing untouched. You prepare that one frame in an image editor;
|
| 35 |
+
from there the video stage runs no pose estimator, segmenter, face tracker or text encoder. Two
|
| 36 |
+
inputs, three forward passes, 26 seconds a shot on one GPU.
|
| 37 |
+
|
| 38 |
+
**It is strongest where replacement is hardest: fast motion, and pose transfer accurate enough to
|
| 39 |
+
follow it.** Whipping heads, full kicks, jumps — tracked frame for frame, not smeared through.
|
| 40 |
+
|
| 41 |
+
## Abstract
|
| 42 |
+
|
| 43 |
+
Controlled character replacement is usually built on intermediate representations — pose skeletons,
|
| 44 |
+
segmentation masks, background plates, face crops. Each needs its own extractor, and each extractor
|
| 45 |
+
is another model to run and another place to lose information. Recent work drops the skeleton but
|
| 46 |
+
keeps a mask channel. **Viggle-Animate uses neither.** Its two inputs are a driving video and one of
|
| 47 |
+
that video's own frames with the character repainted, and its only task is to propagate that edit
|
| 48 |
+
across the shot.
|
| 49 |
+
|
| 50 |
+
Because the reference is a frame of the clip, its pose, camera, framing and lighting already agree
|
| 51 |
+
with the footage, and nothing downstream has to align them again. The model is never told what the
|
| 52 |
+
new character is: no class, no identity encoder, and no user-provided text prompt.
|
| 53 |
+
**Viggle-Animate is a 33.1 B full finetune of MiniMax-H3's `ref2va` transformer, jointly distilled
|
| 54 |
+
with DMD to three forward passes.** In a matched comparison on the same machine and B200 GPU, using
|
| 55 |
+
the same source videos, output resolution and frame count, it renders 124 frames in 26 s, 6.1×
|
| 56 |
+
faster per clip than Wan2.2-Animate-14B.
|
| 57 |
+
|
| 58 |
+
## Method
|
| 59 |
+
|
| 60 |
+
Character replacement asks two questions at once: *what does the new character look like*, and *how
|
| 61 |
+
does it move through this shot*. Systems that condition on a standalone character photograph must
|
| 62 |
+
answer both, and reconciling a photograph with footage it was never part of is what the scaffolding
|
| 63 |
+
exists for.
|
| 64 |
+
|
| 65 |
+
State-of-the-art image models have finished that job. Give `gpt-image` a frame and an instruction
|
| 66 |
+
and it replaces the character while following the prompt exactly — transferring the pose, matching
|
| 67 |
+
the lighting, preserving the background. The hard reconciliation is already solved, once, on one
|
| 68 |
+
image. This model is the second half of that pipeline, not the whole of it.
|
| 69 |
+
|
| 70 |
+
<img src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/pipeline.svg"
|
| 71 |
+
alt="Two inputs — a driving video and one of its own frames, repainted in any image editor — enter Viggle-Animate. No pose skeleton, segmentation mask, face crop, background plate, depth map or user-provided text prompt enters the video model." width="100%">
|
| 72 |
+
|
| 73 |
+
Appearance enters only through the repainted frame; geometry enters only through the driving video.
|
| 74 |
+
**The text encoder is never loaded.** Conditioning is one frozen embedding shipped with the weights
|
| 75 |
+
([`assets/fixed_prompt.txt`](assets/fixed_prompt.txt)), identical for every render.
|
| 76 |
+
|
| 77 |
+
Left panel is the driving video, right panel is this model:
|
| 78 |
+
|
| 79 |
+
<video autoplay controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/hero-duo.mp4"></video>
|
| 80 |
+
|
| 81 |
+
<video autoplay controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/hero-corgi.mp4"></video>
|
| 82 |
+
|
| 83 |
+
**It is fast twice over.** Once the repainted frame exists there is nothing else to run — no pose
|
| 84 |
+
estimator, no segmenter, no face tracker, no text encoder. And the sampler is distilled, so a
|
| 85 |
+
finished clip is three forward passes rather than thirty. The two compound: one model, one GPU,
|
| 86 |
+
and no orchestration to get wrong.
|
| 87 |
+
|
| 88 |
+
The distillation is **joint, across two teachers split by noise level.** Our finetune supervises the
|
| 89 |
+
high-noise end of the schedule, where the replacement itself is decided — it is the model that gets
|
| 90 |
+
the swap right. The original MiniMax-H3 supervises the low-noise end, where detail and texture are
|
| 91 |
+
decided — it is the model with the better image quality. Distilling each end against the teacher that
|
| 92 |
+
owns it keeps both properties in one student, instead of inheriting the finetune's visual
|
| 93 |
+
regressions along with its replacement ability.
|
| 94 |
+
|
| 95 |
+
**It generalizes past humans**, because nothing in the loop assumes one. A pose skeleton has a neck
|
| 96 |
+
and two arms; a mask has a person-shaped hole. We have neither, so the model holds no representation
|
| 97 |
+
that a character must be a person. What it can animate is bounded by what you can paint.
|
| 98 |
+
|
| 99 |
+
## Efficiency
|
| 100 |
+
|
| 101 |
+
<div align="center">
|
| 102 |
+
<div style="display:flex;gap:10px;flex-wrap:wrap;justify-content:center;margin:20px 0 6px;text-align:left">
|
| 103 |
+
|
| 104 |
+
<div style="flex:1 1 170px;border:1px solid rgba(128,128,128,.35);border-top:3px solid #00D94F;border-radius:10px;padding:12px 14px">
|
| 105 |
+
<div style="font-size:1.7em;font-weight:700;line-height:1.15">26 s</div>
|
| 106 |
+
<b>per render</b><br>
|
| 107 |
+
<small style="opacity:.7">124 frames at 24 fps, 480×832, a single B200</small>
|
| 108 |
+
</div>
|
| 109 |
+
|
| 110 |
+
<div style="flex:1 1 170px;border:1px solid rgba(128,128,128,.35);border-top:3px solid #00D94F;border-radius:10px;padding:12px 14px">
|
| 111 |
+
<div style="font-size:1.7em;font-weight:700;line-height:1.15">3</div>
|
| 112 |
+
<b>forward passes</b><br>
|
| 113 |
+
<small style="opacity:.7"><code>--steps 4</code> names four sigma boundaries, so three passes</small>
|
| 114 |
+
</div>
|
| 115 |
+
|
| 116 |
+
<div style="flex:1 1 170px;border:1px solid rgba(128,128,128,.35);border-top:3px solid #00D94F;border-radius:10px;padding:12px 14px">
|
| 117 |
+
<div style="font-size:1.7em;font-weight:700;line-height:1.15">2</div>
|
| 118 |
+
<b>inputs</b><br>
|
| 119 |
+
<small style="opacity:.7">a clip, and one of its own frames repainted</small>
|
| 120 |
+
</div>
|
| 121 |
+
|
| 122 |
+
<div style="flex:1 1 170px;border:1px solid rgba(128,128,128,.35);border-top:3px solid #00D94F;border-radius:10px;padding:12px 14px">
|
| 123 |
+
<div style="font-size:1.7em;font-weight:700;line-height:1.15">0</div>
|
| 124 |
+
<b>other models</b><br>
|
| 125 |
+
<small style="opacity:.7">in the video stage — no pose estimator, segmenter, face tracker or text encoder</small>
|
| 126 |
+
</div>
|
| 127 |
+
|
| 128 |
+
</div>
|
| 129 |
+
</div>
|
| 130 |
+
|
| 131 |
+
One B200, 480×832, 124 frames at 24 fps, bf16, no compile, no offload. Wan ran its documented
|
| 132 |
+
replacement recipe — 20 steps, `sample_shift 5.0`, `--refert_num 1 --replace_flag
|
| 133 |
+
--use_relighting_lora`, `--w_len 1 --h_len 1` — after its own preprocessing pass.
|
| 134 |
+
|
| 135 |
+
| | Viggle-Animate | Wan2.2-Animate-14B |
|
| 136 |
+
|---|---|---|
|
| 137 |
+
| Inputs | driving video + one repainted frame | driving video + character image, then a **preprocessing pass** producing pose, face, mask and background tracks |
|
| 138 |
+
| Render, after weights load | **26 s** | 160 s |
|
| 139 |
+
| — of which sampling | **13.6 s** | 140 s |
|
| 140 |
+
| Forward passes | **3** | 40 (20 steps × 2 chunks) |
|
| 141 |
+
| Parameters | 33.1 B | 17.3 B |
|
| 142 |
+
|
| 143 |
+
**6.1× faster per render, 10.3× on sampling alone.** Wan's preprocessing pass is not counted in
|
| 144 |
+
its 160 s.
|
| 145 |
+
|
| 146 |
+
### Qualitative comparison
|
| 147 |
+
|
| 148 |
+
Four panels each: **painted reference · driving video · this model · Wan2.2-Animate-14B**, the last
|
| 149 |
+
at its documented replacement settings. The gap is widest under fast motion: where the comparison
|
| 150 |
+
smears, this model stays sharp and lands the pose on the right frame.
|
| 151 |
+
|
| 152 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/compare-fastmotion.mp4"></video>
|
| 153 |
+
|
| 154 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/compare-highkick.mp4"></video>
|
| 155 |
+
|
| 156 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/compare-cosplay.mp4"></video>
|
| 157 |
+
|
| 158 |
+
## Generalization
|
| 159 |
+
|
| 160 |
+
The model is never told what it is animating, so how far the character can get from a person is an
|
| 161 |
+
empirical question rather than a list of supported categories. Three panels each: **painted
|
| 162 |
+
reference · driving video · this model.** Every clip below is one paint and one render at the
|
| 163 |
+
shipped defaults, `--seed 42` — no best-of-N.
|
| 164 |
+
|
| 165 |
+
**Animals.** Ears, eye patches and flippers move on limbs the driving clip does not have — the paint
|
| 166 |
+
places them, the render animates them as if they had always been arms and a head.
|
| 167 |
+
|
| 168 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-corgi.mp4"></video>
|
| 169 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-panda.mp4"></video>
|
| 170 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-penguin.mp4"></video>
|
| 171 |
+
|
| 172 |
+
**Not humanoid.** The airliner is the hardest case we have: the paint binds wings to arms and landing
|
| 173 |
+
gear to legs, and the model's job is to keep that binding for 124 frames. The robot has to relight
|
| 174 |
+
specular metal as it turns.
|
| 175 |
+
|
| 176 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-airliner.mp4"></video>
|
| 177 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-robot.mp4"></video>
|
| 178 |
+
|
| 179 |
+
**Stylized.** The clay figure holds its style boundary for the whole clip.
|
| 180 |
+
|
| 181 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-claymation.mp4"></video>
|
| 182 |
+
|
| 183 |
+
**More than one character.** The work moves into the paint prompt, which has to bind each one to a
|
| 184 |
+
position — "the one on the left". The last clip is a wide arena shot, each figure a few dozen pixels
|
| 185 |
+
tall.
|
| 186 |
+
|
| 187 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-anime-duo.mp4"></video>
|
| 188 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-photoreal-duo.mp4"></video>
|
| 189 |
+
<video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-wushu-animals.mp4"></video>
|
| 190 |
+
|
| 191 |
+
## Limitations
|
| 192 |
+
|
| 193 |
+
**It inherits the image edit.** What the paint does not show, the model will not add, and where
|
| 194 |
+
paint and video disagree the video wins. Appearance comes from the paint but shape comes from the
|
| 195 |
+
driving pose: a LEGO minifigure kept its palette and yellow claw hands, yet reverted to human
|
| 196 |
+
anatomy — the airliner held because the paint tied its wings to real arms.
|
| 197 |
+
|
| 198 |
+
~~**Lip-sync is weak.** Mouth shapes do not track speech closely in close-ups. Identity and expression
|
| 199 |
+
hold; it is the sync that lags, and we believe that is a training-data limit rather than anything
|
| 200 |
+
structural.~~
|
| 201 |
+
|
| 202 |
+
**Retracted — and since improved.** Broader user testing already put lip-sync and facial
|
| 203 |
+
expression ahead of our own first read of them. The rest turned out not to be a training-data limit
|
| 204 |
+
either. The model animates the mouth to *its own* audio track, and the fixed prompt asks that track
|
| 205 |
+
to be silence — so in the shipped configuration there is nothing for the mouth to sync to. Give it a
|
| 206 |
+
real soundtrack instead: encode the driving clip's audio with the audio VAE and hold it in the
|
| 207 |
+
target audio rows as a **clean** latent for the whole denoise (H3's flow convention is reversed, so
|
| 208 |
+
clean is `t = 1`). The audio rows stop being something the model predicts and become something it
|
| 209 |
+
conditions on, and the mouth tracks that speech. This is an inference-time change only — no
|
| 210 |
+
retraining, no new weights — and the pairing of noisy video rows with clean audio rows is not a
|
| 211 |
+
state the model was trained on, but it holds up across the clips we have run.
|
| 212 |
+
Pass `--audio` to do it here; the [hosted demo](https://huggingface.co/spaces/Viggle/viggle-animate)
|
| 213 |
+
does it automatically whenever the clip you upload carries sound.
|
| 214 |
+
|
| 215 |
+
**Turn the sound on for these two.** The driving clip, then the same shot with a different actor in
|
| 216 |
+
it, speaking her lines:
|
| 217 |
+
|
| 218 |
+
<video controls playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/audio-pin-driving.mp4"></video>
|
| 219 |
+
|
| 220 |
+
<video controls playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/audio-pin-out.mp4"></video>
|
| 221 |
+
|
| 222 |
+
All three files ship with the weights — the driving clip, the repainted still it is animated from
|
| 223 |
+
([`audio-pin-ref.png`](examples/media/audio-pin-ref.png)), and the take above — so the second clip
|
| 224 |
+
is a command rather than a claim. The same file is passed twice, once for the motion and once for
|
| 225 |
+
the soundtrack:
|
| 226 |
+
|
| 227 |
+
```bash
|
| 228 |
+
python inference/sample.py --model-dir ./MiniMax-H3 \
|
| 229 |
+
--cond examples/media/audio-pin-driving.mp4 \
|
| 230 |
+
--ref examples/media/audio-pin-ref.png \
|
| 231 |
+
--audio examples/media/audio-pin-driving.mp4 \
|
| 232 |
+
--out audio-pin-out.mp4
|
| 233 |
+
```
|
| 234 |
+
|
| 235 |
+
Everything else is the shipped default: 124 frames at 832x480, four steps, `--seed 42`; 51 s of
|
| 236 |
+
render on a B200 after the weights load, and it comes back at 46 dB against the file above. The
|
| 237 |
+
track you get back is the driving clip's own audio put through the audio VAE and decoded again —
|
| 238 |
+
same level, peaks rounded off (0.95 to 0.76 here); that round trip is the evidence the rows really
|
| 239 |
+
were pinned. Drop `--audio` and the same command writes a file whose audio track is digital silence,
|
| 240 |
+
sample for sample, and a mouth that moves without saying anything.
|
| 241 |
+
|
| 242 |
+
**Complex scenes are harder than single subjects.** Several characters at once, close interaction
|
| 243 |
+
between them, and shots that cut are all cases where quality drops off — enough that we would not
|
| 244 |
+
call them solved.
|
| 245 |
+
|
| 246 |
+
**We are training a substantially better model right now**, aimed squarely at what is left. This
|
| 247 |
+
release is the version we can ship today, not the ceiling.
|
| 248 |
+
|
| 249 |
+
## What this repository contains
|
| 250 |
+
|
| 251 |
+
Two parts, both derived from [`MiniMaxAI/MiniMax-H3`](https://huggingface.co/MiniMaxAI/MiniMax-H3)'s
|
| 252 |
+
`ref2va` transformer:
|
| 253 |
+
|
| 254 |
+
| | |
|
| 255 |
+
|---|---|
|
| 256 |
+
| `transformer/` | 33.1 B, bf16, 14 shards. A **full finetune** of the base `transformer_ref` on a character-replacement objective |
|
| 257 |
+
| `lora/` | rank 128 over 302 linear layers, 2.5 GB. A **DMD2-distilled** delta on that finetune — this is what collapses the sampler to three forward passes |
|
| 258 |
+
|
| 259 |
+
The LoRA is a delta on the *finetuned* transformer — loading it onto stock `transformer_ref` produces
|
| 260 |
+
garbage.
|
| 261 |
+
|
| 262 |
+
## Quickstart
|
| 263 |
+
|
| 264 |
+
This repository ships only the transformer and the LoRA — the VAE, audio VAE and schedulers load
|
| 265 |
+
from your own copy of the base model. Inference touches 11 GB of its 269 GB:
|
| 266 |
+
|
| 267 |
+
```bash
|
| 268 |
+
hf download MiniMaxAI/MiniMax-H3 --local-dir ./MiniMax-H3 \
|
| 269 |
+
--include "modular_model_index.json" "vae/*" "audio_vae/*" \
|
| 270 |
+
"scheduler/*" "audio_scheduler/*" "assets/ref2va.mp4"
|
| 271 |
+
|
| 272 |
+
hf download Viggle/Viggle-Animate --local-dir ./Viggle-Animate
|
| 273 |
+
|
| 274 |
+
pip install torch "git+https://github.com/huggingface/diffusers@d6726f3" av
|
| 275 |
+
|
| 276 |
+
python Viggle-Animate/inference/sample.py \
|
| 277 |
+
--model-dir ./MiniMax-H3 \
|
| 278 |
+
--cond driving.mp4 --ref ref.png --out swapped.mp4
|
| 279 |
+
```
|
| 280 |
+
|
| 281 |
+
`d6726f3` is the tested `diffusers` commit; the upstream `minimax_h3` modular pipeline is enough,
|
| 282 |
+
no fork or patch. The last `--include` is the clip [`examples/demo.sh`](examples/demo.sh) needs.
|
| 283 |
+
|
| 284 |
+
**One 80 GB card is not enough at bf16** — the transformer is 62 GiB resident and a 480×832 /
|
| 285 |
+
124-frame render peaks at 80.1 GiB allocated. Use a card with ≥ 96 GB, or pass `--offload` to
|
| 286 |
+
stream blocks from CPU (~12 GB resident, much slower).
|
| 287 |
+
|
| 288 |
+
**It also runs quantized on consumer hardware.** We deploy it on a single **RTX 5090 (32 GB)**:
|
| 289 |
+
NVFP4 weights — 4.5 bits/param, dispatching to the real sm_120 cutlass block-scaled kernel, 2.70×
|
| 290 |
+
bf16 per compiled linear — plus a low-rank `adaln_proj` and `torch.compile`. Quantization alone is
|
| 291 |
+
not enough for 32 GB: 13.0 B of the 33.1 B parameters sit in `adaln_proj`, which the linear-layer
|
| 292 |
+
quantizer does not touch. That deployment path is not shipped in this repository.
|
| 293 |
+
|
| 294 |
+
Defaults are the evaluated configuration: `--steps 4 --flow-shift 3 --num-frames 124` (≈ 5.2 s at
|
| 295 |
+
24 fps) `--seed 42`. Output geometry follows the driving clip and must be a multiple of 32 on both
|
| 296 |
+
axes. Weights load in ~21 s, once per process. **Four steps is the operating point, not a shortcut**
|
| 297 |
+
— the distilled model already renders sharper than its teacher, and raising the step count tips that
|
| 298 |
+
into over-sharpening. `--steps 4` is four sigma boundaries and therefore three forward passes.
|
| 299 |
+
`sample.py` drops the driving clip's audio at input and the model emits its own track, which the
|
| 300 |
+
fixed prompt asks to be silence. `--audio <file>` overrides that and pins a real soundtrack instead,
|
| 301 |
+
which is what makes the mouth track speech — see [Limitations](#limitations).
|
| 302 |
+
|
| 303 |
+
Pull a frame with `ffmpeg -ss 1.5 -i driving.mp4 -frames:v 1 ref.png` and edit it at the same
|
| 304 |
+
resolution. **It does not have to be the first frame.** The still is passed to the model with no
|
| 305 |
+
frame index, so nothing downstream knows where in the clip it came from — pick whichever frame shows
|
| 306 |
+
the character most clearly, front-on and unoccluded. Name the change, and pin down what must *not*
|
| 307 |
+
change: pose, hands, props, framing, background, light. An editor that quietly reframes the shot will
|
| 308 |
+
fight the driving motion. Prefer clips that keep one side to camera, and bind any new limb to a real
|
| 309 |
+
one.
|
| 310 |
+
|
| 311 |
+
[`examples/demo.sh`](examples/demo.sh) runs a swap end to end on a clip that ships with the base
|
| 312 |
+
model, so it needs no media from you — and it is the fastest way to check that the LoRA loaded.
|
| 313 |
+
|
| 314 |
+
## Citation
|
| 315 |
+
|
| 316 |
+
```bibtex
|
| 317 |
+
@misc{viggle2026animate,
|
| 318 |
+
title = {Viggle-Animate: Character Replacement in Video from a Single Repainted Frame},
|
| 319 |
+
author = {Viggle Research},
|
| 320 |
+
year = {2026},
|
| 321 |
+
url = {https://huggingface.co/Viggle/Viggle-Animate}
|
| 322 |
+
}
|
| 323 |
+
```
|
| 324 |
+
|
| 325 |
+
## License
|
| 326 |
+
|
| 327 |
+
The weights are a Model Derivative of MiniMax H3, so the
|
| 328 |
+
[MiniMax H3 Community License](LICENSE) applies to them — read it before you redistribute them or
|
| 329 |
+
ship a product on them. Our changes are listed in [`MODIFICATIONS.md`](MODIFICATIONS.md).
|
| 330 |
+
|
| 331 |
+
The code in [`inference/`](inference) and [`examples/`](examples) is Apache 2.0
|
| 332 |
+
([`LICENSE-CODE`](LICENSE-CODE)).
|
| 333 |
+
|
| 334 |
+
Music in the teaser at the top of this page: "Electrodoodle" by Kevin MacLeod
|
| 335 |
+
([incompetech.com](https://incompetech.com)), licensed under
|
| 336 |
+
[Creative Commons: By Attribution 4.0](http://creativecommons.org/licenses/by/4.0/).
|
| 337 |
+
|
| 338 |
+
## Intended use
|
| 339 |
+
|
| 340 |
+
This model exists to put a consenting performer into footage they did not shoot, and it will just
|
| 341 |
+
as readily put someone into footage they never agreed to appear in. Note where that decision is
|
| 342 |
+
made: **the identity comes from the frame you paint**, so an image editor's safeguards are upstream
|
| 343 |
+
of this model and none of them are in it. It cannot verify identity or consent. Do not run it on
|
| 344 |
+
people who have not agreed to it, label what you generate as AI-generated, and see Section V.5 of
|
| 345 |
+
the Agreement if you offer this as a service.
|
| 346 |
+
|
| 347 |
+
Powered by MiniMax H3.
|
assets/fixed_embed_fwd_anyframe.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e0ae90929caf7b790f5d0de599e868cc6c179f3e969577627971c9d78936a058
|
| 3 |
+
size 3714413
|
assets/fixed_prompt.txt
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<Video 1>: <0.2 seconds><1.2 seconds><2.2 seconds><3.2 seconds><4.2 seconds><5.0 seconds><Picture 1>: subject_definitions:
|
| 2 |
+
<Video 1> is the source video for the editing task.
|
| 3 |
+
<Picture 1> is one frame of the target video.
|
| 4 |
+
|
| 5 |
+
summary:
|
| 6 |
+
[video editing + character replacement] The target video is an edited version of <Video 1> in which every performer is replaced by a different person. <Picture 1> is one frame of the target video: it already shows the replacement, together with the background, camera framing and lighting it happens in. Every other frame of the target video shows those same people in that same place, following the motion, timing and camera of <Video 1>.
|
| 7 |
+
|
| 8 |
+
retention_analysis:
|
| 9 |
+
<Video 1> (source video editing): fully_preserved - the camera framing, the background, the lighting, and the motion and timing of every performance are copied frame for frame.
|
| 10 |
+
<Picture 1> (appears in [Shot 1]): reference - one frame of the target video, pixel for pixel. The people in it, their faces, hair, skin tone, build and clothing, and the background, the framing and the lighting are all taken from <Picture 1> and held unchanged from the first frame to the last.
|
| 11 |
+
|
| 12 |
+
detailed_description:
|
| 13 |
+
[Shot 1] The people of <Picture 1> perform exactly the motion of the corresponding performers in <Video 1>, in the same framing, on the same background, under the same lighting. Nothing outside the people changes. The camera framing never changes through the end of the video.
|
| 14 |
+
|
| 15 |
+
overall_soundscape:
|
| 16 |
+
No music and no speech.
|
examples/demo.sh
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env bash
|
| 2 |
+
# Reproduce the demo on the model card, and in doing so check your install.
|
| 3 |
+
#
|
| 4 |
+
# This repository bundles no driving footage. It builds the driving clip from a demo
|
| 5 |
+
# video that ships with the base model, using the exact crop documented below, and
|
| 6 |
+
# pairs it with the repainted first frame in media/reference.png.
|
| 7 |
+
#
|
| 8 |
+
# The result should match media/output.mp4. On the same GPU model we get it back
|
| 9 |
+
# bit-identical; on different hardware bf16 kernel scheduling shifts things, and a mean
|
| 10 |
+
# absolute difference around 1.5/255 is normal. What matters is that it is the same
|
| 11 |
+
# elderly woman holding the same black lamb. If it comes back as the young man from the
|
| 12 |
+
# source video instead, the LoRA did not load. If it comes back as noise, the weights
|
| 13 |
+
# are wrong.
|
| 14 |
+
#
|
| 15 |
+
# ./examples/demo.sh /path/to/MiniMax-H3
|
| 16 |
+
#
|
| 17 |
+
# About a minute on a B200, most of it loading weights.
|
| 18 |
+
|
| 19 |
+
set -euo pipefail
|
| 20 |
+
|
| 21 |
+
MODEL_DIR="${1:?usage: demo.sh /path/to/MiniMax-H3}"
|
| 22 |
+
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
| 23 |
+
OUT="${OUT:-$HERE/demo_out}"
|
| 24 |
+
mkdir -p "$OUT"
|
| 25 |
+
|
| 26 |
+
SRC="$MODEL_DIR/assets/ref2va.mp4"
|
| 27 |
+
[ -f "$SRC" ] || { echo "missing $SRC -- see the download command on the model card"; exit 1; }
|
| 28 |
+
|
| 29 |
+
# The source is 1344x768 and exactly 124 frames, which is the sampler's window. Crop a
|
| 30 |
+
# 512x768 portrait window around the figure; it stays in frame for the whole push-in, so
|
| 31 |
+
# no scaling and no padding are needed. media/reference.png is this crop's first frame,
|
| 32 |
+
# repainted.
|
| 33 |
+
ffmpeg -y -loglevel error -i "$SRC" \
|
| 34 |
+
-vf "crop=512:768:389:0" -frames:v 124 -an "$OUT/driving.mp4"
|
| 35 |
+
|
| 36 |
+
python "$HERE/../inference/sample.py" \
|
| 37 |
+
--model-dir "$MODEL_DIR" \
|
| 38 |
+
--cond "$OUT/driving.mp4" \
|
| 39 |
+
--ref "$HERE/media/reference.png" \
|
| 40 |
+
--out "$OUT/output.mp4"
|
| 41 |
+
|
| 42 |
+
echo
|
| 43 |
+
echo "wrote $OUT/output.mp4 -- compare against $HERE/media/output.mp4"
|
examples/media/audio-pin-driving.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fc3dd85a5be2e6246fb40afe46e337cb75d6c3558105fd81fa02f1ff286a5e92
|
| 3 |
+
size 4223283
|
examples/media/audio-pin-out.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:25fc5938c25a4dc48d93f388833358ed50cbb8876753ba75ae4e89045c63cf8e
|
| 3 |
+
size 389192
|
examples/media/audio-pin-ref.png
ADDED
|
Git LFS Details
|
examples/media/before-after.png
ADDED
|
Git LFS Details
|
examples/media/community-tweet.png
ADDED
|
Git LFS Details
|
examples/media/compare-cosplay.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6bd63ca750de511cbc6391664def22a9671f79d6885729c8d20bc92978fc8843
|
| 3 |
+
size 1099913
|
examples/media/compare-costume.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2a45ac3e161df80bff9e4b220e3282855821d450b1d14974b4e886609e001d8e
|
| 3 |
+
size 447259
|
examples/media/compare-fastmotion.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b79b19f959b370431dc5e090a80068c9046b07504064977d28d47d81e3ef50b7
|
| 3 |
+
size 1183899
|
examples/media/compare-gamechar.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:69e34a0ec75f588dfa3c2cdf882fab5bb7107e357e64ceb0d1881d8dbc23452c
|
| 3 |
+
size 858158
|
examples/media/compare-highkick.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6a137d9d19701eb1c75aa9e6b67776cfbf069619d9c0668885a717098f42fe37
|
| 3 |
+
size 847770
|
examples/media/compare-prop.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4128c0e35bf144b5d0e72c5837cc1958421a749337af395977de04fa07b92fea
|
| 3 |
+
size 678070
|
examples/media/compare-style.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:692ff38363a0162b965d0c6944e9adb98455c8a4e92419b3f9bb764ce9bd6778
|
| 3 |
+
size 536776
|
examples/media/hero-corgi.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4cb6c4bb9252352e257b9c8b153b5825d7b3a51227c263631169fa609663c3d5
|
| 3 |
+
size 425955
|
examples/media/hero-duo.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cbf81086b77ebbce703d33b9ced930d4f3cabd679c77fc764ad713a5e2bc2d27
|
| 3 |
+
size 187010
|
examples/media/output.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7b89a33a902cd92a1068307b701205029682bb5e5db0326775e3490767f1df04
|
| 3 |
+
size 880425
|
examples/media/pipeline.png
ADDED
|
Git LFS Details
|
examples/media/pipeline.svg
ADDED
|
|
examples/media/reference.png
ADDED
|
Git LFS Details
|
examples/media/swap-airliner.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5a7a31fd225e258633e208f13c18ff8523517834a9ea2c67b97ed060972becf3
|
| 3 |
+
size 397277
|
examples/media/swap-anime-duo.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:205700e55841a5fd22f517c6eb08c447068d0417a8e902bc91c863dede743d86
|
| 3 |
+
size 108986
|
examples/media/swap-claymation.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dd08b83d8a57e77408203710cf55eccdf8d1ee136c22123cc42ed8ac14434528
|
| 3 |
+
size 286267
|
examples/media/swap-corgi.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5954aeab0360ac8c1c5c62b5d35935520ac0feffdb8d11b8c55a6ff84f9d9239
|
| 3 |
+
size 272402
|
examples/media/swap-panda.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bec174f093bf553b39d1f022b6fedd46b9616a6ac30fa10a939753cf5e4a5150
|
| 3 |
+
size 390485
|
examples/media/swap-penguin.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ea2ccf8ce410331d194403903e2d1e836079b2d8fbafebf113265e52aafb3cd0
|
| 3 |
+
size 346127
|
examples/media/swap-photoreal-duo.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a2d7c7fb36c5b2779176484b0a2fcb1735578ea3cad2fa0a2c0b057218881554
|
| 3 |
+
size 108311
|
examples/media/swap-robot.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e8147fb8f85eca14e372ecd26f6d7b4556c649f5c980e0dd514a582ac8619077
|
| 3 |
+
size 194541
|
examples/media/swap-wushu-animals.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1ed178d0c88053df75803d904bef26c83c81da78c33d575ec8f5897701b622a7
|
| 3 |
+
size 204001
|
examples/media/swap-wushu-robots.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:63cc5dfe297ca472b7c54a9ef32aaa575f84e6cc1088a42d01236e40888785b2
|
| 3 |
+
size 201465
|
examples/media/teaser.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cbe5dd05057bafa9f97750788523d600c8dc1963be181fe56431393b9558ecb0
|
| 3 |
+
size 14442381
|
inference/sample.py
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# SPDX-License-Identifier: Apache-2.0
|
| 2 |
+
"""Viggle-Animate: replace the performers in a video with the person in a still.
|
| 3 |
+
|
| 4 |
+
python inference/sample.py --cond driving.mp4 --ref character.png --out swapped.mp4
|
| 5 |
+
|
| 6 |
+
`--cond` supplies the motion, camera framing, background and lighting; `--ref` supplies
|
| 7 |
+
who is in it. Everything about the render except the people is copied from `--cond`.
|
| 8 |
+
|
| 9 |
+
By default the model generates its own soundtrack, and the fixed prompt asks for silence --
|
| 10 |
+
so there is nothing for the mouth to sync to. `--audio` pins a real one instead:
|
| 11 |
+
|
| 12 |
+
python inference/sample.py --cond driving.mp4 --ref character.png \
|
| 13 |
+
--audio driving.mp4 --out swapped.mp4
|
| 14 |
+
|
| 15 |
+
The soundtrack is encoded once and held in the target audio rows as a *clean* latent for the
|
| 16 |
+
whole denoise, so the model conditions on it rather than predicting it, and the mouth tracks
|
| 17 |
+
that speech. The track written to `--out` is then that same audio, back through the audio VAE.
|
| 18 |
+
|
| 19 |
+
The text encoder is never loaded. Conditioning comes from `assets/fixed_embed_fwd_anyframe.pt`,
|
| 20 |
+
a frozen 362 x 5120 tensor computed once from the fixed prompt in `assets/fixed_prompt.txt`,
|
| 21 |
+
so Qwen3-VL (63 GB of the base repo) stays on disk and the text block of the packed sequence
|
| 22 |
+
is 362 rows instead of several thousand. There is no per-clip prompt and no caption: nothing
|
| 23 |
+
in the output comes from text you write.
|
| 24 |
+
|
| 25 |
+
Needs `--model-dir` pointing at a local copy of MiniMaxAI/MiniMax-H3 for the VAE, the audio
|
| 26 |
+
VAE and the schedulers. This repository ships only the transformer and the LoRA.
|
| 27 |
+
"""
|
| 28 |
+
|
| 29 |
+
import argparse
|
| 30 |
+
import os
|
| 31 |
+
import time
|
| 32 |
+
|
| 33 |
+
import torch
|
| 34 |
+
from diffusers import MiniMaxH3Transformer3DModel, ModularPipeline
|
| 35 |
+
from diffusers.modular_pipelines.minimax_h3 import (MiniMaxH3AudioReference, MiniMaxH3ImageReference,
|
| 36 |
+
MiniMaxH3VideoReference)
|
| 37 |
+
from diffusers.modular_pipelines.minimax_h3.before_encoder import MiniMaxH3Ref2VASetupStep
|
| 38 |
+
from diffusers.modular_pipelines.minimax_h3.encoders import MiniMaxH3Ref2VATextEncoderStep
|
| 39 |
+
from diffusers.modular_pipelines.minimax_h3.modular_pipeline import (align_num_frames,
|
| 40 |
+
audio_latent_num_frames)
|
| 41 |
+
from diffusers.utils.export_utils import encode_video
|
| 42 |
+
|
| 43 |
+
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
| 44 |
+
|
| 45 |
+
parser = argparse.ArgumentParser()
|
| 46 |
+
parser.add_argument("--cond", required=True, help="the video whose motion, framing and background are kept")
|
| 47 |
+
parser.add_argument("--ref", required=True, help="a single still of the person to put in it")
|
| 48 |
+
parser.add_argument("--out", required=True)
|
| 49 |
+
parser.add_argument("--model-dir", required=True,
|
| 50 |
+
help="a local copy of MiniMaxAI/MiniMax-H3, for the VAE / audio VAE / schedulers")
|
| 51 |
+
parser.add_argument("--transformer", default=os.path.join(HERE, "transformer"))
|
| 52 |
+
parser.add_argument("--lora", default=os.path.join(HERE, "lora"))
|
| 53 |
+
parser.add_argument("--embed", default=os.path.join(HERE, "assets", "fixed_embed_fwd_anyframe.pt"))
|
| 54 |
+
parser.add_argument("--num-frames", type=int, default=124, help="at 24 fps; 124 frames is ~5.2 s")
|
| 55 |
+
parser.add_argument("--steps", type=int, default=4,
|
| 56 |
+
help="the distilled student's operating point. More is not monotonically better: "
|
| 57 |
+
"s4 is not a degraded s12")
|
| 58 |
+
parser.add_argument("--flow-shift", type=float, default=3.0,
|
| 59 |
+
help="the base model's released default is 12; the few-step student wants 3")
|
| 60 |
+
parser.add_argument("--height", type=int, default=None, help="defaults to the conditioning clip's own height")
|
| 61 |
+
parser.add_argument("--width", type=int, default=None, help="defaults to the conditioning clip's own width")
|
| 62 |
+
parser.add_argument("--short-edge", type=int, default=None,
|
| 63 |
+
help="the canvas both references are laid out on. Defaults to the conditioning clip's own "
|
| 64 |
+
"short edge, which is what this model was evaluated at")
|
| 65 |
+
parser.add_argument("--offload", action="store_true",
|
| 66 |
+
help="stream the transformer from CPU in groups of 5 blocks: ~12 GB resident instead of 62")
|
| 67 |
+
parser.add_argument("--audio", default=None,
|
| 68 |
+
help="pin the generated soundtrack to this file's audio -- usually the driving clip "
|
| 69 |
+
"itself -- so the mouth tracks real speech instead of the silence the fixed "
|
| 70 |
+
"prompt asks for. Any file PyAV can decode; a video's soundtrack is taken")
|
| 71 |
+
parser.add_argument("--seed", type=int, default=42)
|
| 72 |
+
args = parser.parse_args()
|
| 73 |
+
|
| 74 |
+
fixed = torch.load(args.embed, weights_only=False)
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
def use_fixed_embeds(self, components, state):
|
| 78 |
+
block_state = self.get_block_state(state)
|
| 79 |
+
block_state.prompt_embeds = fixed["prompt_embeds"].to(components._execution_device, torch.bfloat16)
|
| 80 |
+
block_state.text_token_tags = fixed["text_token_tags"]
|
| 81 |
+
self.set_block_state(state, block_state)
|
| 82 |
+
return components, state
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
MiniMaxH3Ref2VATextEncoderStep.__call__ = use_fixed_embeds
|
| 86 |
+
|
| 87 |
+
# `--audio` holds the *target* audio rows at a real soundtrack, clean, for the whole denoise, rather
|
| 88 |
+
# than letting the model generate them. Two of the three things that takes have no argument on the
|
| 89 |
+
# pipeline, so they are patched here; the third is the `audio_latents=` passed to the call below.
|
| 90 |
+
if args.audio:
|
| 91 |
+
from diffusers.modular_pipelines.minimax_h3.before_denoise import MiniMaxH3SetTimestepsStep
|
| 92 |
+
from diffusers.modular_pipelines.minimax_h3.denoise import MiniMaxH3LoopSchedulerStep
|
| 93 |
+
|
| 94 |
+
# (1) Those rows carry finished audio, so they have to be told they are clean. H3's flow
|
| 95 |
+
# convention is reversed -- t = 1 is clean, not 0 -- and the library itself passes a literal 1.0
|
| 96 |
+
# for a *reference* soundtrack. `audio_timestep` is positional argument 6.
|
| 97 |
+
_build_row_timesteps = MiniMaxH3SetTimestepsStep.build_row_timesteps
|
| 98 |
+
MiniMaxH3SetTimestepsStep.build_row_timesteps = staticmethod(
|
| 99 |
+
lambda *a: _build_row_timesteps(*a[:6], 1.0, *a[7:]))
|
| 100 |
+
|
| 101 |
+
# (2) ...and the scheduler must never write them, or the first step would walk them off the
|
| 102 |
+
# soundtrack. Only the video rows are stepped. `num_condition_audio_rows` deliberately stays 0:
|
| 103 |
+
# raising it empties the decoder's `audio_latents[num_condition_audio_rows:]` slice and trips the
|
| 104 |
+
# reference-count check, and these rows are a pinned target, not a reference.
|
| 105 |
+
@torch.no_grad()
|
| 106 |
+
def video_only_step(self, components, block_state, i, t):
|
| 107 |
+
n = block_state.num_condition_video_rows
|
| 108 |
+
block_state.latents[n:] = components.scheduler.step(
|
| 109 |
+
block_state.noise_pred[0, n:].float(), t, block_state.latents[n:], return_dict=False)[0]
|
| 110 |
+
return components, block_state
|
| 111 |
+
|
| 112 |
+
MiniMaxH3LoopSchedulerStep.__call__ = video_only_step
|
| 113 |
+
|
| 114 |
+
# The reference order is frozen: the presentation names `<Video 1>` then `<Picture 1>`, and that order
|
| 115 |
+
# advances the shared rotary clock, so it is part of the layout rather than a detail of the prompt. The
|
| 116 |
+
# driving clip's own soundtrack is dropped here, as it is in training: `--audio` puts one back, but as
|
| 117 |
+
# the *target* to be matched rather than as a reference to be imitated.
|
| 118 |
+
video = MiniMaxH3VideoReference.from_file(args.cond)
|
| 119 |
+
video.audio, video.sample_rate = None, None
|
| 120 |
+
|
| 121 |
+
# Passing an orientation that disagrees with the clip silently generates a transposed video, so the output
|
| 122 |
+
# geometry is derived from the clip rather than typed.
|
| 123 |
+
height = args.height or video.frames.shape[1]
|
| 124 |
+
width = args.width or video.frames.shape[2]
|
| 125 |
+
short_edge = args.short_edge or min(height, width)
|
| 126 |
+
|
| 127 |
+
pipe = ModularPipeline.from_pretrained(args.model_dir, workflow="ref2va")
|
| 128 |
+
# Both references are pinned to the target's own short edge. The base model's released defaults (768 for the
|
| 129 |
+
# video reference, 2048 for the image) put the references on a grid the target never shares; this model was
|
| 130 |
+
# finetuned and evaluated with them nested, and changing it changes the take.
|
| 131 |
+
pipe.register_to_config(canvas_short_edge=short_edge,
|
| 132 |
+
canvas_max_pixels=short_edge * max(height, width),
|
| 133 |
+
reference_image_short_edge=short_edge)
|
| 134 |
+
|
| 135 |
+
t0 = time.time()
|
| 136 |
+
# `transformer_ref` is deliberately absent: our finetune replaces it outright, so loading the base copy
|
| 137 |
+
# first would read 62 GB off disk only to drop it.
|
| 138 |
+
pipe.load_components(names=["vae", "audio_vae", "scheduler", "audio_scheduler"],
|
| 139 |
+
pretrained_model_name_or_path=args.model_dir, dtype=torch.bfloat16)
|
| 140 |
+
pipe.transformer_ref = MiniMaxH3Transformer3DModel.from_pretrained(args.transformer, torch_dtype=torch.bfloat16)
|
| 141 |
+
# `prefix=None` and the explicit `weight_name` are both required. The loader defaults to looking for a `.bin`
|
| 142 |
+
# (raises) and to filtering keys for a `transformer.` prefix, which these bare keys do not have -- that
|
| 143 |
+
# mismatch loads *nothing* and only warns, so the default would silently render the un-distilled model.
|
| 144 |
+
pipe.transformer_ref.load_lora_adapter(args.lora, weight_name="pytorch_lora_weights.safetensors", prefix=None)
|
| 145 |
+
pipe.scheduler.set_shift(args.flow_shift)
|
| 146 |
+
|
| 147 |
+
# The pipeline snaps `num_frames` up to the next `17 * n + 5` the video VAE can encode. Doing it here
|
| 148 |
+
# too means the pinned audio is cut to the length that is really rendered rather than the one asked
|
| 149 |
+
# for -- off by one grid step and the rows no longer line up with the video.
|
| 150 |
+
num_frames = align_num_frames(args.num_frames, pipe.vae_frames_per_chunk, pipe.vae_latents_per_chunk)
|
| 151 |
+
|
| 152 |
+
if args.offload:
|
| 153 |
+
pipe.transformer_ref.enable_group_offload(
|
| 154 |
+
onload_device=torch.device("cuda"), offload_type="block_level", num_blocks_per_group=5,
|
| 155 |
+
non_blocking=True, use_stream=True, record_stream=True)
|
| 156 |
+
pipe.vae.to("cuda")
|
| 157 |
+
pipe.audio_vae.to("cuda")
|
| 158 |
+
else:
|
| 159 |
+
pipe.to("cuda")
|
| 160 |
+
print(f"loaded in {time.time() - t0:.0f}s; canvas {height}x{width}, references on short edge {short_edge}")
|
| 161 |
+
|
| 162 |
+
# The soundtrack becomes target rows the same way the pipeline turns a *reference* soundtrack into
|
| 163 |
+
# reference rows: truncate at the source rate, resample once, take the posterior mean, normalize.
|
| 164 |
+
# Reusing its own helper is what keeps the two paths from drifting apart.
|
| 165 |
+
audio_latents = None
|
| 166 |
+
if args.audio:
|
| 167 |
+
# One video frame more than the render needs, so the encoder cannot come up short. A source
|
| 168 |
+
# shorter than the grid it renders on is padded, and that tail is real silence.
|
| 169 |
+
n_samp = round((num_frames + 1) / pipe.fps * pipe.audio_sampling_rate)
|
| 170 |
+
track = MiniMaxH3AudioReference.from_file(args.audio)
|
| 171 |
+
wav = MiniMaxH3Ref2VASetupStep._normalize_audio_condition(
|
| 172 |
+
track.audio, track.sample_rate or pipe.audio_sampling_rate, pipe.audio_sampling_rate,
|
| 173 |
+
max_duration=(num_frames + 1) / pipe.fps)
|
| 174 |
+
have = wav.shape[1]
|
| 175 |
+
wav = torch.nn.functional.pad(wav, (0, max(0, n_samp - have)))[:, :n_samp]
|
| 176 |
+
with torch.no_grad():
|
| 177 |
+
# `encode` casts to the encoder's own dtype, so a float32 waveform is fine against a bf16 VAE.
|
| 178 |
+
posterior = pipe.audio_vae.encode(wav[:, None].to(pipe.audio_vae.device), return_dict=False)[0]
|
| 179 |
+
mean = torch.tensor(pipe.audio_vae.config.latents_mean).view(1, 1, -1)
|
| 180 |
+
std = torch.tensor(pipe.audio_vae.config.latents_std).view(1, 1, -1)
|
| 181 |
+
n_lat = audio_latent_num_frames(num_frames, pipe.fps)
|
| 182 |
+
# Channel-major rows: the two stereo channels are two batch items of the mono audio VAE.
|
| 183 |
+
rows = (posterior.mode().float().cpu().transpose(1, 2)[:, :n_lat] - mean) / std
|
| 184 |
+
if rows.shape[1] != n_lat:
|
| 185 |
+
raise RuntimeError(f"the soundtrack encoded to {rows.shape[1]} latents, short of the {n_lat} "
|
| 186 |
+
f"that {num_frames} frames need")
|
| 187 |
+
audio_latents = rows.permute(0, 2, 1).contiguous()
|
| 188 |
+
print(f"pinned {min(have, n_samp) / pipe.audio_sampling_rate:.2f}s of audio -> "
|
| 189 |
+
f"{tuple(audio_latents.shape)}, over {n_samp / pipe.audio_sampling_rate:.2f}s of video")
|
| 190 |
+
|
| 191 |
+
t0 = time.time()
|
| 192 |
+
result = pipe(
|
| 193 |
+
prompt=fixed["presentation"],
|
| 194 |
+
references=[video, MiniMaxH3ImageReference.from_file(args.ref)],
|
| 195 |
+
num_frames=num_frames,
|
| 196 |
+
height=height,
|
| 197 |
+
width=width,
|
| 198 |
+
num_inference_steps=args.steps,
|
| 199 |
+
audio_latents=audio_latents,
|
| 200 |
+
generator=torch.Generator().manual_seed(args.seed),
|
| 201 |
+
output=["videos", "audio", "sampling_rate"],
|
| 202 |
+
)
|
| 203 |
+
encode_video(result["videos"][0], fps=24, output_path=args.out,
|
| 204 |
+
audio=result["audio"][0], audio_sample_rate=result["sampling_rate"])
|
| 205 |
+
print(f"{time.time() - t0:.0f}s, peak {torch.cuda.max_memory_allocated() / 2**30:.1f} GiB -> {args.out}")
|
lora/pytorch_lora_weights.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bcdd60965d95e45a0b06f7d27435f21cbc2c8bd4f26f7b1cd8e66af7122c4af8
|
| 3 |
+
size 2666439896
|
requirements.txt
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Tested with these. Nothing here is pinned tightly except diffusers, which is pinned to a
|
| 2 |
+
# commit rather than a release: the `minimax_h3` modular pipeline is newer than any tag.
|
| 3 |
+
torch==2.9.1
|
| 4 |
+
git+https://github.com/huggingface/diffusers@d6726f3
|
| 5 |
+
transformers==4.57.3
|
| 6 |
+
safetensors==0.7.0
|
| 7 |
+
av==16.1.0
|
transformer/config.json
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_class_name": "MiniMaxH3Transformer3DModel",
|
| 3 |
+
"_diffusers_version": "0.36.0.dev0",
|
| 4 |
+
"num_attention_heads": 56,
|
| 5 |
+
"attention_head_dim": 128,
|
| 6 |
+
"hidden_size": 5376,
|
| 7 |
+
"num_layers": 50,
|
| 8 |
+
"num_refiner_layers": 2,
|
| 9 |
+
"ffn_dim": 14336,
|
| 10 |
+
"in_channels": 24,
|
| 11 |
+
"audio_in_channels": 32,
|
| 12 |
+
"patch_size": [
|
| 13 |
+
1,
|
| 14 |
+
2,
|
| 15 |
+
2
|
| 16 |
+
],
|
| 17 |
+
"text_dim": 5120,
|
| 18 |
+
"freq_dim": 256,
|
| 19 |
+
"time_embed_hidden_dim": 5376,
|
| 20 |
+
"time_embed_dim": 2688,
|
| 21 |
+
"rope_freq_dim": 16,
|
| 22 |
+
"rope_theta": 10000.0,
|
| 23 |
+
"norm_eps": 1e-05,
|
| 24 |
+
"qk_norm_eps": 1e-05,
|
| 25 |
+
"final_norm_eps": 1e-05
|
| 26 |
+
}
|
transformer/diffusion_pytorch_model-00001-of-00014.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fb30791125c5814baaefb5bac1e3de2cb9595e027866ffa3a1d1c946789ef427
|
| 3 |
+
size 4945656288
|
transformer/diffusion_pytorch_model-00002-of-00014.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:58d254c72eb9637b86b11be13f2d648e4dd7c37faddf0fd09f7808faef83e6a4
|
| 3 |
+
size 4490213792
|
transformer/diffusion_pytorch_model-00003-of-00014.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bd8410138703641d2f6475f5650aaaa6e2c335b12a992b88aab4cb6465666f3d
|
| 3 |
+
size 4701942704
|
transformer/diffusion_pytorch_model-00004-of-00014.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c42d981e9fa75379f14622855f045f2dee79f75b91ced7eeb0247bd4aab1cd95
|
| 3 |
+
size 4933368952
|
transformer/diffusion_pytorch_model-00005-of-00014.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2b6fb580aef372b651083570c7c25edfb9dce065b49a3f605499b31109d0f9d2
|
| 3 |
+
size 4567284240
|
transformer/diffusion_pytorch_model-00006-of-00014.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a3183d396d2fc2d695145c98988742849658561e7e2f69ce8c3a78d414a363a0
|
| 3 |
+
size 4701942704
|