developerjeremylive yycc commited on
Commit
b78f054
·
0 Parent(s):

Duplicate from Viggle/Viggle-Animate

Browse files

Co-authored-by: Yun Chen <yycc@users.noreply.huggingface.co>

This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +68 -0
  2. LICENSE +84 -0
  3. LICENSE-CODE +201 -0
  4. MODIFICATIONS.md +90 -0
  5. MiniMaxH3TokenRefinerBlock/package.pt2 +3 -0
  6. MiniMaxH3TransformerBlock/package.pt2 +3 -0
  7. NOTICE +13 -0
  8. README.md +347 -0
  9. assets/fixed_embed_fwd_anyframe.pt +3 -0
  10. assets/fixed_prompt.txt +16 -0
  11. examples/demo.sh +43 -0
  12. examples/media/audio-pin-driving.mp4 +3 -0
  13. examples/media/audio-pin-out.mp4 +3 -0
  14. examples/media/audio-pin-ref.png +3 -0
  15. examples/media/before-after.png +3 -0
  16. examples/media/community-tweet.png +3 -0
  17. examples/media/compare-cosplay.mp4 +3 -0
  18. examples/media/compare-costume.mp4 +3 -0
  19. examples/media/compare-fastmotion.mp4 +3 -0
  20. examples/media/compare-gamechar.mp4 +3 -0
  21. examples/media/compare-highkick.mp4 +3 -0
  22. examples/media/compare-prop.mp4 +3 -0
  23. examples/media/compare-style.mp4 +3 -0
  24. examples/media/hero-corgi.mp4 +3 -0
  25. examples/media/hero-duo.mp4 +3 -0
  26. examples/media/output.mp4 +3 -0
  27. examples/media/pipeline.png +3 -0
  28. examples/media/pipeline.svg +81 -0
  29. examples/media/reference.png +3 -0
  30. examples/media/swap-airliner.mp4 +3 -0
  31. examples/media/swap-anime-duo.mp4 +3 -0
  32. examples/media/swap-claymation.mp4 +3 -0
  33. examples/media/swap-corgi.mp4 +3 -0
  34. examples/media/swap-panda.mp4 +3 -0
  35. examples/media/swap-penguin.mp4 +3 -0
  36. examples/media/swap-photoreal-duo.mp4 +3 -0
  37. examples/media/swap-robot.mp4 +3 -0
  38. examples/media/swap-wushu-animals.mp4 +3 -0
  39. examples/media/swap-wushu-robots.mp4 +3 -0
  40. examples/media/teaser.mp4 +3 -0
  41. inference/sample.py +205 -0
  42. lora/pytorch_lora_weights.safetensors +3 -0
  43. requirements.txt +7 -0
  44. transformer/config.json +26 -0
  45. transformer/diffusion_pytorch_model-00001-of-00014.safetensors +3 -0
  46. transformer/diffusion_pytorch_model-00002-of-00014.safetensors +3 -0
  47. transformer/diffusion_pytorch_model-00003-of-00014.safetensors +3 -0
  48. transformer/diffusion_pytorch_model-00004-of-00014.safetensors +3 -0
  49. transformer/diffusion_pytorch_model-00005-of-00014.safetensors +3 -0
  50. transformer/diffusion_pytorch_model-00006-of-00014.safetensors +3 -0
.gitattributes ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ examples/media/before-after.png filter=lfs diff=lfs merge=lfs -text
37
+ examples/media/output.mp4 filter=lfs diff=lfs merge=lfs -text
38
+ examples/media/reference.png filter=lfs diff=lfs merge=lfs -text
39
+ examples/media/compare-monkey.mp4 filter=lfs diff=lfs merge=lfs -text
40
+ examples/media/compare-parka.mp4 filter=lfs diff=lfs merge=lfs -text
41
+ examples/media/compare-horns.mp4 filter=lfs diff=lfs merge=lfs -text
42
+ examples/media/compare-costume.mp4 filter=lfs diff=lfs merge=lfs -text
43
+ examples/media/compare-prop.mp4 filter=lfs diff=lfs merge=lfs -text
44
+ examples/media/compare-style.mp4 filter=lfs diff=lfs merge=lfs -text
45
+ examples/media/swap-airliner.mp4 filter=lfs diff=lfs merge=lfs -text
46
+ examples/media/swap-anime-duo.mp4 filter=lfs diff=lfs merge=lfs -text
47
+ examples/media/swap-claymation.mp4 filter=lfs diff=lfs merge=lfs -text
48
+ examples/media/swap-corgi.mp4 filter=lfs diff=lfs merge=lfs -text
49
+ examples/media/swap-panda.mp4 filter=lfs diff=lfs merge=lfs -text
50
+ examples/media/swap-penguin.mp4 filter=lfs diff=lfs merge=lfs -text
51
+ examples/media/swap-photoreal-duo.mp4 filter=lfs diff=lfs merge=lfs -text
52
+ examples/media/swap-robot.mp4 filter=lfs diff=lfs merge=lfs -text
53
+ examples/media/swap-wushu-animals.mp4 filter=lfs diff=lfs merge=lfs -text
54
+ examples/media/swap-wushu-robots.mp4 filter=lfs diff=lfs merge=lfs -text
55
+ examples/media/hero-corgi.mp4 filter=lfs diff=lfs merge=lfs -text
56
+ examples/media/hero-duo.mp4 filter=lfs diff=lfs merge=lfs -text
57
+ examples/media/compare-cosplay.mp4 filter=lfs diff=lfs merge=lfs -text
58
+ examples/media/compare-gamechar.mp4 filter=lfs diff=lfs merge=lfs -text
59
+ examples/media/teaser.mp4 filter=lfs diff=lfs merge=lfs -text
60
+ MiniMaxH3TransformerBlock/package.pt2 filter=lfs diff=lfs merge=lfs -text
61
+ MiniMaxH3TokenRefinerBlock/package.pt2 filter=lfs diff=lfs merge=lfs -text
62
+ examples/media/compare-fastmotion.mp4 filter=lfs diff=lfs merge=lfs -text
63
+ examples/media/compare-highkick.mp4 filter=lfs diff=lfs merge=lfs -text
64
+ examples/media/pipeline.png filter=lfs diff=lfs merge=lfs -text
65
+ examples/media/community-tweet.png filter=lfs diff=lfs merge=lfs -text
66
+ examples/media/audio-pin-driving.mp4 filter=lfs diff=lfs merge=lfs -text
67
+ examples/media/audio-pin-out.mp4 filter=lfs diff=lfs merge=lfs -text
68
+ examples/media/audio-pin-ref.png filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,84 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ MiniMax H3 COMMUNITY LICENSE AGREEMENT
2
+ MiniMax H3 release date/License date: August 2, 2026.
3
+ The scope of this License Agreement (this “Agreement”) is expressly limited to the “Applicable Territory” as defined below.
4
+ By clicking to accept, or by using, reproducing, modifying, distributing, running, or displaying any portion or element of the MiniMax H3 Works (including through any Hosted Services) in any manner, you acknowledge and accept the terms of this Agreement, and this Agreement shall take immediate effect upon the occurrence of such act.
5
+ I. Definitions
6
+ 1. “Acceptable Use Policy” means the policy published by MiniMax in Exhibit A.
7
+ 2. “Agreement” means the terms and conditions set forth herein that govern the use, reproduction, distribution, modification, running, and display of the MiniMax H3 Works or any portion or element thereof.
8
+ 3. “Applicable Territory” means worldwide, excluding the Excluded Territories.
9
+ 4. “Documentation” means the specifications, manuals, and documentation concerning MiniMax H3 that are publicly released by MiniMax.
10
+ 5. “Excluded Territories” means the European Union, the United Kingdom, the Republic of Korea and the United States of America.
11
+ 6. “MiniMax H3” means the video generation model, together with its software and algorithms, including trained model weights, parameters (including optimizer states), machine-learning model code, inference-supporting code, and other elements thereof made publicly available by Us, as released at https://huggingface.co/MiniMaxAI/MiniMax-H3.
12
+ 7. “MiniMax H3 Works” means (i) the Materials, (ii) the Model Derivatives, and (iii) all derivatives thereof.
13
+ 8. “Hosted Services” means hosted services provided via application programming interfaces (APIs), web access, or any other electronic or remote means.
14
+ 9. “Licensee,” “you,” or “your” means the natural or legal person exercising rights and/or using the MiniMax H3 Works for any purpose in any field of use under this Agreement.
15
+ 10. “Materials” means, collectively, MiniMax H3 and the Documentation (and any portion thereof), in each case as made available by MiniMax under this Agreement and proprietary to MiniMax.
16
+ 11. “Model Derivatives” means all of the following: (i) any modification of MiniMax H3 or any Model Derivative thereof; (ii) any work based on MiniMax H3 or any Model Derivative thereof; or (iii) any other machine learning model created by transferring the patterns of the weights, parameters, operational patterns, or Outputs of MiniMax H3 or any Model Derivative thereof to another model, such that the latter model exhibits behavior similar to MiniMax H3 or its Model Derivatives, including by distillation methods, methods using intermediate data representations, or methods based on training using synthetic-data Outputs generated by MiniMax H3 or its Model Derivatives. For the avoidance of doubt, Outputs are not deemed Model Derivatives.
17
+ 12. “Output” means any result of operating or otherwise using MiniMax H3 or any Model Derivatives (including through Hosted Services).
18
+ 13. “Third Party” means any natural or legal person that is not under common control with us or with you.
19
+ 14. “Including” means “including but not limited to.”
20
+ 15. “We,” “Us” or “MiniMax” means Nanonoble Pte. Ltd..
21
+ II. Grant of Rights
22
+ Solely within the Applicable Territory, we grant you a non-exclusive, non-transferable, royalty-free, limited license to use, reproduce, distribute, create derivative works (including Model Derivatives), and modify the Materials in accordance with the terms of this Agreement and the Acceptable Use Policy, based on the intellectual property and other rights owned by MiniMax that are embodied in or used by the Materials. You shall not violate (or encourage or permit any person to violate) any term of this Agreement or the Acceptable Use Policy.
23
+ We will continuously evaluate the applicable laws, regulations and compliance requirements for the Excluded Territories. In the meantime, should any person in such Excluded Territories be interested in deploying our models, you are welcome to contact us about obtaining a license, which will be granted based on robust controls and guardrails for purposes of complying with the laws, regulations and compliance requirements of the Excluded Territories.
24
+ III. Distribution and Redistribution
25
+ Subject to and conditioned on your continuing compliance with this Agreement, including its territorial restrictions and the Acceptable Use Policy, and solely within the Applicable Territory, you may distribute or make available the MiniMax H3 Works to Third Parties within the Applicable Territory; provided, that all of the following conditions are met:
26
+ 1. You must provide a copy of this Agreement to all such Third Parties who receive the MiniMax H3 Works or use your products or services related thereto;
27
+ 2. You must cause any modified files to carry prominent notices stating that you have modified such files;
28
+ 3. You are encouraged to:
29
+ a. display a notice on any product or service developed using MiniMax H3 indicating that the product or service is “Powered by MiniMax H3”;
30
+ b. add an AI-generation identifier to files produced using generative AI models including MiniMax H3; and
31
+ c. publish at least one technical blog post or a public statement describing your experience using MiniMax H3 Works;
32
+ 4. All distributions to Third Parties (other than through Hosted Services) must be accompanied by a “NOTICE” text file containing the following notice:
33
+ “MiniMax H3 is licensed under the MiniMax H3 Community License Agreement, Copyright © 2026 MiniMax. All Rights Reserved.”
34
+ You may add your own copyright notices on your modifications; except as provided in this Section and in Section V, however, you may not impose additional or different terms and conditions on the use, reproduction, or distribution of your modifications or of any aggregate Model Derivatives, and your use, reproduction, modification, distribution, running, and display of the work must otherwise comply with the terms and conditions of this Agreement (including the provisions concerning the Applicable Territory). If you receive the MiniMax H3 Works from a Licensee as part of an integrated end-user product, the provisions of Section III of this Agreement do not apply to you, but Section V and Exhibit A remain applicable.
35
+ IV. Additional Commercial Terms
36
+ 1. You shall obtain a separate, prior written authorization from MiniMax by contacting api@minimax.io with the subject line “MiniMax H3 licensing - authorization request”, if your commercial products and services generate more than 20 million US dollars (or equivalent in other currencies) in yearly revenue.
37
+ 2. You shall prominently display “MiniMax H3”on the user interface of commercial product or service that uses MiniMax H3 or MiniMax H3 Works.
38
+ V. Use Restrictions
39
+ 1. Your use of the MiniMax H3 Works must comply with applicable laws and regulations (including trade-compliance laws and regulations) and must comply with the Acceptable Use Policy for the MiniMax H3 Works, which is incorporated into this Agreement by reference.
40
+ 2. Before providing access to the MiniMax H3 Works or any product, service, or Hosted Service incorporating them, you must bind each recipient or user to enforceable terms at least as protective as the use restrictions in this Section V and Exhibit A, and you must notify each recipient or user that those restrictions apply.
41
+ 3. You may not use the MiniMax H3 Works or any of their Outputs or results to improve any other artificial intelligence model (other than MiniMax H3 or its Model Derivatives).
42
+ 4. You may not use, reproduce, modify, distribute, or display the MiniMax H3 Works or any of their Outputs or results outside the Applicable Territory. Any such use outside the Applicable Territory is not authorized by this Agreement.
43
+ 5. If you provide or make available to any Third Party a product, service, or Hosted Service that permits the generation of Outputs using MiniMax H3 or any Model Derivative, you must, before making that product or service available and throughout its operation, implement, maintain, test, and periodically review reasonable and proportionate technical and organizational safeguards designed to prevent and mitigate access, uses, and Outputs that violate this Section V or Exhibit A, including uses or Outputs that infringe, misappropriate, or otherwise violate any Third Party’s intellectual-property or other rights. You must not knowingly disable, materially weaken, or permit the circumvention of those safeguards. You must maintain a reasonably accessible mechanism for reporting suspected violations. Upon receiving a good-faith report or otherwise obtaining actual knowledge of a violation, you must promptly investigate and take reasonable steps within your control to stop or mitigate the violation, including removing or disabling access to offending content or services and suspending or terminating repeat violators where appropriate. You are responsible for implementing and enforcing these requirements with respect to your products, services, systems, users, and downstream recipients.
44
+ VI. Intellectual Property
45
+ 1. Subject to MiniMax’s rights in the MiniMax H3 Works (and the intellectual property therein), and to your compliance with the terms and conditions of this Agreement, as between you and MiniMax, you will own the derivative works and modifications of the Materials that you have created or had created, as well as any Model Derivatives.
46
+ 2. Except for the limited license expressly granted in this paragraph, no trademark license is granted under this Agreement; with respect to MiniMax H3 Works, the Licensee may not use any name or mark owned by or associated with MiniMax or any of its affiliates, except as reasonably and customarily necessary to describe and distribute the MiniMax H3 Works. MiniMax hereby grants you a license to use the “MiniMax H3” mark (the “Mark”) within the Applicable Territory solely for the purpose of complying with Section III.3; provided, that you comply with all applicable trademark-protection laws. All goodwill arising from your use of the Mark shall inure to the benefit of MiniMax.
47
+ 3. If you bring or assert any suit or other legal proceeding (including a cross-claim or counterclaim in any action) against us or any other natural or legal person alleging that the Materials, any Output, or any portion of the foregoing infringes any intellectual property right or other right owned by you or for which you can obtain a license, all licenses granted to you under this Agreement will terminate as of the date such suit or proceeding is filed. You shall defend, indemnify, and hold us harmless against any Third-Party claim arising out of or related to the use or distribution of the MiniMax H3 Works by you or by any Third Party.
48
+ 4. MiniMax claims no rights over the Outputs you generate. You and your users are entirely responsible for the Outputs and any subsequent use thereof.
49
+ VII. Disclaimers and Limitations of Liability
50
+ 1. We have no obligation to support, update, provide training for, or develop any further version of the MiniMax H3 Works, or to grant any license with respect thereto.
51
+ 2. UNLESS AND ONLY TO THE EXTENT REQUIRED BY APPLICABLE LAW, THE MINIMAX H3 WORKS AND ANY OUTPUT AND RESULTS THEREFROM ARE PROVIDED “AS IS” WITHOUT ANY EXPRESS OR IMPLIED WARRANTIES OF ANY KIND INCLUDING ANY WARRANTIES OF TITLE, MERCHANTABILITY, NONINFRINGEMENT, COURSE OF DEALING, USAGE OF TRADE, OR FITNESS FOR A PARTICULAR PURPOSE. YOU ARE SOLELY RESPONSIBLE FOR DETERMINING THE APPROPRIATENESS OF USING, REPRODUCING, MODIFYING, PERFORMING, DISPLAYING OR DISTRIBUTING ANY OF THE MINIMAX H3 WORKS OR OUTPUTS AND ASSUME ANY AND ALL RISKS ASSOCIATED WITH YOUR OR A THIRD PARTY’S USE OR DISTRIBUTION OF ANY OF THE MINIMAX H3 WORKS OR OUTPUTS AND YOUR EXERCISE OF RIGHTS AND PERMISSIONS UNDER THIS AGREEMENT.
52
+ 3. TO THE FULLEST EXTENT PERMITTED BY APPLICABLE LAW, IN NO EVENT SHALL MINIMAX OR ITS AFFILIATES BE LIABLE UNDER ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, TORT, NEGLIGENCE, PRODUCTS LIABILITY, OR OTHERWISE, FOR ANY DAMAGES, INCLUDING ANY DIRECT, INDIRECT, SPECIAL, INCIDENTAL, EXEMPLARY, CONSEQUENTIAL OR PUNITIVE DAMAGES, OR LOST PROFITS OF ANY KIND ARISING FROM THIS AGREEMENT OR RELATED TO ANY OF THE MINIMAX H3 WORKS OR OUTPUTS, EVEN IF MINIMAX OR ITS AFFILIATES HAVE BEEN ADVISED OF THE POSSIBILITY OF ANY OF THE FOREGOING.
53
+ VIII. Term and Termination
54
+ 1. This Agreement is effective from the moment you accept this Agreement or begin accessing the Materials, and, subject to your compliance with its terms and conditions, will remain in effect until terminated as provided herein.
55
+ 2. If you breach any term or condition of this Agreement, we have the right to terminate this Agreement. Upon termination, you must immediately cease accessing, using, and distributing the MiniMax H3 Works; delete or destroy all copies within your possession or control; and notify each downstream recipient that your authorization has ended. The obligations in the preceding sentence and Sections VI.1, VI.3, VII, and IX survive termination.
56
+ IX. Governing Law and Jurisdiction
57
+ 1. This Agreement, and any dispute arising out of or related to this Agreement, shall be governed by the laws of the Hong Kong Special Administrative Region of the People’s Republic of China, without regard to its conflict-of-laws rules. The United Nations Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
58
+ 2. Any dispute arising out of or related to this Agreement shall be subject to the exclusive jurisdiction of the courts of the Hong Kong Special Administrative Region of the People’s Republic of China with competent jurisdiction. Both MiniMax and the Licensee hereby consent to the exclusive jurisdiction of such courts for any such dispute.
59
+ Additional Note: Please note that the encoder of MiniMax H3 uses Qwen3-VL-32B, which is licensed under Apache 2.0 License: https://github.com/QwenLM/Qwen3-VL/blob/main/LICENSE.
60
+
61
+ Exhibit A — Acceptable Use Policy
62
+ MiniMax reserves the right to update this Acceptable Use Policy from time to time.
63
+ Last revised: August 2, 2026.
64
+ MiniMax is committed to promoting the safe and fair use of its tools and features, including MiniMax H3. You agree not to use MiniMax H3, any Model Derivatives, or any Output in any of the following ways:
65
+ 1. Use outside the Applicable Territory;
66
+ 2. Use in any manner that violates any applicable national, federal, state, local, or international law, regulation, or other legal requirement, or that infringes, misappropriates, or otherwise violates any Third Party’s intellectual-property or other proprietary rights, including through unauthorized reproduction, distribution, public display, public performance, or creation of derivative works;
67
+ 3. Use in any manner that may harm yourself or others;
68
+ 4. Use to repurpose or distribute the Outputs of MiniMax H3 or any Model Derivatives in order to harm yourself or others;
69
+ 5. Use to circumvent or bypass any safety guardrails or safeguards we have implemented;
70
+ 6. Use in any manner that exploits or harms, or intends to exploit or harm, minors;
71
+ 7. Use to generate or disseminate verifiably false information and/or content for the purpose of harming others or influencing elections;
72
+ 8. Use to manufacture or facilitate false online engagement, including fake reviews and other means of false online engagement;
73
+ 9. Use to intentionally defame, disparage, or otherwise harass others;
74
+ 10. Use to generate and/or disseminate malware (including ransomware) or any other content intended to damage electronic systems;
75
+ 11. Use to generate or disseminate personally identifiable information for the purpose of harming others;
76
+ 12. Use to generate or disseminate information (including images, code, posts, or articles) in or to any public environment (including via bot tweets or similar means) without clearly and prominently disclosing that such information and/or content is machine-generated;
77
+ 13. Use to impersonate another person without that person’s consent, authorization, or lawful right to do so;
78
+ 14. Use to make high-risk automated decisions in critical domains that affect individual safety, rights, or well-being (such as law enforcement, immigration, healthcare or medical services, critical-infrastructure management, product-safety components, essential services, credit, employment, housing, education, social scoring, or insurance);
79
+ 15. Use in any manner that violates or disregards the social, ethical, or moral standards of other countries or regions;
80
+ 16. Use to carry out, assist, threaten, incite, plan, advocate for, or encourage violent extremism or terrorism;
81
+ 17. Use for any purpose intended to discriminate against, or harm, individuals or groups based on protected characteristics or categories, online or offline social behavior, or known or predicted personality traits;
82
+ 18. Use to intentionally exploit the vulnerabilities of specific populations based on age, social, physical, or psychological characteristics, so as to materially distort the behavior of a member of that group in a manner that causes, or is likely to cause, physical or psychological harm to that person or to others;
83
+ 19. Use for military purposes;
84
+ 20. Use to engage in any unauthorized or unlicensed professional activity, including but not limited to financial, legal, medical or healthcare, or other professional practice.
LICENSE-CODE ADDED
@@ -0,0 +1,201 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright [yyyy] [name of copyright owner]
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
MODIFICATIONS.md ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Modified files
2
+
3
+ Section III.2 of the MiniMax H3 Community License Agreement requires that modified
4
+ files carry a prominent notice saying so. This file is that notice.
5
+
6
+ Everything below is derived from
7
+ [`MiniMaxAI/MiniMax-H3`](https://huggingface.co/MiniMaxAI/MiniMax-H3).
8
+
9
+ ## `transformer/` — modified
10
+
11
+ **Every weight file in `transformer/` has been modified.** It started as the base
12
+ model's `transformer_ref/` (the `ref2va` transformer, 33.1 B parameters) and every
13
+ parameter was updated by a full finetune on a character-replacement objective. The
14
+ architecture, `config.json` and tensor names are unchanged, so it is a drop-in
15
+ replacement for the base `transformer_ref/`; the numbers in it are not the base
16
+ model's numbers.
17
+
18
+ The file layout also differs: the finetune was written as one 61.7 GiB safetensors
19
+ file and re-sharded here into 14 parts, because HuggingFace rejects single files
20
+ above 50 GB. The 638 tensors and their contents are unchanged by that re-sharding.
21
+
22
+ ## `lora/pytorch_lora_weights.safetensors` — new
23
+
24
+ Not a MiniMax file. A rank-128 LoRA over 302 linear layers of `transformer/`,
25
+ trained by us with DMD2 distillation. It is a delta on the finetuned transformer
26
+ above, not on the base model — loading it onto stock `transformer_ref/` produces
27
+ garbage.
28
+
29
+ ## `assets/fixed_embed_fwd_anyframe.pt` — new
30
+
31
+ Not a MiniMax file. A frozen 362 × 5120 text-conditioning tensor we computed once
32
+ with the base model's own text encoder, so that inference never has to load
33
+ Qwen3-VL. It is an *output* of the base model's encoder in the sense of Section
34
+ I.12, computed from the prompt in `assets/fixed_prompt.txt`.
35
+
36
+ ## `assets/fixed_prompt.txt` — new
37
+
38
+ Not a MiniMax file. The prompt text the tensor above was computed from, included so
39
+ that what conditions every render is readable rather than opaque.
40
+
41
+ ## `inference/sample.py`, `examples/demo.sh` — new
42
+
43
+ Not MiniMax files. Written by us against the public `diffusers` API.
44
+
45
+ ## `LICENSE-CODE`, `NOTICE` — new
46
+
47
+ Not MiniMax files. `LICENSE-CODE` is the Apache 2.0 text, and it covers `inference/` and
48
+ `examples/` only — each file there carries an `SPDX-License-Identifier: Apache-2.0` header.
49
+ `NOTICE` records that the weights are *not* Apache 2.0. `LICENSE` is the MiniMax H3
50
+ Community License Agreement itself, included unmodified as Section III.1 requires.
51
+
52
+ ## `examples/media/` — new
53
+
54
+ Two kinds of media here, with different provenance.
55
+
56
+ **The demo** — `reference.png`, `output.mp4`, `before-after.png` — derives from
57
+ `assets/ref2va.mp4`, a video MiniMax published with the base model. That clip is itself a
58
+ MiniMax H3 generation rather than camera footage. All three files are a 512 × 768 portrait
59
+ crop of it (`crop=512:768:389:0`, no scaling):
60
+
61
+ - `reference.png` — the crop's first frame with the young man repainted as an invented
62
+ elderly woman. Produced with OpenAI's `gpt-image-2`; the character is fictional and is
63
+ not a real person or an existing property.
64
+ - `output.mp4` — that reference propagated across 124 frames by this model.
65
+ - `before-after.png` — frames from the driving crop above frames from the output.
66
+
67
+ The driving clip itself is **not** bundled. `examples/demo.sh` rebuilds it, with the
68
+ documented crop, from your own copy of the base model.
69
+
70
+ **The comparison clips** — `compare-prop.mp4`, `compare-costume.mp4`, `compare-style.mp4` — are a
71
+ different matter. Their driving videos are
72
+ real filmed footage that we hold the rights to, and they are among the clips this model was
73
+ evaluated against. Each file is a four-panel stack: painted reference, driving video, this model,
74
+ Wan2.2-Animate-14B. The Wan2.2-Animate panels were rendered by us from the official
75
+ [`Wan-AI/Wan2.2-Animate-14B`](https://huggingface.co/Wan-AI/Wan2.2-Animate-14B) weights and code,
76
+ unmodified, at the replacement-mode settings its own README documents (20 steps, `sample_shift 5.0`,
77
+ `--refert_num 1 --replace_flag --use_relighting_lora`, preprocessing at `--w_len 1 --h_len 1`) —
78
+ that panel is Wan's output, not ours.
79
+
80
+ Wan's `generate.py` hardcodes a 30 fps output timebase regardless of the source, so its raw files
81
+ claim 4.10 s for motion that is 24 fps. The panels are retimed (`setpts`), not resampled, so no
82
+ frames are dropped and both models play at the same speed. Every panel is letterboxed into the
83
+ driving clip's own geometry; nothing is stretched.
84
+
85
+ ## Not redistributed here
86
+
87
+ The VAE, audio VAE, schedulers, text encoder, tokenizer and processor are **not**
88
+ included in this repository and are not modified. They are loaded at runtime from
89
+ your own copy of `MiniMaxAI/MiniMax-H3`, which you must download separately and
90
+ under its own license terms.
MiniMaxH3TokenRefinerBlock/package.pt2 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1fc10961935eb10891d1e2449b18185533d5714a40bb6f6929b8d6b7c1b4d39f
3
+ size 604596
MiniMaxH3TransformerBlock/package.pt2 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:58c651aa49a7455aa6a49e17ae7c74b4fd73e9306456a1baaa3300d0aef1ee09
3
+ size 853072
NOTICE ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ MiniMax H3 is licensed under the MiniMax H3 Community License Agreement,
2
+ Copyright © 2026 MiniMax. All Rights Reserved.
3
+
4
+ ---
5
+
6
+ Viggle-Animate is a Model Derivative of MiniMax H3, as that term is defined in
7
+ Section I.11 of the MiniMax H3 Community License Agreement. It is distributed
8
+ under that same Agreement, a copy of which is included in this repository as
9
+ LICENSE. See MODIFICATIONS.md for the list of files that were modified.
10
+
11
+ Powered by MiniMax H3.
12
+
13
+ Modifications and additions Copyright © 2026 Viggle AI.
README.md ADDED
@@ -0,0 +1,347 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: minimax-h3-community-license
4
+ license_link: LICENSE
5
+ base_model: MiniMaxAI/MiniMax-H3
6
+ pipeline_tag: video-to-video
7
+ tags:
8
+ - video-editing
9
+ - character-replacement
10
+ - video-to-video
11
+ - distillation
12
+ - dmd
13
+ ---
14
+
15
+ # Viggle-Animate
16
+
17
+ ### Character Replacement in Video from a Single Repainted Frame
18
+
19
+ **[Try the demo](https://huggingface.co/spaces/Viggle/viggle-animate)** &nbsp;·&nbsp;
20
+ **[Research write-up](https://viggle.ai/research/viggle-animate-character-replacement-from-a-repainted-frame?utm_source=huggingface&utm_medium=social&utm_campaign=viggle-animate&utm_content=viggle/viggle-animate)** &nbsp;·&nbsp;
21
+ **[ComfyUI nodes](https://github.com/Saganaki22/ComfyUI-Viggle-Animate-H3)** &nbsp;·&nbsp;
22
+ **[viggle.ai/h3](https://viggle.ai/h3)** &nbsp;·&nbsp;
23
+ Built on **[MiniMaxAI/MiniMax-H3](https://huggingface.co/MiniMaxAI/MiniMax-H3)**
24
+
25
+ <a href="https://x.com/cocktailpeanut/status/2097332291844399514" style="display:block;max-width:460px;margin:0 0 .5em">
26
+ <img src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/community-tweet.png" width="100%" style="display:block;border-radius:14px" alt="@cocktailpeanut on X: two AI CEOs dropped into a film scene with Viggle-Animate - 1.5M views, 14K likes">
27
+ </a>
28
+
29
+ <sub>Made with Viggle-Animate by <a href="https://x.com/cocktailpeanut/status/2097332291844399514">@cocktailpeanut</a>, not by us. Click through to watch it on X.</sub>
30
+
31
+ <video autoplay controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/teaser.mp4"></video>
32
+
33
+ **Viggle-Animate replaces the character in a video with whatever you paint into one of its own
34
+ frames** — motion, camera and timing untouched. You prepare that one frame in an image editor;
35
+ from there the video stage runs no pose estimator, segmenter, face tracker or text encoder. Two
36
+ inputs, three forward passes, 26 seconds a shot on one GPU.
37
+
38
+ **It is strongest where replacement is hardest: fast motion, and pose transfer accurate enough to
39
+ follow it.** Whipping heads, full kicks, jumps — tracked frame for frame, not smeared through.
40
+
41
+ ## Abstract
42
+
43
+ Controlled character replacement is usually built on intermediate representations — pose skeletons,
44
+ segmentation masks, background plates, face crops. Each needs its own extractor, and each extractor
45
+ is another model to run and another place to lose information. Recent work drops the skeleton but
46
+ keeps a mask channel. **Viggle-Animate uses neither.** Its two inputs are a driving video and one of
47
+ that video's own frames with the character repainted, and its only task is to propagate that edit
48
+ across the shot.
49
+
50
+ Because the reference is a frame of the clip, its pose, camera, framing and lighting already agree
51
+ with the footage, and nothing downstream has to align them again. The model is never told what the
52
+ new character is: no class, no identity encoder, and no user-provided text prompt.
53
+ **Viggle-Animate is a 33.1 B full finetune of MiniMax-H3's `ref2va` transformer, jointly distilled
54
+ with DMD to three forward passes.** In a matched comparison on the same machine and B200 GPU, using
55
+ the same source videos, output resolution and frame count, it renders 124 frames in 26 s, 6.1×
56
+ faster per clip than Wan2.2-Animate-14B.
57
+
58
+ ## Method
59
+
60
+ Character replacement asks two questions at once: *what does the new character look like*, and *how
61
+ does it move through this shot*. Systems that condition on a standalone character photograph must
62
+ answer both, and reconciling a photograph with footage it was never part of is what the scaffolding
63
+ exists for.
64
+
65
+ State-of-the-art image models have finished that job. Give `gpt-image` a frame and an instruction
66
+ and it replaces the character while following the prompt exactly — transferring the pose, matching
67
+ the lighting, preserving the background. The hard reconciliation is already solved, once, on one
68
+ image. This model is the second half of that pipeline, not the whole of it.
69
+
70
+ <img src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/pipeline.svg"
71
+ alt="Two inputs — a driving video and one of its own frames, repainted in any image editor — enter Viggle-Animate. No pose skeleton, segmentation mask, face crop, background plate, depth map or user-provided text prompt enters the video model." width="100%">
72
+
73
+ Appearance enters only through the repainted frame; geometry enters only through the driving video.
74
+ **The text encoder is never loaded.** Conditioning is one frozen embedding shipped with the weights
75
+ ([`assets/fixed_prompt.txt`](assets/fixed_prompt.txt)), identical for every render.
76
+
77
+ Left panel is the driving video, right panel is this model:
78
+
79
+ <video autoplay controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/hero-duo.mp4"></video>
80
+
81
+ <video autoplay controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/hero-corgi.mp4"></video>
82
+
83
+ **It is fast twice over.** Once the repainted frame exists there is nothing else to run — no pose
84
+ estimator, no segmenter, no face tracker, no text encoder. And the sampler is distilled, so a
85
+ finished clip is three forward passes rather than thirty. The two compound: one model, one GPU,
86
+ and no orchestration to get wrong.
87
+
88
+ The distillation is **joint, across two teachers split by noise level.** Our finetune supervises the
89
+ high-noise end of the schedule, where the replacement itself is decided — it is the model that gets
90
+ the swap right. The original MiniMax-H3 supervises the low-noise end, where detail and texture are
91
+ decided — it is the model with the better image quality. Distilling each end against the teacher that
92
+ owns it keeps both properties in one student, instead of inheriting the finetune's visual
93
+ regressions along with its replacement ability.
94
+
95
+ **It generalizes past humans**, because nothing in the loop assumes one. A pose skeleton has a neck
96
+ and two arms; a mask has a person-shaped hole. We have neither, so the model holds no representation
97
+ that a character must be a person. What it can animate is bounded by what you can paint.
98
+
99
+ ## Efficiency
100
+
101
+ <div align="center">
102
+ <div style="display:flex;gap:10px;flex-wrap:wrap;justify-content:center;margin:20px 0 6px;text-align:left">
103
+
104
+ <div style="flex:1 1 170px;border:1px solid rgba(128,128,128,.35);border-top:3px solid #00D94F;border-radius:10px;padding:12px 14px">
105
+ <div style="font-size:1.7em;font-weight:700;line-height:1.15">26 s</div>
106
+ <b>per render</b><br>
107
+ <small style="opacity:.7">124 frames at 24 fps, 480×832, a single B200</small>
108
+ </div>
109
+
110
+ <div style="flex:1 1 170px;border:1px solid rgba(128,128,128,.35);border-top:3px solid #00D94F;border-radius:10px;padding:12px 14px">
111
+ <div style="font-size:1.7em;font-weight:700;line-height:1.15">3</div>
112
+ <b>forward passes</b><br>
113
+ <small style="opacity:.7"><code>--steps 4</code> names four sigma boundaries, so three passes</small>
114
+ </div>
115
+
116
+ <div style="flex:1 1 170px;border:1px solid rgba(128,128,128,.35);border-top:3px solid #00D94F;border-radius:10px;padding:12px 14px">
117
+ <div style="font-size:1.7em;font-weight:700;line-height:1.15">2</div>
118
+ <b>inputs</b><br>
119
+ <small style="opacity:.7">a clip, and one of its own frames repainted</small>
120
+ </div>
121
+
122
+ <div style="flex:1 1 170px;border:1px solid rgba(128,128,128,.35);border-top:3px solid #00D94F;border-radius:10px;padding:12px 14px">
123
+ <div style="font-size:1.7em;font-weight:700;line-height:1.15">0</div>
124
+ <b>other models</b><br>
125
+ <small style="opacity:.7">in the video stage — no pose estimator, segmenter, face tracker or text encoder</small>
126
+ </div>
127
+
128
+ </div>
129
+ </div>
130
+
131
+ One B200, 480×832, 124 frames at 24 fps, bf16, no compile, no offload. Wan ran its documented
132
+ replacement recipe — 20 steps, `sample_shift 5.0`, `--refert_num 1 --replace_flag
133
+ --use_relighting_lora`, `--w_len 1 --h_len 1` — after its own preprocessing pass.
134
+
135
+ | | Viggle-Animate | Wan2.2-Animate-14B |
136
+ |---|---|---|
137
+ | Inputs | driving video + one repainted frame | driving video + character image, then a **preprocessing pass** producing pose, face, mask and background tracks |
138
+ | Render, after weights load | **26 s** | 160 s |
139
+ | — of which sampling | **13.6 s** | 140 s |
140
+ | Forward passes | **3** | 40 (20 steps × 2 chunks) |
141
+ | Parameters | 33.1 B | 17.3 B |
142
+
143
+ **6.1× faster per render, 10.3× on sampling alone.** Wan's preprocessing pass is not counted in
144
+ its 160 s.
145
+
146
+ ### Qualitative comparison
147
+
148
+ Four panels each: **painted reference · driving video · this model · Wan2.2-Animate-14B**, the last
149
+ at its documented replacement settings. The gap is widest under fast motion: where the comparison
150
+ smears, this model stays sharp and lands the pose on the right frame.
151
+
152
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/compare-fastmotion.mp4"></video>
153
+
154
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/compare-highkick.mp4"></video>
155
+
156
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/compare-cosplay.mp4"></video>
157
+
158
+ ## Generalization
159
+
160
+ The model is never told what it is animating, so how far the character can get from a person is an
161
+ empirical question rather than a list of supported categories. Three panels each: **painted
162
+ reference · driving video · this model.** Every clip below is one paint and one render at the
163
+ shipped defaults, `--seed 42` — no best-of-N.
164
+
165
+ **Animals.** Ears, eye patches and flippers move on limbs the driving clip does not have — the paint
166
+ places them, the render animates them as if they had always been arms and a head.
167
+
168
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-corgi.mp4"></video>
169
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-panda.mp4"></video>
170
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-penguin.mp4"></video>
171
+
172
+ **Not humanoid.** The airliner is the hardest case we have: the paint binds wings to arms and landing
173
+ gear to legs, and the model's job is to keep that binding for 124 frames. The robot has to relight
174
+ specular metal as it turns.
175
+
176
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-airliner.mp4"></video>
177
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-robot.mp4"></video>
178
+
179
+ **Stylized.** The clay figure holds its style boundary for the whole clip.
180
+
181
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-claymation.mp4"></video>
182
+
183
+ **More than one character.** The work moves into the paint prompt, which has to bind each one to a
184
+ position — "the one on the left". The last clip is a wide arena shot, each figure a few dozen pixels
185
+ tall.
186
+
187
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-anime-duo.mp4"></video>
188
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-photoreal-duo.mp4"></video>
189
+ <video controls muted loop playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/swap-wushu-animals.mp4"></video>
190
+
191
+ ## Limitations
192
+
193
+ **It inherits the image edit.** What the paint does not show, the model will not add, and where
194
+ paint and video disagree the video wins. Appearance comes from the paint but shape comes from the
195
+ driving pose: a LEGO minifigure kept its palette and yellow claw hands, yet reverted to human
196
+ anatomy — the airliner held because the paint tied its wings to real arms.
197
+
198
+ ~~**Lip-sync is weak.** Mouth shapes do not track speech closely in close-ups. Identity and expression
199
+ hold; it is the sync that lags, and we believe that is a training-data limit rather than anything
200
+ structural.~~
201
+
202
+ **Retracted — and since improved.** Broader user testing already put lip-sync and facial
203
+ expression ahead of our own first read of them. The rest turned out not to be a training-data limit
204
+ either. The model animates the mouth to *its own* audio track, and the fixed prompt asks that track
205
+ to be silence — so in the shipped configuration there is nothing for the mouth to sync to. Give it a
206
+ real soundtrack instead: encode the driving clip's audio with the audio VAE and hold it in the
207
+ target audio rows as a **clean** latent for the whole denoise (H3's flow convention is reversed, so
208
+ clean is `t = 1`). The audio rows stop being something the model predicts and become something it
209
+ conditions on, and the mouth tracks that speech. This is an inference-time change only — no
210
+ retraining, no new weights — and the pairing of noisy video rows with clean audio rows is not a
211
+ state the model was trained on, but it holds up across the clips we have run.
212
+ Pass `--audio` to do it here; the [hosted demo](https://huggingface.co/spaces/Viggle/viggle-animate)
213
+ does it automatically whenever the clip you upload carries sound.
214
+
215
+ **Turn the sound on for these two.** The driving clip, then the same shot with a different actor in
216
+ it, speaking her lines:
217
+
218
+ <video controls playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/audio-pin-driving.mp4"></video>
219
+
220
+ <video controls playsinline width="100%" src="https://huggingface.co/Viggle/Viggle-Animate/resolve/main/examples/media/audio-pin-out.mp4"></video>
221
+
222
+ All three files ship with the weights — the driving clip, the repainted still it is animated from
223
+ ([`audio-pin-ref.png`](examples/media/audio-pin-ref.png)), and the take above — so the second clip
224
+ is a command rather than a claim. The same file is passed twice, once for the motion and once for
225
+ the soundtrack:
226
+
227
+ ```bash
228
+ python inference/sample.py --model-dir ./MiniMax-H3 \
229
+ --cond examples/media/audio-pin-driving.mp4 \
230
+ --ref examples/media/audio-pin-ref.png \
231
+ --audio examples/media/audio-pin-driving.mp4 \
232
+ --out audio-pin-out.mp4
233
+ ```
234
+
235
+ Everything else is the shipped default: 124 frames at 832x480, four steps, `--seed 42`; 51 s of
236
+ render on a B200 after the weights load, and it comes back at 46 dB against the file above. The
237
+ track you get back is the driving clip's own audio put through the audio VAE and decoded again —
238
+ same level, peaks rounded off (0.95 to 0.76 here); that round trip is the evidence the rows really
239
+ were pinned. Drop `--audio` and the same command writes a file whose audio track is digital silence,
240
+ sample for sample, and a mouth that moves without saying anything.
241
+
242
+ **Complex scenes are harder than single subjects.** Several characters at once, close interaction
243
+ between them, and shots that cut are all cases where quality drops off — enough that we would not
244
+ call them solved.
245
+
246
+ **We are training a substantially better model right now**, aimed squarely at what is left. This
247
+ release is the version we can ship today, not the ceiling.
248
+
249
+ ## What this repository contains
250
+
251
+ Two parts, both derived from [`MiniMaxAI/MiniMax-H3`](https://huggingface.co/MiniMaxAI/MiniMax-H3)'s
252
+ `ref2va` transformer:
253
+
254
+ | | |
255
+ |---|---|
256
+ | `transformer/` | 33.1 B, bf16, 14 shards. A **full finetune** of the base `transformer_ref` on a character-replacement objective |
257
+ | `lora/` | rank 128 over 302 linear layers, 2.5 GB. A **DMD2-distilled** delta on that finetune — this is what collapses the sampler to three forward passes |
258
+
259
+ The LoRA is a delta on the *finetuned* transformer — loading it onto stock `transformer_ref` produces
260
+ garbage.
261
+
262
+ ## Quickstart
263
+
264
+ This repository ships only the transformer and the LoRA — the VAE, audio VAE and schedulers load
265
+ from your own copy of the base model. Inference touches 11 GB of its 269 GB:
266
+
267
+ ```bash
268
+ hf download MiniMaxAI/MiniMax-H3 --local-dir ./MiniMax-H3 \
269
+ --include "modular_model_index.json" "vae/*" "audio_vae/*" \
270
+ "scheduler/*" "audio_scheduler/*" "assets/ref2va.mp4"
271
+
272
+ hf download Viggle/Viggle-Animate --local-dir ./Viggle-Animate
273
+
274
+ pip install torch "git+https://github.com/huggingface/diffusers@d6726f3" av
275
+
276
+ python Viggle-Animate/inference/sample.py \
277
+ --model-dir ./MiniMax-H3 \
278
+ --cond driving.mp4 --ref ref.png --out swapped.mp4
279
+ ```
280
+
281
+ `d6726f3` is the tested `diffusers` commit; the upstream `minimax_h3` modular pipeline is enough,
282
+ no fork or patch. The last `--include` is the clip [`examples/demo.sh`](examples/demo.sh) needs.
283
+
284
+ **One 80 GB card is not enough at bf16** — the transformer is 62 GiB resident and a 480×832 /
285
+ 124-frame render peaks at 80.1 GiB allocated. Use a card with ≥ 96 GB, or pass `--offload` to
286
+ stream blocks from CPU (~12 GB resident, much slower).
287
+
288
+ **It also runs quantized on consumer hardware.** We deploy it on a single **RTX 5090 (32 GB)**:
289
+ NVFP4 weights — 4.5 bits/param, dispatching to the real sm_120 cutlass block-scaled kernel, 2.70×
290
+ bf16 per compiled linear — plus a low-rank `adaln_proj` and `torch.compile`. Quantization alone is
291
+ not enough for 32 GB: 13.0 B of the 33.1 B parameters sit in `adaln_proj`, which the linear-layer
292
+ quantizer does not touch. That deployment path is not shipped in this repository.
293
+
294
+ Defaults are the evaluated configuration: `--steps 4 --flow-shift 3 --num-frames 124` (≈ 5.2 s at
295
+ 24 fps) `--seed 42`. Output geometry follows the driving clip and must be a multiple of 32 on both
296
+ axes. Weights load in ~21 s, once per process. **Four steps is the operating point, not a shortcut**
297
+ — the distilled model already renders sharper than its teacher, and raising the step count tips that
298
+ into over-sharpening. `--steps 4` is four sigma boundaries and therefore three forward passes.
299
+ `sample.py` drops the driving clip's audio at input and the model emits its own track, which the
300
+ fixed prompt asks to be silence. `--audio <file>` overrides that and pins a real soundtrack instead,
301
+ which is what makes the mouth track speech — see [Limitations](#limitations).
302
+
303
+ Pull a frame with `ffmpeg -ss 1.5 -i driving.mp4 -frames:v 1 ref.png` and edit it at the same
304
+ resolution. **It does not have to be the first frame.** The still is passed to the model with no
305
+ frame index, so nothing downstream knows where in the clip it came from — pick whichever frame shows
306
+ the character most clearly, front-on and unoccluded. Name the change, and pin down what must *not*
307
+ change: pose, hands, props, framing, background, light. An editor that quietly reframes the shot will
308
+ fight the driving motion. Prefer clips that keep one side to camera, and bind any new limb to a real
309
+ one.
310
+
311
+ [`examples/demo.sh`](examples/demo.sh) runs a swap end to end on a clip that ships with the base
312
+ model, so it needs no media from you — and it is the fastest way to check that the LoRA loaded.
313
+
314
+ ## Citation
315
+
316
+ ```bibtex
317
+ @misc{viggle2026animate,
318
+ title = {Viggle-Animate: Character Replacement in Video from a Single Repainted Frame},
319
+ author = {Viggle Research},
320
+ year = {2026},
321
+ url = {https://huggingface.co/Viggle/Viggle-Animate}
322
+ }
323
+ ```
324
+
325
+ ## License
326
+
327
+ The weights are a Model Derivative of MiniMax H3, so the
328
+ [MiniMax H3 Community License](LICENSE) applies to them — read it before you redistribute them or
329
+ ship a product on them. Our changes are listed in [`MODIFICATIONS.md`](MODIFICATIONS.md).
330
+
331
+ The code in [`inference/`](inference) and [`examples/`](examples) is Apache 2.0
332
+ ([`LICENSE-CODE`](LICENSE-CODE)).
333
+
334
+ Music in the teaser at the top of this page: "Electrodoodle" by Kevin MacLeod
335
+ ([incompetech.com](https://incompetech.com)), licensed under
336
+ [Creative Commons: By Attribution 4.0](http://creativecommons.org/licenses/by/4.0/).
337
+
338
+ ## Intended use
339
+
340
+ This model exists to put a consenting performer into footage they did not shoot, and it will just
341
+ as readily put someone into footage they never agreed to appear in. Note where that decision is
342
+ made: **the identity comes from the frame you paint**, so an image editor's safeguards are upstream
343
+ of this model and none of them are in it. It cannot verify identity or consent. Do not run it on
344
+ people who have not agreed to it, label what you generate as AI-generated, and see Section V.5 of
345
+ the Agreement if you offer this as a service.
346
+
347
+ Powered by MiniMax H3.
assets/fixed_embed_fwd_anyframe.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e0ae90929caf7b790f5d0de599e868cc6c179f3e969577627971c9d78936a058
3
+ size 3714413
assets/fixed_prompt.txt ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <Video 1>: <0.2 seconds><1.2 seconds><2.2 seconds><3.2 seconds><4.2 seconds><5.0 seconds><Picture 1>: subject_definitions:
2
+ <Video 1> is the source video for the editing task.
3
+ <Picture 1> is one frame of the target video.
4
+
5
+ summary:
6
+ [video editing + character replacement] The target video is an edited version of <Video 1> in which every performer is replaced by a different person. <Picture 1> is one frame of the target video: it already shows the replacement, together with the background, camera framing and lighting it happens in. Every other frame of the target video shows those same people in that same place, following the motion, timing and camera of <Video 1>.
7
+
8
+ retention_analysis:
9
+ <Video 1> (source video editing): fully_preserved - the camera framing, the background, the lighting, and the motion and timing of every performance are copied frame for frame.
10
+ <Picture 1> (appears in [Shot 1]): reference - one frame of the target video, pixel for pixel. The people in it, their faces, hair, skin tone, build and clothing, and the background, the framing and the lighting are all taken from <Picture 1> and held unchanged from the first frame to the last.
11
+
12
+ detailed_description:
13
+ [Shot 1] The people of <Picture 1> perform exactly the motion of the corresponding performers in <Video 1>, in the same framing, on the same background, under the same lighting. Nothing outside the people changes. The camera framing never changes through the end of the video.
14
+
15
+ overall_soundscape:
16
+ No music and no speech.
examples/demo.sh ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ # Reproduce the demo on the model card, and in doing so check your install.
3
+ #
4
+ # This repository bundles no driving footage. It builds the driving clip from a demo
5
+ # video that ships with the base model, using the exact crop documented below, and
6
+ # pairs it with the repainted first frame in media/reference.png.
7
+ #
8
+ # The result should match media/output.mp4. On the same GPU model we get it back
9
+ # bit-identical; on different hardware bf16 kernel scheduling shifts things, and a mean
10
+ # absolute difference around 1.5/255 is normal. What matters is that it is the same
11
+ # elderly woman holding the same black lamb. If it comes back as the young man from the
12
+ # source video instead, the LoRA did not load. If it comes back as noise, the weights
13
+ # are wrong.
14
+ #
15
+ # ./examples/demo.sh /path/to/MiniMax-H3
16
+ #
17
+ # About a minute on a B200, most of it loading weights.
18
+
19
+ set -euo pipefail
20
+
21
+ MODEL_DIR="${1:?usage: demo.sh /path/to/MiniMax-H3}"
22
+ HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
23
+ OUT="${OUT:-$HERE/demo_out}"
24
+ mkdir -p "$OUT"
25
+
26
+ SRC="$MODEL_DIR/assets/ref2va.mp4"
27
+ [ -f "$SRC" ] || { echo "missing $SRC -- see the download command on the model card"; exit 1; }
28
+
29
+ # The source is 1344x768 and exactly 124 frames, which is the sampler's window. Crop a
30
+ # 512x768 portrait window around the figure; it stays in frame for the whole push-in, so
31
+ # no scaling and no padding are needed. media/reference.png is this crop's first frame,
32
+ # repainted.
33
+ ffmpeg -y -loglevel error -i "$SRC" \
34
+ -vf "crop=512:768:389:0" -frames:v 124 -an "$OUT/driving.mp4"
35
+
36
+ python "$HERE/../inference/sample.py" \
37
+ --model-dir "$MODEL_DIR" \
38
+ --cond "$OUT/driving.mp4" \
39
+ --ref "$HERE/media/reference.png" \
40
+ --out "$OUT/output.mp4"
41
+
42
+ echo
43
+ echo "wrote $OUT/output.mp4 -- compare against $HERE/media/output.mp4"
examples/media/audio-pin-driving.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fc3dd85a5be2e6246fb40afe46e337cb75d6c3558105fd81fa02f1ff286a5e92
3
+ size 4223283
examples/media/audio-pin-out.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:25fc5938c25a4dc48d93f388833358ed50cbb8876753ba75ae4e89045c63cf8e
3
+ size 389192
examples/media/audio-pin-ref.png ADDED

Git LFS Details

  • SHA256: 6bf55661c6e0b3ba9a0b14999462deab9d6de8333d35b6cbd360c3b3cbf855e7
  • Pointer size: 131 Bytes
  • Size of remote file: 442 kB
examples/media/before-after.png ADDED

Git LFS Details

  • SHA256: 958c7d84b72c4756c687401ae4bfb84e06015f2237c88b8e1f36e4411676b10c
  • Pointer size: 132 Bytes
  • Size of remote file: 1.34 MB
examples/media/community-tweet.png ADDED

Git LFS Details

  • SHA256: 84ce1d8e96c07aa3537a6a39f2bbd3f9de2900141a031e8142ce77aa52e1b10b
  • Pointer size: 131 Bytes
  • Size of remote file: 360 kB
examples/media/compare-cosplay.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6bd63ca750de511cbc6391664def22a9671f79d6885729c8d20bc92978fc8843
3
+ size 1099913
examples/media/compare-costume.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2a45ac3e161df80bff9e4b220e3282855821d450b1d14974b4e886609e001d8e
3
+ size 447259
examples/media/compare-fastmotion.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b79b19f959b370431dc5e090a80068c9046b07504064977d28d47d81e3ef50b7
3
+ size 1183899
examples/media/compare-gamechar.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:69e34a0ec75f588dfa3c2cdf882fab5bb7107e357e64ceb0d1881d8dbc23452c
3
+ size 858158
examples/media/compare-highkick.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6a137d9d19701eb1c75aa9e6b67776cfbf069619d9c0668885a717098f42fe37
3
+ size 847770
examples/media/compare-prop.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4128c0e35bf144b5d0e72c5837cc1958421a749337af395977de04fa07b92fea
3
+ size 678070
examples/media/compare-style.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:692ff38363a0162b965d0c6944e9adb98455c8a4e92419b3f9bb764ce9bd6778
3
+ size 536776
examples/media/hero-corgi.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4cb6c4bb9252352e257b9c8b153b5825d7b3a51227c263631169fa609663c3d5
3
+ size 425955
examples/media/hero-duo.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cbf81086b77ebbce703d33b9ced930d4f3cabd679c77fc764ad713a5e2bc2d27
3
+ size 187010
examples/media/output.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7b89a33a902cd92a1068307b701205029682bb5e5db0326775e3490767f1df04
3
+ size 880425
examples/media/pipeline.png ADDED

Git LFS Details

  • SHA256: 451cbcf8322877fbda31338f362a3748440dd48ec67b11ee7ef48e3f22ecad84
  • Pointer size: 131 Bytes
  • Size of remote file: 105 kB
examples/media/pipeline.svg ADDED
examples/media/reference.png ADDED

Git LFS Details

  • SHA256: 5a01824f3023f305bb04b054b236c498030c2e9a3b76e1a3ccacdddea19f49cb
  • Pointer size: 131 Bytes
  • Size of remote file: 541 kB
examples/media/swap-airliner.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5a7a31fd225e258633e208f13c18ff8523517834a9ea2c67b97ed060972becf3
3
+ size 397277
examples/media/swap-anime-duo.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:205700e55841a5fd22f517c6eb08c447068d0417a8e902bc91c863dede743d86
3
+ size 108986
examples/media/swap-claymation.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dd08b83d8a57e77408203710cf55eccdf8d1ee136c22123cc42ed8ac14434528
3
+ size 286267
examples/media/swap-corgi.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5954aeab0360ac8c1c5c62b5d35935520ac0feffdb8d11b8c55a6ff84f9d9239
3
+ size 272402
examples/media/swap-panda.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bec174f093bf553b39d1f022b6fedd46b9616a6ac30fa10a939753cf5e4a5150
3
+ size 390485
examples/media/swap-penguin.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ea2ccf8ce410331d194403903e2d1e836079b2d8fbafebf113265e52aafb3cd0
3
+ size 346127
examples/media/swap-photoreal-duo.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a2d7c7fb36c5b2779176484b0a2fcb1735578ea3cad2fa0a2c0b057218881554
3
+ size 108311
examples/media/swap-robot.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e8147fb8f85eca14e372ecd26f6d7b4556c649f5c980e0dd514a582ac8619077
3
+ size 194541
examples/media/swap-wushu-animals.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ed178d0c88053df75803d904bef26c83c81da78c33d575ec8f5897701b622a7
3
+ size 204001
examples/media/swap-wushu-robots.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:63cc5dfe297ca472b7c54a9ef32aaa575f84e6cc1088a42d01236e40888785b2
3
+ size 201465
examples/media/teaser.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cbe5dd05057bafa9f97750788523d600c8dc1963be181fe56431393b9558ecb0
3
+ size 14442381
inference/sample.py ADDED
@@ -0,0 +1,205 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Viggle-Animate: replace the performers in a video with the person in a still.
3
+
4
+ python inference/sample.py --cond driving.mp4 --ref character.png --out swapped.mp4
5
+
6
+ `--cond` supplies the motion, camera framing, background and lighting; `--ref` supplies
7
+ who is in it. Everything about the render except the people is copied from `--cond`.
8
+
9
+ By default the model generates its own soundtrack, and the fixed prompt asks for silence --
10
+ so there is nothing for the mouth to sync to. `--audio` pins a real one instead:
11
+
12
+ python inference/sample.py --cond driving.mp4 --ref character.png \
13
+ --audio driving.mp4 --out swapped.mp4
14
+
15
+ The soundtrack is encoded once and held in the target audio rows as a *clean* latent for the
16
+ whole denoise, so the model conditions on it rather than predicting it, and the mouth tracks
17
+ that speech. The track written to `--out` is then that same audio, back through the audio VAE.
18
+
19
+ The text encoder is never loaded. Conditioning comes from `assets/fixed_embed_fwd_anyframe.pt`,
20
+ a frozen 362 x 5120 tensor computed once from the fixed prompt in `assets/fixed_prompt.txt`,
21
+ so Qwen3-VL (63 GB of the base repo) stays on disk and the text block of the packed sequence
22
+ is 362 rows instead of several thousand. There is no per-clip prompt and no caption: nothing
23
+ in the output comes from text you write.
24
+
25
+ Needs `--model-dir` pointing at a local copy of MiniMaxAI/MiniMax-H3 for the VAE, the audio
26
+ VAE and the schedulers. This repository ships only the transformer and the LoRA.
27
+ """
28
+
29
+ import argparse
30
+ import os
31
+ import time
32
+
33
+ import torch
34
+ from diffusers import MiniMaxH3Transformer3DModel, ModularPipeline
35
+ from diffusers.modular_pipelines.minimax_h3 import (MiniMaxH3AudioReference, MiniMaxH3ImageReference,
36
+ MiniMaxH3VideoReference)
37
+ from diffusers.modular_pipelines.minimax_h3.before_encoder import MiniMaxH3Ref2VASetupStep
38
+ from diffusers.modular_pipelines.minimax_h3.encoders import MiniMaxH3Ref2VATextEncoderStep
39
+ from diffusers.modular_pipelines.minimax_h3.modular_pipeline import (align_num_frames,
40
+ audio_latent_num_frames)
41
+ from diffusers.utils.export_utils import encode_video
42
+
43
+ HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
44
+
45
+ parser = argparse.ArgumentParser()
46
+ parser.add_argument("--cond", required=True, help="the video whose motion, framing and background are kept")
47
+ parser.add_argument("--ref", required=True, help="a single still of the person to put in it")
48
+ parser.add_argument("--out", required=True)
49
+ parser.add_argument("--model-dir", required=True,
50
+ help="a local copy of MiniMaxAI/MiniMax-H3, for the VAE / audio VAE / schedulers")
51
+ parser.add_argument("--transformer", default=os.path.join(HERE, "transformer"))
52
+ parser.add_argument("--lora", default=os.path.join(HERE, "lora"))
53
+ parser.add_argument("--embed", default=os.path.join(HERE, "assets", "fixed_embed_fwd_anyframe.pt"))
54
+ parser.add_argument("--num-frames", type=int, default=124, help="at 24 fps; 124 frames is ~5.2 s")
55
+ parser.add_argument("--steps", type=int, default=4,
56
+ help="the distilled student's operating point. More is not monotonically better: "
57
+ "s4 is not a degraded s12")
58
+ parser.add_argument("--flow-shift", type=float, default=3.0,
59
+ help="the base model's released default is 12; the few-step student wants 3")
60
+ parser.add_argument("--height", type=int, default=None, help="defaults to the conditioning clip's own height")
61
+ parser.add_argument("--width", type=int, default=None, help="defaults to the conditioning clip's own width")
62
+ parser.add_argument("--short-edge", type=int, default=None,
63
+ help="the canvas both references are laid out on. Defaults to the conditioning clip's own "
64
+ "short edge, which is what this model was evaluated at")
65
+ parser.add_argument("--offload", action="store_true",
66
+ help="stream the transformer from CPU in groups of 5 blocks: ~12 GB resident instead of 62")
67
+ parser.add_argument("--audio", default=None,
68
+ help="pin the generated soundtrack to this file's audio -- usually the driving clip "
69
+ "itself -- so the mouth tracks real speech instead of the silence the fixed "
70
+ "prompt asks for. Any file PyAV can decode; a video's soundtrack is taken")
71
+ parser.add_argument("--seed", type=int, default=42)
72
+ args = parser.parse_args()
73
+
74
+ fixed = torch.load(args.embed, weights_only=False)
75
+
76
+
77
+ def use_fixed_embeds(self, components, state):
78
+ block_state = self.get_block_state(state)
79
+ block_state.prompt_embeds = fixed["prompt_embeds"].to(components._execution_device, torch.bfloat16)
80
+ block_state.text_token_tags = fixed["text_token_tags"]
81
+ self.set_block_state(state, block_state)
82
+ return components, state
83
+
84
+
85
+ MiniMaxH3Ref2VATextEncoderStep.__call__ = use_fixed_embeds
86
+
87
+ # `--audio` holds the *target* audio rows at a real soundtrack, clean, for the whole denoise, rather
88
+ # than letting the model generate them. Two of the three things that takes have no argument on the
89
+ # pipeline, so they are patched here; the third is the `audio_latents=` passed to the call below.
90
+ if args.audio:
91
+ from diffusers.modular_pipelines.minimax_h3.before_denoise import MiniMaxH3SetTimestepsStep
92
+ from diffusers.modular_pipelines.minimax_h3.denoise import MiniMaxH3LoopSchedulerStep
93
+
94
+ # (1) Those rows carry finished audio, so they have to be told they are clean. H3's flow
95
+ # convention is reversed -- t = 1 is clean, not 0 -- and the library itself passes a literal 1.0
96
+ # for a *reference* soundtrack. `audio_timestep` is positional argument 6.
97
+ _build_row_timesteps = MiniMaxH3SetTimestepsStep.build_row_timesteps
98
+ MiniMaxH3SetTimestepsStep.build_row_timesteps = staticmethod(
99
+ lambda *a: _build_row_timesteps(*a[:6], 1.0, *a[7:]))
100
+
101
+ # (2) ...and the scheduler must never write them, or the first step would walk them off the
102
+ # soundtrack. Only the video rows are stepped. `num_condition_audio_rows` deliberately stays 0:
103
+ # raising it empties the decoder's `audio_latents[num_condition_audio_rows:]` slice and trips the
104
+ # reference-count check, and these rows are a pinned target, not a reference.
105
+ @torch.no_grad()
106
+ def video_only_step(self, components, block_state, i, t):
107
+ n = block_state.num_condition_video_rows
108
+ block_state.latents[n:] = components.scheduler.step(
109
+ block_state.noise_pred[0, n:].float(), t, block_state.latents[n:], return_dict=False)[0]
110
+ return components, block_state
111
+
112
+ MiniMaxH3LoopSchedulerStep.__call__ = video_only_step
113
+
114
+ # The reference order is frozen: the presentation names `<Video 1>` then `<Picture 1>`, and that order
115
+ # advances the shared rotary clock, so it is part of the layout rather than a detail of the prompt. The
116
+ # driving clip's own soundtrack is dropped here, as it is in training: `--audio` puts one back, but as
117
+ # the *target* to be matched rather than as a reference to be imitated.
118
+ video = MiniMaxH3VideoReference.from_file(args.cond)
119
+ video.audio, video.sample_rate = None, None
120
+
121
+ # Passing an orientation that disagrees with the clip silently generates a transposed video, so the output
122
+ # geometry is derived from the clip rather than typed.
123
+ height = args.height or video.frames.shape[1]
124
+ width = args.width or video.frames.shape[2]
125
+ short_edge = args.short_edge or min(height, width)
126
+
127
+ pipe = ModularPipeline.from_pretrained(args.model_dir, workflow="ref2va")
128
+ # Both references are pinned to the target's own short edge. The base model's released defaults (768 for the
129
+ # video reference, 2048 for the image) put the references on a grid the target never shares; this model was
130
+ # finetuned and evaluated with them nested, and changing it changes the take.
131
+ pipe.register_to_config(canvas_short_edge=short_edge,
132
+ canvas_max_pixels=short_edge * max(height, width),
133
+ reference_image_short_edge=short_edge)
134
+
135
+ t0 = time.time()
136
+ # `transformer_ref` is deliberately absent: our finetune replaces it outright, so loading the base copy
137
+ # first would read 62 GB off disk only to drop it.
138
+ pipe.load_components(names=["vae", "audio_vae", "scheduler", "audio_scheduler"],
139
+ pretrained_model_name_or_path=args.model_dir, dtype=torch.bfloat16)
140
+ pipe.transformer_ref = MiniMaxH3Transformer3DModel.from_pretrained(args.transformer, torch_dtype=torch.bfloat16)
141
+ # `prefix=None` and the explicit `weight_name` are both required. The loader defaults to looking for a `.bin`
142
+ # (raises) and to filtering keys for a `transformer.` prefix, which these bare keys do not have -- that
143
+ # mismatch loads *nothing* and only warns, so the default would silently render the un-distilled model.
144
+ pipe.transformer_ref.load_lora_adapter(args.lora, weight_name="pytorch_lora_weights.safetensors", prefix=None)
145
+ pipe.scheduler.set_shift(args.flow_shift)
146
+
147
+ # The pipeline snaps `num_frames` up to the next `17 * n + 5` the video VAE can encode. Doing it here
148
+ # too means the pinned audio is cut to the length that is really rendered rather than the one asked
149
+ # for -- off by one grid step and the rows no longer line up with the video.
150
+ num_frames = align_num_frames(args.num_frames, pipe.vae_frames_per_chunk, pipe.vae_latents_per_chunk)
151
+
152
+ if args.offload:
153
+ pipe.transformer_ref.enable_group_offload(
154
+ onload_device=torch.device("cuda"), offload_type="block_level", num_blocks_per_group=5,
155
+ non_blocking=True, use_stream=True, record_stream=True)
156
+ pipe.vae.to("cuda")
157
+ pipe.audio_vae.to("cuda")
158
+ else:
159
+ pipe.to("cuda")
160
+ print(f"loaded in {time.time() - t0:.0f}s; canvas {height}x{width}, references on short edge {short_edge}")
161
+
162
+ # The soundtrack becomes target rows the same way the pipeline turns a *reference* soundtrack into
163
+ # reference rows: truncate at the source rate, resample once, take the posterior mean, normalize.
164
+ # Reusing its own helper is what keeps the two paths from drifting apart.
165
+ audio_latents = None
166
+ if args.audio:
167
+ # One video frame more than the render needs, so the encoder cannot come up short. A source
168
+ # shorter than the grid it renders on is padded, and that tail is real silence.
169
+ n_samp = round((num_frames + 1) / pipe.fps * pipe.audio_sampling_rate)
170
+ track = MiniMaxH3AudioReference.from_file(args.audio)
171
+ wav = MiniMaxH3Ref2VASetupStep._normalize_audio_condition(
172
+ track.audio, track.sample_rate or pipe.audio_sampling_rate, pipe.audio_sampling_rate,
173
+ max_duration=(num_frames + 1) / pipe.fps)
174
+ have = wav.shape[1]
175
+ wav = torch.nn.functional.pad(wav, (0, max(0, n_samp - have)))[:, :n_samp]
176
+ with torch.no_grad():
177
+ # `encode` casts to the encoder's own dtype, so a float32 waveform is fine against a bf16 VAE.
178
+ posterior = pipe.audio_vae.encode(wav[:, None].to(pipe.audio_vae.device), return_dict=False)[0]
179
+ mean = torch.tensor(pipe.audio_vae.config.latents_mean).view(1, 1, -1)
180
+ std = torch.tensor(pipe.audio_vae.config.latents_std).view(1, 1, -1)
181
+ n_lat = audio_latent_num_frames(num_frames, pipe.fps)
182
+ # Channel-major rows: the two stereo channels are two batch items of the mono audio VAE.
183
+ rows = (posterior.mode().float().cpu().transpose(1, 2)[:, :n_lat] - mean) / std
184
+ if rows.shape[1] != n_lat:
185
+ raise RuntimeError(f"the soundtrack encoded to {rows.shape[1]} latents, short of the {n_lat} "
186
+ f"that {num_frames} frames need")
187
+ audio_latents = rows.permute(0, 2, 1).contiguous()
188
+ print(f"pinned {min(have, n_samp) / pipe.audio_sampling_rate:.2f}s of audio -> "
189
+ f"{tuple(audio_latents.shape)}, over {n_samp / pipe.audio_sampling_rate:.2f}s of video")
190
+
191
+ t0 = time.time()
192
+ result = pipe(
193
+ prompt=fixed["presentation"],
194
+ references=[video, MiniMaxH3ImageReference.from_file(args.ref)],
195
+ num_frames=num_frames,
196
+ height=height,
197
+ width=width,
198
+ num_inference_steps=args.steps,
199
+ audio_latents=audio_latents,
200
+ generator=torch.Generator().manual_seed(args.seed),
201
+ output=["videos", "audio", "sampling_rate"],
202
+ )
203
+ encode_video(result["videos"][0], fps=24, output_path=args.out,
204
+ audio=result["audio"][0], audio_sample_rate=result["sampling_rate"])
205
+ print(f"{time.time() - t0:.0f}s, peak {torch.cuda.max_memory_allocated() / 2**30:.1f} GiB -> {args.out}")
lora/pytorch_lora_weights.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bcdd60965d95e45a0b06f7d27435f21cbc2c8bd4f26f7b1cd8e66af7122c4af8
3
+ size 2666439896
requirements.txt ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ # Tested with these. Nothing here is pinned tightly except diffusers, which is pinned to a
2
+ # commit rather than a release: the `minimax_h3` modular pipeline is newer than any tag.
3
+ torch==2.9.1
4
+ git+https://github.com/huggingface/diffusers@d6726f3
5
+ transformers==4.57.3
6
+ safetensors==0.7.0
7
+ av==16.1.0
transformer/config.json ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "MiniMaxH3Transformer3DModel",
3
+ "_diffusers_version": "0.36.0.dev0",
4
+ "num_attention_heads": 56,
5
+ "attention_head_dim": 128,
6
+ "hidden_size": 5376,
7
+ "num_layers": 50,
8
+ "num_refiner_layers": 2,
9
+ "ffn_dim": 14336,
10
+ "in_channels": 24,
11
+ "audio_in_channels": 32,
12
+ "patch_size": [
13
+ 1,
14
+ 2,
15
+ 2
16
+ ],
17
+ "text_dim": 5120,
18
+ "freq_dim": 256,
19
+ "time_embed_hidden_dim": 5376,
20
+ "time_embed_dim": 2688,
21
+ "rope_freq_dim": 16,
22
+ "rope_theta": 10000.0,
23
+ "norm_eps": 1e-05,
24
+ "qk_norm_eps": 1e-05,
25
+ "final_norm_eps": 1e-05
26
+ }
transformer/diffusion_pytorch_model-00001-of-00014.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fb30791125c5814baaefb5bac1e3de2cb9595e027866ffa3a1d1c946789ef427
3
+ size 4945656288
transformer/diffusion_pytorch_model-00002-of-00014.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:58d254c72eb9637b86b11be13f2d648e4dd7c37faddf0fd09f7808faef83e6a4
3
+ size 4490213792
transformer/diffusion_pytorch_model-00003-of-00014.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bd8410138703641d2f6475f5650aaaa6e2c335b12a992b88aab4cb6465666f3d
3
+ size 4701942704
transformer/diffusion_pytorch_model-00004-of-00014.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c42d981e9fa75379f14622855f045f2dee79f75b91ced7eeb0247bd4aab1cd95
3
+ size 4933368952
transformer/diffusion_pytorch_model-00005-of-00014.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2b6fb580aef372b651083570c7c25edfb9dce065b49a3f605499b31109d0f9d2
3
+ size 4567284240
transformer/diffusion_pytorch_model-00006-of-00014.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a3183d396d2fc2d695145c98988742849658561e7e2f69ce8c3a78d414a363a0
3
+ size 4701942704