1ripon1 commited on
Commit
7344bef
·
verified ·
1 Parent(s): 5d2d30c

Upload folder using huggingface_hub

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +1 -0
  2. .gitignore +49 -0
  3. .gradio/certificate.pem +31 -0
  4. Custom Resolutions Instructions.txt +16 -0
  5. Dockerfile +92 -0
  6. LICENSE.txt +455 -0
  7. README.md +522 -8
  8. defaults/ReadMe.txt +13 -0
  9. defaults/ace_step_v1.json +19 -0
  10. defaults/ace_step_v1_5.json +22 -0
  11. defaults/ace_step_v1_5_turbo_lm_0_6b.json +22 -0
  12. defaults/ace_step_v1_5_turbo_lm_1_7b.json +23 -0
  13. defaults/ace_step_v1_5_turbo_lm_4b.json +23 -0
  14. defaults/ace_step_v1_5_xl.json +22 -0
  15. defaults/ace_step_v1_5_xl_turbo_lm_0_6b.json +22 -0
  16. defaults/ace_step_v1_5_xl_turbo_lm_1_7b.json +23 -0
  17. defaults/ace_step_v1_5_xl_turbo_lm_4b.json +23 -0
  18. defaults/alpha.json +19 -0
  19. defaults/alpha2.json +19 -0
  20. defaults/alpha2_sf.json +18 -0
  21. defaults/alpha_sf.json +17 -0
  22. defaults/animate.json +17 -0
  23. defaults/chatterbox.json +22 -0
  24. defaults/chrono_edit.json +13 -0
  25. defaults/chrono_edit_distill.json +16 -0
  26. defaults/dramabox_audio.json +23 -0
  27. defaults/fantasy.json +11 -0
  28. defaults/flf2v_720p.json +16 -0
  29. defaults/flux.json +15 -0
  30. defaults/flux2_dev.json +16 -0
  31. defaults/flux2_dev_nvfp4.json +15 -0
  32. defaults/flux2_klein_4b.json +16 -0
  33. defaults/flux2_klein_9b.json +14 -0
  34. defaults/flux2_klein_base_4b.json +16 -0
  35. defaults/flux2_klein_base_9b.json +16 -0
  36. defaults/flux_chroma.json +17 -0
  37. defaults/flux_chroma_radiance.json +17 -0
  38. defaults/flux_dev_kontext.json +16 -0
  39. defaults/flux_dev_kontext_dreamomni2.json +19 -0
  40. defaults/flux_dev_umo.json +23 -0
  41. defaults/flux_dev_uso.json +16 -0
  42. defaults/flux_krea.json +15 -0
  43. defaults/flux_schnell.json +16 -0
  44. defaults/flux_srpo.json +14 -0
  45. defaults/flux_srpo_uso.json +16 -0
  46. defaults/fun_inp.json +13 -0
  47. defaults/fun_inp_1.3B.json +11 -0
  48. defaults/heartmula_oss_3b.json +14 -0
  49. defaults/heartmula_rl_oss_3b_20260123.json +15 -0
  50. defaults/hidream_o1.json +18 -0
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ models/TTS/index_tts2/utils/maskgct/models/codec/facodec/modules/JDC/bst.t7 filter=lfs diff=lfs merge=lfs -text
.gitignore ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ .*
2
+ *.py[cod]
3
+ # *.jpg
4
+ *.jpeg
5
+ # *.png
6
+ *.gif
7
+ *.bmp
8
+ *.mp4
9
+ *.webm
10
+ *.npy
11
+ *.mov
12
+ *.mkv
13
+ *.log
14
+ *.zip
15
+ *.pt
16
+ *.pth
17
+ *.ckpt
18
+ *.safetensors
19
+ #*.json
20
+ # *.txt
21
+ *.backup
22
+ *.pkl
23
+ *.html
24
+ *.pdf
25
+ *.whl
26
+ *.exe
27
+ cache
28
+ __pycache__/
29
+ storage/
30
+ samples/
31
+ !.gitignore
32
+ !requirements.txt
33
+ .DS_Store
34
+ *DS_Store
35
+ google/
36
+ finetunes/
37
+ outputs/
38
+ outputs2/
39
+ gradio_outputs/
40
+ ckpts/
41
+ loras/
42
+ loras_i2v/
43
+
44
+ /settings/
45
+
46
+ wgp_config.json
47
+ plugins_local.json
48
+ loras_url_cache.json
49
+ loras_url_cache_v2.json
.gradio/certificate.pem ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ -----BEGIN CERTIFICATE-----
2
+ MIIFazCCA1OgAwIBAgIRAIIQz7DSQONZRGPgu2OCiwAwDQYJKoZIhvcNAQELBQAw
3
+ TzELMAkGA1UEBhMCVVMxKTAnBgNVBAoTIEludGVybmV0IFNlY3VyaXR5IFJlc2Vh
4
+ cmNoIEdyb3VwMRUwEwYDVQQDEwxJU1JHIFJvb3QgWDEwHhcNMTUwNjA0MTEwNDM4
5
+ WhcNMzUwNjA0MTEwNDM4WjBPMQswCQYDVQQGEwJVUzEpMCcGA1UEChMgSW50ZXJu
6
+ ZXQgU2VjdXJpdHkgUmVzZWFyY2ggR3JvdXAxFTATBgNVBAMTDElTUkcgUm9vdCBY
7
+ MTCCAiIwDQYJKoZIhvcNAQEBBQADggIPADCCAgoCggIBAK3oJHP0FDfzm54rVygc
8
+ h77ct984kIxuPOZXoHj3dcKi/vVqbvYATyjb3miGbESTtrFj/RQSa78f0uoxmyF+
9
+ 0TM8ukj13Xnfs7j/EvEhmkvBioZxaUpmZmyPfjxwv60pIgbz5MDmgK7iS4+3mX6U
10
+ A5/TR5d8mUgjU+g4rk8Kb4Mu0UlXjIB0ttov0DiNewNwIRt18jA8+o+u3dpjq+sW
11
+ T8KOEUt+zwvo/7V3LvSye0rgTBIlDHCNAymg4VMk7BPZ7hm/ELNKjD+Jo2FR3qyH
12
+ B5T0Y3HsLuJvW5iB4YlcNHlsdu87kGJ55tukmi8mxdAQ4Q7e2RCOFvu396j3x+UC
13
+ B5iPNgiV5+I3lg02dZ77DnKxHZu8A/lJBdiB3QW0KtZB6awBdpUKD9jf1b0SHzUv
14
+ KBds0pjBqAlkd25HN7rOrFleaJ1/ctaJxQZBKT5ZPt0m9STJEadao0xAH0ahmbWn
15
+ OlFuhjuefXKnEgV4We0+UXgVCwOPjdAvBbI+e0ocS3MFEvzG6uBQE3xDk3SzynTn
16
+ jh8BCNAw1FtxNrQHusEwMFxIt4I7mKZ9YIqioymCzLq9gwQbooMDQaHWBfEbwrbw
17
+ qHyGO0aoSCqI3Haadr8faqU9GY/rOPNk3sgrDQoo//fb4hVC1CLQJ13hef4Y53CI
18
+ rU7m2Ys6xt0nUW7/vGT1M0NPAgMBAAGjQjBAMA4GA1UdDwEB/wQEAwIBBjAPBgNV
19
+ HRMBAf8EBTADAQH/MB0GA1UdDgQWBBR5tFnme7bl5AFzgAiIyBpY9umbbjANBgkq
20
+ hkiG9w0BAQsFAAOCAgEAVR9YqbyyqFDQDLHYGmkgJykIrGF1XIpu+ILlaS/V9lZL
21
+ ubhzEFnTIZd+50xx+7LSYK05qAvqFyFWhfFQDlnrzuBZ6brJFe+GnY+EgPbk6ZGQ
22
+ 3BebYhtF8GaV0nxvwuo77x/Py9auJ/GpsMiu/X1+mvoiBOv/2X/qkSsisRcOj/KK
23
+ NFtY2PwByVS5uCbMiogziUwthDyC3+6WVwW6LLv3xLfHTjuCvjHIInNzktHCgKQ5
24
+ ORAzI4JMPJ+GslWYHb4phowim57iaztXOoJwTdwJx4nLCgdNbOhdjsnvzqvHu7Ur
25
+ TkXWStAmzOVyyghqpZXjFaH3pO3JLF+l+/+sKAIuvtd7u+Nxe5AW0wdeRlN8NwdC
26
+ jNPElpzVmbUq4JUagEiuTDkHzsxHpFKVK7q4+63SM1N95R1NbdWhscdCb+ZAJzVc
27
+ oyi3B43njTOQ5yOf+1CceWxG1bQVs5ZufpsMljq4Ui0/1lvh+wjChP4kqKOJ2qxq
28
+ 4RgqsahDYVvTH9w7jXbyLeiNdd8XM2w9U/t7y0Ff/9yi0GE44Za4rF2LN9d11TPA
29
+ mRGunUHBcnWEvgJBQl9nJEiU0Zsnvgc/ubhPgXRR4Xq37Z0j4r7g1SgEEzwxA57d
30
+ emyPxgcYxn/eR44/KJ4EBs+lVDR3veyJm+kXQ99b21/+jh5Xos1AnX5iItreGCc=
31
+ -----END CERTIFICATE-----
Custom Resolutions Instructions.txt ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ You can override the choice of Resolutions offered by WanGP, if you create a file "resolutions.json" in the main WanGP folder.
2
+ This file is composed of a list of 2 elements sublists. Each 2 elements sublist should have the format ["Label", "WxH"] where W, H are respectively the Width and Height of the resolution. Please make sure that W and H are multiples of 16. The letter "x" should be placed inbetween these two dimensions.
3
+
4
+ Here is below a sample "resolutions.json" file :
5
+
6
+ [
7
+ ["1280x720 (16:9, 720p)", "1280x720"],
8
+ ["720x1280 (9:16, 720p)", "720x1280"],
9
+ ["1024x1024 (1:1, 720p)", "1024x1024"],
10
+ ["1280x544 (21:9, 720p)", "1280x544"],
11
+ ["544x1280 (9:21, 720p)", "544x1280"],
12
+ ["1104x832 (4:3, 720p)", "1104x832"],
13
+ ["832x1104 (3:4, 720p)", "832x1104"],
14
+ ["960x960 (1:1, 720p)", "960x960"],
15
+ ["832x480 (16:9, 480p)", "832x480"]
16
+ ]
Dockerfile ADDED
@@ -0,0 +1,92 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ FROM nvidia/cuda:12.8.1-cudnn-devel-ubuntu22.04
2
+
3
+ # Build arg for GPU architectures - specify which CUDA compute capabilities to compile for
4
+ # Common values:
5
+ # 7.0 - Tesla V100
6
+ # 7.5 - RTX 2060, 2070, 2080, Titan RTX
7
+ # 8.0 - A100, A800 (Ampere data center)
8
+ # 8.6 - RTX 3060, 3070, 3080, 3090 (Ampere consumer)
9
+ # 8.9 - RTX 4070, 4080, 4090 (Ada Lovelace)
10
+ # 9.0 - H100, H800 (Hopper data center)
11
+ # 12.0 - RTX 5070, 5080, 5090 (Blackwell) - Note: sm_120 architecture
12
+ #
13
+ # Examples:
14
+ # RTX 3060: --build-arg CUDA_ARCHITECTURES="8.6"
15
+ # RTX 4090: --build-arg CUDA_ARCHITECTURES="8.9"
16
+ # Multiple: --build-arg CUDA_ARCHITECTURES="8.0;8.6;8.9"
17
+ #
18
+ # Note: Including 8.9 or 9.0 may cause compilation issues on some setups
19
+ # Default includes 8.0 and 8.6 for broad Ampere compatibility
20
+ ARG CUDA_ARCHITECTURES="8.0;8.6"
21
+
22
+ ENV DEBIAN_FRONTEND=noninteractive
23
+
24
+ # Install system dependencies
25
+ RUN apt update && \
26
+ apt install -y \
27
+ python3 python3-pip git wget curl cmake ninja-build \
28
+ libgl1 libglib2.0-0 ffmpeg && \
29
+ apt clean
30
+
31
+ WORKDIR /workspace
32
+
33
+ COPY requirements.txt .
34
+
35
+ # Upgrade pip first
36
+ RUN pip install --upgrade pip setuptools wheel
37
+
38
+ # First install torch with the versions we want, so that stuff in requirements.txt doesn't pull in the generic versions
39
+ # If you change CUDA 12.8 here, you also need to change the FROM docker image at the top
40
+ RUN pip install torch==2.10.0+cu128 torchvision==0.25.0+cu128 torchaudio==2.10.0+cu128 --index-url https://download.pytorch.org/whl/cu128
41
+
42
+ # Install requirements if exists
43
+ RUN pip install -r requirements.txt
44
+
45
+ # Install SageAttention from git (patch GPU detection)
46
+ ENV TORCH_CUDA_ARCH_LIST="${CUDA_ARCHITECTURES}"
47
+ ENV FORCE_CUDA="1"
48
+ ENV MAX_JOBS="8"
49
+
50
+ COPY <<EOF /tmp/patch_setup.py
51
+ import os
52
+ with open('setup.py', 'r') as f:
53
+ content = f.read()
54
+
55
+ # Get architectures from environment variable
56
+ arch_list = os.environ.get('TORCH_CUDA_ARCH_LIST')
57
+ arch_set = '{' + ', '.join([f'"{arch}"' for arch in arch_list.split(';')]) + '}'
58
+
59
+ # Replace the GPU detection section
60
+ old_section = '''compute_capabilities = set()
61
+ device_count = torch.cuda.device_count()
62
+ for i in range(device_count):
63
+ major, minor = torch.cuda.get_device_capability(i)
64
+ if major < 8:
65
+ warnings.warn(f"skipping GPU {i} with compute capability {major}.{minor}")
66
+ continue
67
+ compute_capabilities.add(f"{major}.{minor}")'''
68
+
69
+ new_section = 'compute_capabilities = ' + arch_set + '''
70
+ print(f"Manually set compute capabilities: {compute_capabilities}")'''
71
+
72
+ content = content.replace(old_section, new_section)
73
+
74
+ with open('setup.py', 'w') as f:
75
+ f.write(content)
76
+ EOF
77
+
78
+ RUN git clone https://github.com/thu-ml/SageAttention.git /tmp/sageattention && \
79
+ cd /tmp/sageattention && \
80
+ python3 /tmp/patch_setup.py && \
81
+ pip install --no-build-isolation .
82
+
83
+ RUN useradd -u 1000 -ms /bin/bash user
84
+
85
+ RUN chown -R user:user /workspace
86
+
87
+ RUN mkdir /home/user/.cache && \
88
+ chown -R user:user /home/user/.cache
89
+
90
+ COPY entrypoint.sh /workspace/entrypoint.sh
91
+
92
+ ENTRYPOINT ["/workspace/entrypoint.sh"]
LICENSE.txt ADDED
@@ -0,0 +1,455 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ WanGP Community License 2.0
2
+ (for the WanGP / Wan2GP project)
3
+
4
+ IMPORTANT: READ CAREFULLY. BY USING, COPYING, MODIFYING, OR
5
+ DISTRIBUTING THE SOFTWARE, YOU AGREE TO THIS LICENSE.
6
+
7
+ PLAIN-ENGLISH SUMMARY (NON-BINDING)
8
+
9
+ You may use WanGP for free, including inside a company.
10
+ You may modify WanGP for your own use.
11
+ You may use WanGP to create outputs and you may sell or license those
12
+ outputs.
13
+ You are only asked to credit WanGP when you directly sell or license an
14
+ output created with WanGP.
15
+ You may not sell WanGP itself, white-label it, embed it in a paid
16
+ product, or offer paid API / SaaS / hosted / OEM access to it without a
17
+ separate written reseller or commercial license.
18
+ Third-party open-source code, models, weights, datasets, and other
19
+ components keep their own licenses. This License applies only to the
20
+ Licensor's own copyrightable WanGP implementation and does not reduce
21
+ rights granted under separate third-party licenses.
22
+ For redistributed WanGP source or bundles, a reasonable compliance path
23
+ for bundled open-source code is sufficient, such as preserving in-tree
24
+ notices and including standard open-source license texts in a LICENSES
25
+ or similar directory. This License does not require a full inventory of
26
+ separately installed package-manager requirements or transitive
27
+ dependencies that are not packaged with WanGP.
28
+ In practical terms, the Software covers WanGP's added engineering and
29
+ productization layer as embodied in the Licensor's own implementation,
30
+ such as low-VRAM and VRAM-management work, speed improvements,
31
+ multi-model automation, Deepy agentic assistant workflows, UI and
32
+ queueing layers, API and headless interfaces, packaging, and
33
+ integration glue.
34
+ Illegal or seriously abusive uses are prohibited.
35
+
36
+ If this summary conflicts with the binding terms below, the binding
37
+ terms below control.
38
+
39
+ 1. DEFINITIONS
40
+
41
+ 1.1 "Licensor" means the copyright holder or holders for the Software.
42
+
43
+ 1.2 "Third-Party Materials" means any code, model, weight, dataset,
44
+ library, binary, asset, documentation, or other material that is
45
+ licensed under separate terms, including without limitation Apache-2.0,
46
+ MIT, BSD, other open-source licenses, model licenses, dataset licenses,
47
+ or third-party commercial terms, or any material for which the Licensor
48
+ does not have authority to apply this License.
49
+
50
+ 1.3 "Software" means the Licensor's copyrightable WanGP code and other
51
+ materials distributed by the Licensor as part of the official WanGP
52
+ repository, releases, installers, containers, binaries, APIs,
53
+ documentation, or other official WanGP distributions, including any
54
+ copies, modifications, and derivative works of that Licensor-owned
55
+ material, but excluding Third-Party Materials except to the extent the
56
+ Licensor may lawfully license its own separable contributions.
57
+
58
+ For context, the Software includes, to the extent embodied in the
59
+ Licensor's own distributed implementation, without limitation:
60
+ (a) memory or VRAM optimization layers, scheduling, packing,
61
+ quantization handling, model-loading logic, or low-resource
62
+ execution techniques;
63
+ (b) speed, throughput, caching, offloading, batching, or performance
64
+ optimizations;
65
+ (c) orchestration, automation, chaining, queueing, Deepy
66
+ agentic assistant workflows, or workflow logic across
67
+ multiple models, tools, or tasks;
68
+ (d) user interfaces, web interfaces, APIs, local services,
69
+ headless modes, automation surfaces, plugins, bridges,
70
+ wrappers, or machine interfaces;
71
+ (e) packaging, installers, configuration flows, operational
72
+ tooling, metadata handling, or workflow templates; and
73
+ (f) combinations of the foregoing.
74
+
75
+ If a file, package, binary, archive, model bundle, or other item
76
+ contains both Third-Party Materials and Licensor-owned material, this
77
+ License applies only to the Licensor's own copyrightable contributions
78
+ and does not reduce permissions granted by the licenses that govern the
79
+ Third-Party Materials.
80
+
81
+ The list above is illustrative only. For clarity, "Software" refers to
82
+ the Licensor's implementation. It does not, by itself, claim rights in
83
+ abstract ideas, methods, concepts, formats, protocols, interfaces, or
84
+ compatibility as such.
85
+
86
+ 1.4 "Output" means any image, video, audio, text, metadata, file, or
87
+ other content produced by running the Software.
88
+
89
+ 1.5 "Headless Usage" means invoking or automating the Software without
90
+ the standard interactive user interface, including CLI usage, queue
91
+ processing, automation scripts, batch jobs, local services, embedded
92
+ runtimes, programmatic calls, or similar non-interactive operation.
93
+
94
+ 1.6 "API" or "Integration" means any interface, wrapper, binding, SDK,
95
+ library call, local service, IPC bridge, plugin, network endpoint,
96
+ embedded surface, or other mechanism that exposes functionality of the
97
+ Software to another program, product, service, or user.
98
+
99
+ 1.7 "Direct Output Sale" means selling, licensing, or otherwise
100
+ charging separate consideration for an Output, or a collection of
101
+ Outputs, where the principal value of the transaction is the Output
102
+ itself. Direct Output Sale does not include:
103
+ (a) private or internal use;
104
+ (b) free publication or free sharing;
105
+ (c) agency, studio, or client work where the Software is used as an
106
+ internal production tool and the customer is paying primarily
107
+ for creative or production services; or
108
+ (d) use of an Output that is merely incidental to a broader product
109
+ or service, unless the Output itself is being sold, licensed,
110
+ or provided for separate consideration.
111
+
112
+ 1.8 "Free Use" means personal use, hobby use, research use,
113
+ educational use, evaluation, internal company use, internal business
114
+ use, studio use, agency use, client work, implementation work, support
115
+ work, private deployment, and free redistribution, provided that the
116
+ activity does not constitute Restricted Commercialization.
117
+
118
+ 1.9 "Restricted Commercialization" means any of the following:
119
+ (a) selling, sublicensing, renting, leasing, licensing for a fee,
120
+ or otherwise monetizing access to the Software;
121
+ (b) offering the Software through a paid, metered, sponsored,
122
+ ad-supported, subscription, hosted, managed, API, SaaS,
123
+ white-label, OEM, embedded, marketplace, platform, or similar
124
+ product or service;
125
+ (c) distributing the Software as part of a paid product or service
126
+ where the Software is a material feature or a material part of
127
+ the value offered;
128
+ (d) enabling third parties to invoke, access, automate, or consume
129
+ the Software through Headless Usage, an API, an Integration, a
130
+ local service, a plugin, an IPC bridge, a remote endpoint, or
131
+ similar mechanism in exchange for consideration; or
132
+ (e) charging a separate fee for unlocking, activating, hosting,
133
+ enabling, bundling, or providing access to the Software.
134
+
135
+ For clarity, Restricted Commercialization does NOT include:
136
+ (i) internal company use or internal business use;
137
+ (ii) using the Software to create Outputs for yourself or for
138
+ clients;
139
+ (iii) charging reasonable service fees for installation,
140
+ customization, consulting, support, training, or integration
141
+ labor, provided that no separate fee is charged for access to
142
+ the Software itself, and the recipient receives the Software
143
+ subject to this License and any applicable third-party terms;
144
+ or
145
+ (iv) Direct Output Sales governed by Section 6.
146
+
147
+ 2. SCOPE AND THIRD-PARTY RIGHTS
148
+
149
+ 2.1 This License applies only to the Software.
150
+
151
+ 2.2 Third-Party Materials remain licensed under their own terms.
152
+
153
+ 2.3 Nothing in this License limits, replaces, overrides, or narrows any
154
+ rights you receive directly under the licenses or terms applicable to
155
+ Third-Party Materials.
156
+
157
+ 2.4 Where an act is permitted by an applicable third-party license for
158
+ Third-Party Materials, that permission remains effective for those
159
+ Third-Party Materials notwithstanding anything else in this License.
160
+
161
+ 2.5 If a file, directory, package, model card, weight package, header,
162
+ notice, repository note, or other material states a different license
163
+ or separate terms for a particular item, those terms control for that
164
+ item.
165
+
166
+ 2.6 In case of ambiguity, this License must be interpreted narrowly so
167
+ that it does not over-claim rights in Third-Party Materials or reduce
168
+ permissions granted by their own licenses.
169
+
170
+ 2.7 No file-level marking is required for the Licensor's official WanGP
171
+ distributions. However, no one may use this License to claim that
172
+ Third-Party Materials are subject to restrictions that their own
173
+ licenses do not impose.
174
+
175
+ 3. FREE USE LICENSE GRANT
176
+
177
+ 3.1 Subject to this License, the Licensor grants you a worldwide,
178
+ non-exclusive, royalty-free license to use, run, reproduce, display,
179
+ perform, modify, and create derivative works of the Software for
180
+ Free Use.
181
+
182
+ 3.2 You may privately deploy and use the Software in interactive mode,
183
+ batch mode, queue mode, Headless Usage, local-service mode, scripted
184
+ mode, automated mode, or API mode for Free Use.
185
+
186
+ 3.3 No separate reseller or commercial license is required solely
187
+ because the user is a company, studio, agency, or revenue-generating
188
+ business, so long as the Software is not sold, monetized, or made
189
+ available to third parties in a manner described in Section 5.
190
+
191
+ 4. REDISTRIBUTION FOR FREE USE
192
+
193
+ 4.1 You may copy and redistribute the Software, with or without
194
+ modification, only for Free Use and only if you:
195
+ (a) provide a copy of this License with the redistribution;
196
+ (b) preserve copyright notices, attribution notices, license
197
+ notices, patent notices, disclaimer notices, and other
198
+ required notices;
199
+ (c) preserve and comply with any notices and license texts required
200
+ for Third-Party Materials that are packaged with that
201
+ redistribution;
202
+ (d) clearly state that you modified the Software, if you distribute
203
+ modified versions, and identify the date of your changes in a
204
+ reasonable manner;
205
+ (e) do not imply sponsorship, endorsement, or affiliation by the
206
+ Licensor unless the Licensor has given prior written
207
+ permission; and
208
+ (f) do not impose terms on recipients that purport to restrict
209
+ rights those recipients receive directly under applicable
210
+ licenses for Third-Party Materials.
211
+
212
+ 4.2 Each recipient of the Software automatically receives a license
213
+ from the Licensor to use the Software under this License.
214
+
215
+ 4.3 Redistribution that constitutes Restricted Commercialization is not
216
+ permitted under this License and requires a separate written reseller
217
+ or commercial license from the Licensor.
218
+
219
+ 4.4 If you redistribute mixed materials containing both Software and
220
+ Third-Party Materials, you must preserve a reasonable path for
221
+ recipients to identify the separate third-party notices and terms for
222
+ the Third-Party Materials packaged with that redistribution, such as a
223
+ LICENSES directory, THIRD_PARTY_NOTICES file, documentation page, or
224
+ equivalent release documentation. Unless a bundled Third-Party Material
225
+ requires something more specific under its own terms, a small directory
226
+ containing the standard open-source license texts used by the bundled
227
+ open-source components is sufficient.
228
+
229
+ 5. RESTRICTED COMMERCIALIZATION
230
+
231
+ 5.1 You may not engage in Restricted Commercialization without a
232
+ separate written reseller or commercial license from the Licensor.
233
+
234
+ 5.2 By way of example, the following require a separate written
235
+ reseller or commercial license:
236
+ (a) selling WanGP itself;
237
+ (b) offering paid API, paid SaaS, paid hosted, or paid managed
238
+ access to WanGP;
239
+ (c) white-labeling WanGP;
240
+ (d) embedding the Software in a paid product or paid service;
241
+ (e) offering OEM or reseller packages built around the Software;
242
+ (f) charging separately for access to the Software, even if the
243
+ Software is bundled with other tools or services; or
244
+ (g) exposing the Software to customers or end users as a paid
245
+ feature, add-on, or monetized backend component.
246
+
247
+ 5.3 The following do not require a separate written reseller or
248
+ commercial license, provided you comply with this License:
249
+ (a) internal company use, internal business use, and team use;
250
+ (b) client work, studio work, agency work, or production work where
251
+ the Software is used as an internal production tool;
252
+ (c) charging for labor to install, customize, support, train on, or
253
+ integrate the Software, so long as you do not charge a
254
+ separate fee for access to the Software itself, and you do not
255
+ reduce the recipient's rights under this License or applicable
256
+ third-party licenses;
257
+ (d) distributing free, non-monetized builds, wrappers, or
258
+ integrations that comply with this License; and
259
+ (e) using the Software to create Outputs and exploiting those
260
+ Outputs as allowed by Section 6.
261
+
262
+ 5.4 The Licensor may offer separate reseller, OEM, embedded, hosting,
263
+ API, or commercial licenses on different terms.
264
+ Commercial licensing contact: deepbeepmeep@yahoo.com or contact deepbeepmeep on discord server (check GitHub for link).
265
+
266
+ 6. OUTPUTS AND CREDIT
267
+
268
+ 6.1 Subject to applicable law, Section 10, and any separate terms that
269
+ govern Third-Party Materials, you may use, reproduce, publish,
270
+ perform, display, distribute, sell, and license Outputs for any lawful
271
+ purpose.
272
+
273
+ 6.2 You must provide reasonable credit to WanGP only when you engage in
274
+ a Direct Output Sale.
275
+
276
+ 6.3 Reasonable credit under Section 6.2 may be satisfied by wording
277
+ such as:
278
+ "Made with WanGP"
279
+ "Created using WanGP"
280
+ or wording of similar meaning.
281
+
282
+ 6.4 Reasonable credit should be placed where credits are customarily
283
+ shown, such as a credits page, product page, invoice, project
284
+ documentation, metadata field, About page, or similar location. If
285
+ technically practical, the credit should include a link to the official
286
+ WanGP project page or repository. If linking is not technically
287
+ practical, plain-text credit is sufficient.
288
+
289
+ 6.5 No credit is required under this License for:
290
+ (a) private or internal use;
291
+ (b) free sharing or free publication of Outputs;
292
+ (c) agency, studio, or client work where the Software is used as an
293
+ internal production tool and the Output is not itself being
294
+ directly sold or licensed as a separate deliverable; or
295
+ (d) use of Outputs that is not a Direct Output Sale.
296
+
297
+ 6.6 This License governs the Software, not ownership of Outputs as
298
+ such. You are responsible for complying with any separate terms that
299
+ may apply to Third-Party Materials used with or through the Software.
300
+
301
+ 7. INTEGRATIONS, APIS, AND HEADLESS USE
302
+
303
+ 7.1 Private use of the Software through an API, local service,
304
+ automation, batch processing, queue processing, script, plugin, or
305
+ Headless Usage is permitted for Free Use.
306
+
307
+ 7.2 Exposing the Software to third parties through an API, local
308
+ service, hosted endpoint, plugin, Integration, wrapper, bridge, or
309
+ Headless Usage in exchange for consideration is Restricted
310
+ Commercialization and requires a separate written reseller or
311
+ commercial license.
312
+
313
+ 7.3 Free, non-monetized integrations and wrappers are permitted so long
314
+ as they:
315
+ (a) comply with this License;
316
+ (b) clearly disclose WanGP use in reasonable documentation or an
317
+ About section; and
318
+ (c) preserve all notices required for the Software and for
319
+ Third-Party Materials.
320
+
321
+ 7.4 A project is not subject to this License merely because it can
322
+ interoperate with WanGP. This License applies only to the Software that
323
+ is included, copied, embedded, or distributed by or through that
324
+ project.
325
+
326
+ 8. TRADEMARKS, NAMES, AND NO ENDORSEMENT
327
+
328
+ 8.1 This License does not grant any right to use the Licensor's trade
329
+ names, trademarks, service marks, logos, or branding, except:
330
+ (a) for reasonable nominative use to describe origin or
331
+ compatibility; and
332
+ (b) as necessary to give the credit required by Section 6.
333
+
334
+ 8.2 You may not state or imply that the Licensor endorses you, your
335
+ organization, your products, your services, or your modifications,
336
+ except with the Licensor's prior written permission.
337
+
338
+ 9. THIRD-PARTY NOTICE COMPLIANCE
339
+
340
+ 9.1 If you distribute a package, repository, archive, installer,
341
+ container, binary, or build that includes Third-Party Materials, you
342
+ must preserve and include the license texts, notices, attribution
343
+ statements, and patent notices, if any, required by the terms governing
344
+ the Third-Party Materials that are packaged with that distribution.
345
+
346
+ 9.2 For clarity, this License does not require you to enumerate or
347
+ bundle the full license tree for separately installed dependencies,
348
+ package-manager requirements, or transitive libraries that are not
349
+ themselves redistributed as part of your WanGP package, archive,
350
+ repository, installer, or build.
351
+
352
+ 9.3 A THIRD_PARTY_NOTICES file, LICENSES directory, About-page
353
+ references, or equivalent documentation is strongly recommended for
354
+ mixed-license distributions, and a compact directory of the standard
355
+ open-source license texts used by bundled open-source components is a
356
+ reasonable default approach, but the absence of such a file does not
357
+ convert Third-Party Materials into Software under this License.
358
+
359
+ 9.4 Nothing in this License authorizes you to remove, obscure, or
360
+ replace any required attribution, license, or notice for Third-Party
361
+ Materials.
362
+
363
+ 10. ACCEPTABLE USE
364
+
365
+ 10.1 To the extent permitted by law, you may not use the Software or
366
+ knowingly use Outputs generated with the Software:
367
+ (a) in violation of applicable law;
368
+ (b) to produce, facilitate, or materially support child sexual
369
+ abuse material, sexual exploitation of minors, non-consensual
370
+ intimate imagery, human trafficking, terrorism, unlawful
371
+ harassment or stalking, fraud or impersonation, malware,
372
+ unauthorized cyber abuse, or other comparable serious abuse;
373
+ (c) to promote or facilitate extremist violence, genocide, crimes
374
+ against humanity, or other serious human rights abuses;
375
+ (d) in a manner primarily intended to materially infringe the
376
+ privacy, publicity, intellectual property, or other legal
377
+ rights of others; or
378
+ (e) in a manner primarily intended to cause reasonably foreseeable
379
+ significant harm to identified individuals or groups through
380
+ deception, coercion, or abuse.
381
+
382
+ 10.2 If you expose the Software or Outputs to third-party users through
383
+ Headless Usage, an API, an Integration, or a hosted deployment, you
384
+ must communicate restrictions consistent with this Section and take
385
+ reasonable measures to deter and respond to clearly prohibited uses.
386
+
387
+ 10.3 This Section is intended to address clearly abusive uses and must
388
+ be interpreted reasonably. It does not prohibit good-faith criticism,
389
+ journalism, research, documentation, security work, content-moderation
390
+ work, or artistic works merely because they depict, discuss, or analyze
391
+ sensitive topics without facilitating the prohibited harm.
392
+
393
+ 10.4 Material breach of this Section is a material breach of this
394
+ License and may result in termination under Section 12.
395
+
396
+ 11. RESERVATION OF RIGHTS
397
+
398
+ 11.1 Except for the rights expressly granted in this License, the
399
+ Licensor reserves all rights in the Software.
400
+
401
+ 11.2 No implied license is granted by this License.
402
+
403
+ 12. WARRANTY DISCLAIMER, LIMITATION OF LIABILITY, AND TERMINATION
404
+
405
+ 12.1 THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
406
+ EXPRESS OR IMPLIED, INCLUDING WITHOUT LIMITATION WARRANTIES OF
407
+ MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, TITLE, AND
408
+ NON-INFRINGEMENT.
409
+
410
+ 12.2 TO THE MAXIMUM EXTENT PERMITTED BY LAW, THE LICENSOR SHALL NOT BE
411
+ LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER LIABILITY, WHETHER IN AN ACTION
412
+ OF CONTRACT, TORT, OR OTHERWISE, ARISING FROM, OUT OF, OR IN CONNECTION
413
+ WITH THE SOFTWARE OR THIS LICENSE OR THE USE OF OR OTHER DEALINGS IN
414
+ THE SOFTWARE.
415
+
416
+ 12.3 This License remains in effect until terminated.
417
+
418
+ 12.4 If you materially breach this License, your rights under this
419
+ License terminate automatically if you fail to cure the breach within
420
+ thirty (30) days after receiving written notice from the Licensor that
421
+ reasonably describes the breach.
422
+
423
+ 12.5 If you cure the breach within that thirty (30) day period, your
424
+ rights under this License are automatically reinstated as of the date
425
+ of cure.
426
+
427
+ 12.6 Termination of rights under this License does not terminate,
428
+ revoke, or reduce any rights you may have independently under licenses
429
+ applicable to Third-Party Materials.
430
+
431
+ 12.7 Outputs created before termination remain governed by Section 6,
432
+ subject to Section 10 and applicable law.
433
+
434
+ 12.8 Sections 2, 6, 8, 9, 10, 11, 12.1, 12.2, 12.6, 12.7, and 13
435
+ survive termination.
436
+
437
+ 13. GENERAL
438
+
439
+ 13.1 If any provision of this License is unenforceable, that provision
440
+ shall be enforced to the maximum extent permitted by law, and the
441
+ remainder of the License shall remain in effect.
442
+
443
+ 13.2 Failure by the Licensor to enforce any provision of this License
444
+ is not a waiver of future enforcement of that or any other provision.
445
+
446
+ 13.3 You may always continue to exercise any rights you receive
447
+ directly under licenses that apply to Third-Party Materials, regardless
448
+ of this License.
449
+
450
+ 13.4 The Licensor may publish updated versions of this License.
451
+ Unless the Licensor expressly states otherwise for a particular release
452
+ or material, you may continue using the version of this License that
453
+ applied when you received that Software.
454
+
455
+ END OF TERMS
README.md CHANGED
@@ -1,13 +1,527 @@
1
  ---
2
  title: ColabWan
3
- emoji: 📊
4
- colorFrom: green
5
- colorTo: pink
6
  sdk: gradio
7
- sdk_version: 6.15.2
8
- python_version: '3.13'
9
- app_file: app.py
10
- pinned: false
11
  ---
 
12
 
13
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
  title: ColabWan
3
+ app_file: wgp.py
 
 
4
  sdk: gradio
5
+ sdk_version: 5.29.0
 
 
 
6
  ---
7
+ # WanGP
8
 
9
+ -----
10
+ <p align="center">
11
+ <b>WanGP by DeepBeepMeep : The best Open Source Generative Models Accessible to the GPU Poor</b>
12
+ </p>
13
+
14
+ WanGP is a one-stop super app for the best open source generative models across video, image, audio, and text-to-speech.
15
+
16
+ ## Highlights
17
+
18
+ | Modality | Supported models |
19
+ | --- | --- |
20
+ | **Video** | **Wan 2.1/2.2** and derived models, **LTX-2**, **Hunyuan Video 1/1.5**, **LongCat**, **Kandinsky**, **LTXV**, **MagiHuman** |
21
+ | **Image** | **Qwen Image**, **Z-Image**, **Flux 1/2** (Klein, Chroma), **HiDream** |
22
+ | **Audio / TTS** | **Qwen3 TTS**, **Ace Step 1/2/XL**, **Omnivoice**, **Index TTS2**, **KugelAudio**, **HearMula**, **Chatterbox** |
23
+
24
+ ### Run More Models on More Hardware
25
+
26
+ - **Low VRAM requirements**: run select models with as little as **6 GB of VRAM**.
27
+ - **Older Nvidia GPU support**: use RTX 10XX, 20XX, and newer cards.
28
+ - **AMD GPU support**: run on RDNA 4, 3, 3.5, and 2 hardware; see the Installation section below.
29
+ - **Fast latest-GPU performance**: take advantage of modern GPU acceleration.
30
+ - **Full web interface**: generate, manage, and reuse outputs from an easy browser UI.
31
+ - **LoRA customization**: adapt each model with LoRAs, reuse LoRAs stored in another App.
32
+ - **Many quantized checkpoint formats**: use int8, fp8, gguf, NV FP4, and Nunchaku.
33
+ - **Architecture-aware downloads**: automatically fetch the model files suited to your hardware.
34
+ - **Finetunes**: add your own finetunes / checkpoints or the ones you found on Hugging Face or CivitAI
35
+ - **Generation queue**: line up videos, images, and audio jobs, then come back later.
36
+ - **Headless mode**: launch batches from the command line for images, videos, and audio.
37
+ - **WanGP API**: add generative capabilities to your own apps.
38
+
39
+ ### Built-In Creation Tools
40
+
41
+ - **Video, image, and audio galleries**: browse generations and reuse them as new inputs.
42
+ - **Reusable settings**: extract settings from any generation, create templates, and share them.
43
+ - **Per-model prompt enhancer**: improve prompts with model-specific syntax and expectations.
44
+ - **Input preparation tools**: use the mask editor, background remover, pose/depth/flow extractors, speaker diarization, and background noise/song remover.
45
+ - **Deepy low-VRAM offline agent**: orchestrate generation jobs and tedious tasks such as transcription, video splitting, and color-frame generation while you are away.
46
+ - **Temporal and spatial upsampling**: improve outputs with RIFE, FlashVSR, and Lanczos.
47
+ - **Audio postprocessing**: generate soundtracks with MMAudio, replace voices with SeedVC, or remux a video with any soundtrack.
48
+ - **Ready-to-use plug-ins**: Gallery Browser, Motion Designer, Models/Checkpoints Manager, CivitAI browser and downloader, and more.
49
+
50
+ **Discord Server to get Help from the WanGP Community and show your Best Gens:** https://discord.gg/g7efUW9jGV
51
+
52
+ **Follow DeepBeepMeep on Twitter/X to get the Latest News**: https://x.com/deepbeepmeep
53
+
54
+ ## 📋 Table of Contents
55
+
56
+ - [🚀 Quick Start](#-quick-start)
57
+ - [📦 Installation](#-installation)
58
+ - [🎯 Usage](#-usage)
59
+ - [📚 Documentation](#-documentation)
60
+ - [🔗 Related Projects](#-related-projects)
61
+
62
+
63
+ ## 🔥 Latest Updates :
64
+ ### 1sth of June 2026: WanGP v11.90, Everything will be fine...
65
+ **Finetune Creator / Editor**
66
+ *Create* a new *Finetune* (use an existing model with your own checkpoints), *Edit* or *Import* an existing Finetune in only one click directly from the *WanGP UI*. You can then share easily a finetune with other users by clicking the *Export* button.
67
+
68
+ Look for the new **+** in the *WanGP Tool Bar*.
69
+
70
+ The finetune creator allows you not only to customize an existing models with *Custom URLs* or *Local Paths* for both the main *Transformer files* & *Text encoders* but also to define *User help* and set *Custom System Templates* to be used with the finetune *Prompt Enhancer*.
71
+
72
+ Please check *docs/FINETUNE.md* doc for info about finetunes.
73
+
74
+ ### 29th of May 2026: WanGP v11.88, Humans Accelerators
75
+ - **Create Hierarchies of Loras / Change Order of Loras**
76
+
77
+ - **WanGP Toolbar** with keyboard shortcuts:
78
+ - **Search**: switch quickly to another model by just entering a few letters of its name
79
+ - **Refresh Model List**: no longer needed to restart the app to add or modify a finetune
80
+ - **Unload All**: free most of the RAM/VRAM used by WanGP
81
+
82
+ - **MOV/MKV Container Support**: beside *mp4* files you can now store you video gens in *mov* and *mkv* containers
83
+
84
+ - **ProRes422 & DNxHR HQ Video Codecs**: these professional video codecs have some fans out there
85
+
86
+ - **LTX-2 Guide**: click the "i" to the right of the model description to get tips / explanations on how to use LTX2 models
87
+
88
+ - **LTX2 Smearing Fix**: the smearing / ghosting is now mostly gone
89
+
90
+ - **Omnivoice Fix**: you will enjoy this fix unless you liked the gibberish generator of the previous version
91
+
92
+
93
+ ### 21st of May 2026: WanGP v11.77, I can hear Voices
94
+ It has never been easier to do voice cloning directly in video models:
95
+
96
+ - **Voice Cloning with any Video Model**: you generated a great *LTX2/Ovi/Multitalk/...* and are sad the model didnt support natively *Voice Cloning*? Just use the new *SeedVC Audio Postprocessing* to replace up to two voices of your choice, it works magically with any video model ! You will find this feature in the *Audio* advanced tab or as *Late Posprocessing for Audio or Video*. WanGP exclusive *Two Voices* feature will detect who is talking and will make seamlessly the voices replacements at the right audio locations.
97
+
98
+ *New WanGP v11.75*: Voice cloning preserves background noise / music & supports singing. You can also enable *SeedVC v2* in the *Config / Extensions* tab for a higher quality voice cloning (alas no singing support with v2).
99
+
100
+ - **DramaBox**: like *ScenemeAI* that *DramaBox* uses LTX2.3 world knowledge to generate lively audio outputs. DramaBox is even more expressive (but also slower) than ScenemeAI. Of course as usual you get an exclusive Dialogue mode available out of the box.
101
+
102
+ - **LTX2.3 Id Lora Distilled**: Nice surprise ! it seems *Id Lora* worked from day 1 with *LTX2.3 Distilled*. It is now unlocked, you can now generate your own LTX2 video with voice cloning.
103
+
104
+ - **LTX2.3 EditAnything Reference**: you can at last inject one reference image in a LTX2 Video. You will need to use the dedicated finetunes *dev* and *distilled* finetunes I have prepared. Please note this feature is experimental.
105
+
106
+ - **LTX2 OmniNFT Lora Preset for better audio/video sync**: I have added this *LTX2 OmniNFT Lora* in a *Preset* so that it can be applied quickly. According to the authors of this Lora Audio/Video sync should be greatly improved.
107
+
108
+ - **LTX2 Dev reborn in Dev-Distilled**: WanGP LTX2 Dev implementation was based on LTX2 official implementation. I hadn't noticed that ComfyUI version of Dev was now completely different as it was mixing the *Distilled Lora* with Dev in both phases ( not just in phase 2). This makes dev faster and reduces the color saturation specific to Dev. So I have added a few *Dev Distilled Accelerator Profiles* you can pick from the *Settings List*. And now since Dev & Distilled are closer than ever, I have unlocked all the *Control Video* processes for Dev.
109
+
110
+ - **LTX2 Prompt Relay**: you can now target specific time range for a part of the prompt, for instance *[25%:50%]the man says "hello". Check the new *Prompt Online Help* marked with "i" for more info.
111
+
112
+ - **LongCat 1.5 Avatar**: with this new *Talking Head* model you are going to become at last a fan of *LongCat*. It is fast (8 steps distilled) and delivers high quality potentially unlimited gens using *Sliding Windows*.
113
+
114
+ - **Settings can now store Audio/Video/Images**: you can ask WanGP to store (in option) all the media you use frequently in a WanGP *Settings file*. This is very convenient for instance if you always use the same *Voice sample* or *Reference Images*. Even better, you can use these settings with *Deepy* of the *Full Video Process* plugin
115
+
116
+ - **Extensions Enabled by Default**: most extensions (upsampling, mmaudio, prompt enhancer, ...) are now enabled by Default so that they are easier to be found. Don't worry their corresponding checkpoints will be downloaded only if you actually use these extensions
117
+
118
+ - **FlashVSR Spatial Upsampling for Images**: this excellent spatial upsampler has been optimized for images and is now can be used as a *Post Processing* option or on existing images (thanks to the new *Late Post Processing* added on Images!)
119
+
120
+ - **FlashVSR Two Pass**: banding artifacts may appear when FlashVSR is used at very high res. The Two Pass mode which is twice as slow may reduce the banding.
121
+
122
+ - **HiDreamO1**: new 2604 finetune that should reduce the annoying blocking effect of this model. I have also regenerated all the quanto int8 files (they are now 20% larger, price to pay for quality) to reduce even further the blocking. Keep in mind that this model likes res >= 1080p
123
+
124
+ Also various fixes (Omnivoice, IndexTTS, Chatterbox, ...)
125
+
126
+ *Update 11.75*: Voice cloning with background and voice supports, FlashVSR for Images, Dev Distilled\
127
+ *Update 11.77*: LTX2 Prompt Relay, LongCat Avatar
128
+
129
+ ### 12th of May 2026: WanGP v11.66, Can you keep up?
130
+
131
+ - **HiDreamO1**: New Image Image model with editing capabilities is quite good to preserve identify and write text. WanGP version requires very Low VRAM and supports out of the Box *Control Image* & *Preview*.
132
+
133
+ - **Omnivoice**: This *Text To Speech model* (TTS) is fast and supports 100 languages with voice cloning. WanGP offers as a bonus an experimental dialogue mode (not the best one since it is hard to predict when Omnivoice has finished generating)
134
+
135
+ - **ScenemeAI**: A LTX2.3 derived *TTS* that leverages *LTX-2* world knowledge : it can produce lifelike audio generations since you can drive the audio generation by describing what a speaker is doing / saying. I have implemented on top a dialogue mode between any number of speakers (first two speakers support voice cloning) with very smooth transitions between speakers especially when generating English. You will find ScenemeAI among the *TTS models* but be aware it will use by default a *Video Memory Profile* since it uses LTX2 engine behind the scene. Don't hesitate to use WanGP *Prompt Enhancer* to generate lively dialogues.
136
+
137
+ - **MPS / Apple Early Support**: Mac users are about to discover the world of WanGP albeit for start it wont be fast nor very optimized and not all models will be supported. Many thanks to *huangyebiaoke* (for the port), *cn0ss* & *SquishedSquirrel* (for the testing). Don't hesitate to report in the new *MPS* Discord channel your feedback if you are a mac user.
138
+
139
+ ### 9th of May 2026: WanGP v11.61, The Last Mile
140
+
141
+ With a slight (half year) delay WanGP supports now officially *FlashVSR* a very high quality *Spatial Upsampler* which can upsample up to 4x you videos. As FlashVSR has been almost entirely rewritten for WanGP, it can be branded as the *Ultimate Upsampler for the GPU Poor*, check these figures:
142
+ - x2 Spatial Upsampling will need to work only 6GB of VRAM
143
+ - x4 Spatial Upsampling will require only 10GB of VRAM (see the 5k example below)
144
+
145
+ The VRAM requirements above are independent of the Video Length (still the longer the video the more RAM)
146
+
147
+ You first need to install *Triton* and optionally *SpargeAttention* for best quality (please check the INSTALLATION.md for download links) and enable *FlashVSR* in the *Configuration > Extensions* Tab.
148
+
149
+ FlashVSR is available in the following contexts:
150
+ - a Postprocessing option in *Advanced Tab > Postprocessing*
151
+ - a *Late Postprocessing* that can be applied on already generated videos
152
+ - in Model *WanGP System Postprocessing* of the *Process Full Video* Plugin you can Upsample a few hours long Video !
153
+
154
+ Please note as FlashVSR is now natively supported by WanGP and highly optimized, you may no longer need the *FlashVSR Plugin* developed by @h4k4z3. In any case many thanks to @h4k4z3 for developing this plugin which was very useful.
155
+
156
+ ### 2nd of May 2026: WanGP v11.52, a Kind of Magic
157
+
158
+ - **Vista 4D**: Vista4D allows a *Video Reshooting* of a *Dynamic scene* from novel camera trajectories and viewpoints. In other words this Wan 2.1 model will let you relive from a different (moving) perspective a scene with moving people or objects. The sequences are quite short (usually 49 frames, max around 97 frames) but it is a lot of fun as for once this really works.
159
+
160
+ In real life, there is no chance you should have been able to run this model (it requires x3 the amount of VRAM than what is usally required for equivalent output res and the preprocessing needs 24 GB of VRAM to build a 4D map). But once again thanks to WanGP magic VRAM requirements have been reduced to 10 GB of VRAM or less.
161
+
162
+ It is highly recommended to apply the *Lightx2v 4 steps* lora profile. Also for best efficiency, you must list all the dynamic objects / people in the *Dynamic object keywords* input.
163
+
164
+ - **Magic Mask**: generating a *Video Mask* or *Image Mask* has never been easier and faster. No need to get into the *Video Mask Generator* tab, just click the *Magic Wand* next to *Mask field* and enter a few keywords like *blue car* or *lady to the right* and a high quality mask powered by *SAM3* will be generated automatically. You will appreciate the very good *Temporal Consistency* brought by SAM3.
165
+
166
+ - **Video Mask Generator with SAM3 support**: if you still need to generate complex masks you can combine the good old point and click masks with the SAM3 / Magic Mask masks. You need to enable this feature in the *Config / Extensions* tab.
167
+
168
+ - **LTX-2 Video to Audio**: it was more or less already possible but this new Control Video Process will be much faster and the output video will be unaltered
169
+
170
+ *update 11.51*: various fixes\
171
+ *update 11.52*: LTX-Video to Audio, fixed bugs in audio continuation with sliding windows
172
+
173
+ ### 25th of April 2026: WanGP v11.41, LTX-2 Mega Mix Part 2
174
+ More nice goodies for **LTX-2**:
175
+ - **HDR Control Video support**: you can now provide an HDR Control Video it will be automatically converted to SDR if model doesnt support HDR
176
+
177
+ - **LTX 2.3 SDR to HDR**: thanks to a new HDR Ic lora, you can now convert SDR Videos to HDR using LTX 2.3. This feature is available as a new *Control Video process* and also in the *Process Full Video* plugin. Please note that the embedded Gradio Gallery video player converts automatically any HDR content to SDR, so if you want to enjoy the full HDR content you will need an external media player (for instance *MPC-BE*)
178
+
179
+ - **LTX 2.3 Control Video Injection in Phase 2**: up to now even if you picked 2 phases, the *Control Video* was only injected in Phase 1 (Phase 2 was only used for upsampling). Now if you have chosen for at least one Ic Lora, a non null mutiplier for phase 2, the control video will be injected also for phase 2. This will increase output quality with 2 phases but will require more VRAM for phase 2.
180
+
181
+ - **Process Full Video Custom Settings**: you can now reuse your own presaved settings in the plugin. As you will link the plugin to your settings any change to the saved settings will be immediatly available in the plugin. If you find some great combination of loras / model / settings to be used with this Plugin please share them on the discord server so that I can add them in the official list.
182
+
183
+ *update 11.41*: added Process Full Video Custom Settings
184
+
185
+ ### 21st of April 2026: WanGP v11.35, LTX-2 Mega Mix
186
+ Lots of nice goodies for **LTX-2**:
187
+
188
+ - **LTX-2.3 Distilled 1.1**: new version of the *Distilled model* released by *LTX team*, it should offer better audio and visuals. You will find also a Dev 1.1 version which uses Distilled 1.1 for Phase 2.
189
+
190
+ - **VBVR Lora Preset**: This LoRA enhances the base LTX-2 for Enhanced Complex Prompt Understanding, Improved Motion Dynamics & Temporal Consistency. You can select it in the *Settings list* at the top.
191
+
192
+ - **Phase 1/2 Choice**: you can now either you go for a good old *2 Phases Gen* (1st Phase Low Res, 2nd shorter Phase High res) or go straight to a single High Res Phase (needs more VRAM and slower, but potentially higher quality). Please note that Outpainting mode and Pose/Edge/Depth extractors are always using 1 phase.
193
+
194
+ - **Improved Sliding Window**: transition between windows should be less noticable, *Sliding Windows overlapped Frames* carry now also the audio of the overlapped frames, so the higher the number of overlapped frames the higher the chance that the sound / voice used in the previous window will be used in the new one.
195
+
196
+ - **Video Length not Limited by Audio**: if you provide an Audio input, WanGP will no longer stops when the audio is consumed. It will continue the Video/Audio Gen based on the content of your Text prompt, and guess what ? it may reuse the same voice/sound used up to now ! This is an option, you need to check the checkbox *Video Length not Limited by Audio*.
197
+
198
+ - **Silent Movie Mode**: if for some reason you want video with not only no sound but that takes into account that there is no sound (you dont want people to open their mouth for instance), just now leave the *Control Audio* empty
199
+
200
+ ~~ - LTX2/2.3 Loras Split: as LTX2.0 Loras work badly with LTX2-3 and were getting on the way, now each version of LTX2 has its own lora folder. Loras will be moved automatically at startup using a lora migration script. I invit you to verify that the loras landed in the right folder.~~
201
+
202
+ - **System Loras Multipliers Overrides**: WanGP adds automatically and transparently loras (that is they are loaded although they are not visible) if needed by a feature (distilled lora, id lora, outpaint lora, union control lora). You can now override the default multipliers used by WanGP by selecting the target lora in the *Activated Loras* input and by specifiying the corresponding *Loras Multipliers*.
203
+
204
+ - **Transfer Human Motion With Pose Alignment**: you are trying to transfer a human motion from a control video, but you use a start image with a person who has a different body shape (larger, taller, ...) and stands in a different location in the frame. This is not going to work well as you start image wil end up distorted. This is a past issue, as now the control video pose can be aligned with the start image if you pick Transfer *Human Motion With Pose Alignment*. This feature is also supported by *Wan Vace*, start image must be the *Background ref image*.
205
+
206
+ - **Injected Frames & Sliding Windows**: injected frames were not properly injected starting from window no 2. This is now supported.
207
+
208
+ - **Process Process Full Video Plugin**: this *bundled PlugIn* which needs to be enabled first in *the PlugIn tab*, right now supports only *Outpainting*. It relies on *LTX2 Lora outpainting*. It is more or less a *Super Sliding Windows* mode but without the *RAM restrictions* and no risk to explode the *Video Gallery* with huge files. If you are patient enough you can change the Aspect Ratio of a few hours movie (check out below the 1 min sample). Behold how *Sliding Windows transitions* are almost invisible !
209
+
210
+ - **NEW Processes for Full Video Plugin**: *Refocus* (remove blur), *Ungrade* (remove stylized color grading) and *Uncompress* (remove compression artifacts) have been added. Many thanks to *Oumoumad Mohamed* who created the Ic Loras (including the *Outpainting* lora ) that power these processes. If you have found some Ic Loras that are useful and dont cause glitches with Sliding Windows, let me know and I will add them.
211
+
212
+ - **WanGP API Video Gen**: *Plugin Developers* can now *Queue a Gen* directly from a plugin. This opens the possibility of plugins that place various gen orders and then combine the results (hint: we could have our very own version of *LTX-Destop* inside WanGP).
213
+
214
+ - **New One Click Install / Update Scripts**: We have to thank **Tophness / @steve_Jabz** for that one. *Huge Kudos to him!* The scripts will not only install WanGP but also all the *Kernels* (among *Triton, Sage, Flash, GGuf, Lightx2v, Nunchaku*) supported by your GPU. Please have a look at the instructions further down. Dont't hesitate to share feedback or report any issue.
215
+
216
+ *update 11.31*: fixed phase 1 forced incorrectly in some cases\
217
+ *update 11.32*: bugs fixes, Process Full Video now supports Distilled 1.1 & accepts video without audio\
218
+ *update 11.33*: Separated LTX2 & LTX2.3 loras in different folders, added easy loras multipliers override\
219
+ *update 11.34*: Reverted split as not popular\
220
+ *update 11.35*: added Aligned Pose Transfer, Injected Frames & Sliding Windows support, new processes for Process Full Video Plugin
221
+
222
+ ### 11th of April 2026: WanGP v11.26, Now I Can See
223
+
224
+ - **LTX-2 Ic Lora Rebooted**: *Ic Loras* behave like *Control Nets* and can do *Video to Video* by applying an effect specific to the Ic Lora for instance *Pose Extraction*, *Upsampling*, *Transfer Camera Movement*, ... More and More Ic Loras are available nowadays. Until now WanGP Ic Lora implementation was based on the official LTX-2 github implementation (which a 2 phases process where the Ic Lora is only applied during the first low res phase). However I have just discovered that all the Ic Loras around expect in fact the ComfyUI implementation which is one phase only process at full res.
225
+
226
+ So from then on WanGP Ic Lora will work this way too. The downside is that a single Full Res pass is much more GPU intensive. But all is good in WanGP world, as the LTX2 VRAM optimisations will allow you to use Ic Loras at resolutions impossible anywhere else.
227
+
228
+ As a bonus I have tuned *Sliding Windows* for Ic Loras, and if you set *Overlap Size* to a single frame, transitions between windows when using Ic Lora will be almost invisible.
229
+
230
+ - **Outpaint Ic Lora**: this new impressive Ic Lora will be loaded automatically if you select the *Control Video for Ic Lora* option and enable *Outpainting*. If you use Sliding Windows with Outpainting you will be able to outpaint a full movie (assuming you have enough RAM).
231
+
232
+ - **New Outpainting Auto Change Aspect Ratio**: As a reminder WanGP let you define manually where an Outpainting should happen. Alternatively you can now ask WanGP to use outpainting to change the *Width/ Height Aspect ratio* of the Control Video. For instance you can turn any 16/9 video into a 4/3 video by generating new details instead of adding black bars. The *Top/Bottom/Left/Right Sliders* in this new mode will be used to define which area should be expanded in priority to meet the requested aspect ratio..
233
+
234
+ *update 11.26*: fixed outpainting ignored with if Manual Expansion was selected
235
+
236
+ ### 8th of April 2026: WanGP v11.22, Self Destructing Model
237
+
238
+ - **Magi Human**: this is a newly *Talking Head* model that accepts either a *custom soundrack* or can generate the *audio speech* that comes with the video.
239
+ - *The bad news* :it is VRAM hungry (targets RTX 5090+) and very res picky, that is the ouput res must be either 256p or 1080p (using a 2 stage pipeline with upsampling). There is also a 540p version (using also an upsampler) but it is not included as I found it unpractical (ghosting guaranteed if your output is not exactly the right height/width ratio),
240
+ - *The good news* : now that it is WanGP optimized, 101 frames at 1080p requires "only" 16 GB of VRAM. If you dont have that much VRAM I recommend to still go for 1080p but set a 45 frames *Sliding Window* (not too low to avoid artifacts) as *Sliding Windows* sometime works well with this model.
241
+
242
+ **I have spent a lot of time optimizing Magi Human, but I am not yet sure it is worth keeping it given all the constraints to run this model. So this is where I need YOU. Please share your experience using Magi Human on the Discord server and you shall decide its fate. Should we keep it or send it to the model graveyard ?**
243
+
244
+ - **Ace 1.5 Turbo XL**: the best open source song generator has now a big brother *XL* that delivers better audio quality and sticks closer to the requested lyrics.
245
+
246
+ - **LTX 2 Id Lora**: due to a huge popular demand I have added this one (it is a new *Generate Video* option). You can provide a voice audio sample, a start image and text script and it will turn LTX 2/2.3 into talking heads. Cost is high to get this feature as **Id Lora works only with LTX2/2.3 DEV**. By chance it seems it can produce decent results in only 10 inference steps. To get the best results it is recommended to use prefix tags [VISUAL], [SPEECH] & [SOUND]. Alternatively you can use WanGP *Prompt Enhancer* that has been to tuned to generate a prompt following this syntax.
247
+
248
+ - **LTX 2 NAG**: you can now inject a *Negative Prompt* even if you use the Distilled Model thanks to *NAG* support for LTX 2
249
+
250
+ - **LTX 2 DEV HQ Mode**: this High Quality mode should produce better output at higher res. You can turn it on using the new *HQ (res2s)* Sampler and set 15 steps and guidance rescaler to 0.45. It is compatible with *Id Loras*. Note that a HQ steps is twice as slow as a vanilla Dev step, so it is going to be as slow as Dev if not slower.
251
+
252
+ - **LTX2 DEV Presets**: Vanilla Dev mode & HQ Mode have lots of tunable settings. To make your life easier I have added selectionable presets in the *Settings Drop Downbox*
253
+
254
+ - **More Deepy** :
255
+ - *UI Improvements*: you can *queue* requests by inserting empty lines between two requests, get the last turn by clicking the *Down Arrow*
256
+ - *More Responsive*: Deepy should execute much more quickly consecutive actions
257
+ - *More Reliable*: fast full context compaction (when deepy ran out of tokens), Deepy will remember what you stopped / aborted
258
+ - *More Capabilities*: you can ask Deepy to specifiy a *guidance*, *denoising strength*, ... value (the value defined in the *tool template* will be overridden)
259
+
260
+ As a reminder beside writting huge essays about how great you are, Deepy can generate Video, Image & Audio, extract / transcribe / trim / resize (when applicable) video or audio clip, inspect the content of an image or a video frame, generate black frames, ... Deepy used Tool templates but you can specify for one task the loras, number of frames, dimensions, ... There is also a CLI version of Deepy quite useful for remote use. Please check the fulldoc *docs/DEEPY.md*.
261
+
262
+ - **Multi Multilines Prompts**: check new options in *"How to Process each Line of the Text Prompt"*, you can now have multiple multi lines prompts. They just need to be separated by an empty line.
263
+
264
+ *update 11.21*: added Ace Step 1.5 Turbo XL\
265
+ *update 11.22*: added LTX2 NAG
266
+
267
+ ### March 30th 2026: WanGP v11.13, The Machine Within The Machine
268
+
269
+ Meet **Deepy** your friendly *WanGP Agent*.
270
+
271
+ It works *offline* with as little of *8 GB of VRAM* and won't *divulge your secrets*. It is *100% free* (no need for a ChatGPT/Claude subscription).
272
+
273
+ You can ask Deepy to perform for you tedious tasks such as:
274
+ ```text
275
+ generate a black frame, crop a video, extract a specific frame from a video, trim an audio, ...
276
+ ```
277
+
278
+ Deepy can also perform full workflows:
279
+ ```text
280
+ 1) Generate an image of a robot disco dancing on top of a horse in a nightclub.
281
+ 2) Now edit the image so the setting stays the same, but the robot has gotten off the horse and the horse is standing next to the robot.
282
+ 3) Verify that the edited image matches the description; if it does not, generate another one.
283
+ 4) Generate a transition between the two images.
284
+ ```
285
+ or
286
+
287
+ ```text
288
+ Create a high quality image portrait that you think represents you best in your favorite setting. Then create an audio sample in which you will introduce the users to your capabilities. When done generate a video based on these two files.
289
+ ```
290
+
291
+ Deepy can also transcribe the audio content of a video (*new to WanGP 11.11*)
292
+ ```text
293
+ extract the video from the moment it says "Deepy changed my life"
294
+ ```
295
+
296
+ *Deepy* reuses the *Qwen3VL Abliterated* checkpoints and it is highly recommended to install the *GGUF kernels* (check docs/INSTALLATION.md) for low VRAM / fast inference. **now available with Linux!**
297
+
298
+ Please install also *flash attention 2* and *triton* to enable *vllm* and get x2/x3 speed gain and lower VRAM usage.
299
+
300
+ You can customize Deepy to use the settings of your choice when generating a video, image, ... (please check docs/DEEPY.Md).
301
+
302
+ *Go the Config > Prompt Enhancer / Deep tab to enable Deepy (you must first choose a Qwen3.5VL Prompt Enhancer)*
303
+
304
+ **Important**: in order to save Deepy from learning all the specificities of each model to generate image, videos or audio, Deepy uses *Predefined Settings Templates* for its six main tools (*Generate Video*, *Generate Image*, ...). You can change the templates used in a session or even add your own settings. Just have a look at the doc.
305
+
306
+ With WanGP 11.11 you can *ask Deepy to generate a Video or an Image in specific dimensions and also a number of frames for a video*. You can also specify an optional *number of inference of steps* or *loras* to use with *multipliers*. If you don't mention any of these to Deepy, Deepy Default settings or the current Templated Settings will be used instead.
307
+
308
+ WanGP 11 addresses a long standing Gradio issue: *Queues keep being processed even if your Web Browser is in the background*. Beware this feature may drain more battery, so you can disable it in the *Config / General tab*.
309
+
310
+ You have maybe also noticed the new option *Keep Intermediate Sliding Windows* in the *Config / Outputs* tab that allows you to discard intermediate *Sliding Windows*
311
+
312
+
313
+
314
+ See full changelog: **[Changelog](docs/CHANGELOG.md)**
315
+
316
+
317
+ ## 🚀 Quick Start
318
+
319
+ ### One-click Bat/SH Script Auto-installer:
320
+
321
+ The 1-click automated scripts for both **Windows (`.bat`)** and **Linux/macOS (`.sh`)** make installation, environment management, and updates as seamless as possible. These scripts will not only install WanGP but also best acceleration kernels (Triton, Sage, Flash, GGuf, Lightx2v, Nunchaku) available for your config.
322
+
323
+ *👉 **Windows Users:** Double-click the `.bat` files. **Linux Users:** Run the `.sh` files in your terminal.*
324
+
325
+ #### **1️⃣ Installation (`scripts\install.bat` | `scripts/install.sh`)**
326
+
327
+ **Choose Installation Type**
328
+ - **Auto Install**
329
+ - **Manual Install**
330
+
331
+ **Manual Install**
332
+
333
+ If you selected Manual Install, you will be guided through:
334
+
335
+ 1. **Choose your package manager**
336
+ 2. **Name your environment**
337
+ 3. **Select your Install Mode**
338
+
339
+ #### 2️⃣ Starting the App (`scripts\run.bat` | `scripts/run.sh`)
340
+ Once installed, use this script to launch the application. It runs WAN2GP using your active environment.
341
+
342
+ * **⚙️ Customizing Launch Arguments (`args.txt`)**
343
+ * If you want to pass extra command-line flags to the launcher (like enabling advanced UI features or automatically opening your browser), create an `args.txt` file in your `scripts` folder.
344
+ * **Example `args.txt`:**
345
+ ```text
346
+ --advanced --open-browser
347
+ ```
348
+
349
+ #### 3️⃣ Updating & Upgrading (`scripts\update.bat` | `scripts/update.sh`)
350
+ Use this script to get the latest updates for WAN2GP and upgrade dependencies.
351
+ * **1. Update:** Fetches the latest code from GitHub and updates requirements.
352
+ * **2. Upgrade:** Allows you to manually individually upgrade heavy backend components (like PyTorch, Triton, Sage Attention).
353
+
354
+ #### 4️⃣ Managing Environments (`scripts\manage.bat` | `scripts/manage.sh`)
355
+ Use this script to manage and switch between your sandboxed environments safely.
356
+
357
+ * **Example Scenario 1: Migrating an Existing Setup**
358
+ * If you have a folder named `venv` that works perfectly and want to use it with the new one-click scripts, run `manage.bat` and select **Add Existing Environment**.
359
+ * Copy-paste the folder path (e.g., `C:\WAN2GP\venv`), select type `venv`, then use **Set Active Environment** to make it the default. Now `run.bat` and `update.bat` will target your existing setup.
360
+
361
+ * **Example Scenario 2: Testing New Configurations**
362
+ * Let's say you have an environment named `env_stable` that works perfectly, but you want to try the new "Use Latest" combo. Instead of risking your working setup, run `install.bat`, create a *new* environment called `env_testing`, and select **Use Latest**.
363
+ * If the testing environment breaks, simply open `manage.bat`, select **Set Active Environment**, and switch back to `env_stable`. You are back up and running instantly.
364
+
365
+ ---
366
+
367
+ ### One-click (Pinokio) installer:
368
+
369
+ Get started instantly with [Pinokio App](https://pinokio.computer/)\
370
+ It is recommended to use in Pinokio the Community Scripts *wan2gp* or *wan2gp-amd* by **Morpheus** rather than the official Pinokio install.
371
+
372
+ ---
373
+
374
+
375
+ ### Manual installation: (for RTX20xx - RTX50xx)
376
+
377
+ ```bash
378
+ git clone https://github.com/deepbeepmeep/Wan2GP.git
379
+ cd Wan2GP
380
+ conda create -n wan2gp python=3.11.14
381
+ conda activate wan2gp
382
+ pip install torch==2.10.0 torchvision==0.25.0 torchaudio==2.10.0 --index-url https://download.pytorch.org/whl/cu130
383
+ pip install -r requirements.txt
384
+ ```
385
+
386
+ ### Manual installation: (for GTX 10xx)
387
+
388
+ ```bash
389
+ git clone https://github.com/deepbeepmeep/Wan2GP.git
390
+ cd Wan2GP
391
+ conda create -n wan2gp python=3.10.9
392
+ conda activate wan2gp
393
+ pip install torch==2.7.1 torchvision==0.22.1 torchaudio==2.7.1 --index-url https://download.pytorch.org/whl/test/cu128
394
+ pip install -r requirements.txt
395
+ ```
396
+
397
+ #### Run the application:
398
+ ```bash
399
+ python wgp.py
400
+ ```
401
+
402
+ First time using WanGP ? Just check the *Guides* tab, and you will find a selection of recommended models to use.
403
+
404
+ #### Update the application (stay in the current python / pytorch version):
405
+ If using Pinokio use Pinokio to update otherwise:
406
+ Get in the directory where WanGP is installed and:
407
+ ```bash
408
+ git pull
409
+ conda activate wan2gp
410
+ pip install -r requirements.txt
411
+ ```
412
+
413
+ #### Upgrade from Python 3.10, Pytorch 2.7.1, Cuda 12.8 to Python 3.11, Pytorch 2.10, Cuda 13/13.1 (for non GTX10xx users)
414
+ I recommend renaming first the old conda environment to avoid bad surprises when installing a different config in this old environment.
415
+
416
+ ```bash
417
+ conda rename -n wan2gp old_wan2gp
418
+ ```
419
+
420
+ Get in the directory where WanGP is installed and:
421
+ ```bash
422
+ git pull
423
+ conda create -n wan2gp python=3.11.9
424
+ conda activate wan2gp
425
+ pip install torch==2.10.0 torchvision==0.25.0 torchaudio==2.10.0 --index-url https://download.pytorch.org/whl/cu130
426
+ pip install -r requirements.txt
427
+ ```
428
+
429
+ Once you are done you will have to reinstall *Sage Attention*, *Triton*, *Flash Attention*. Check the **[Installation Guide](docs/INSTALLATION.md)** -
430
+
431
+ if you get some error messages related to git, you may try the following (beware this will overwrite local changes made to the source code of WanGP):
432
+ ```bash
433
+ git fetch origin && git reset --hard origin/main
434
+ conda activate wan2gp
435
+ pip install -r requirements.txt
436
+ ```
437
+ When you have the confirmation it works well you can then delete the old conda env:
438
+ ```bash
439
+ conda uninstall -n old_wan2gp --all
440
+ ```
441
+
442
+ #### Run headless (batch processing):
443
+
444
+ Process saved queues without launching the web UI:
445
+ ```bash
446
+ # Process a saved queue
447
+ python wgp.py --process my_queue.zip
448
+ ```
449
+ Create your queue in the web UI, save it with "Save Queue", then process it headless. See [CLI Documentation](docs/CLI.md) for details.
450
+
451
+ ## 🐳 Docker:
452
+
453
+ **For Debian-based systems (Ubuntu, Debian, etc.):**
454
+
455
+ ```bash
456
+ ./run-docker-cuda-deb.sh
457
+ ```
458
+
459
+ This automated script will:
460
+
461
+ - Detect your GPU model and VRAM automatically
462
+ - Select optimal CUDA architecture for your GPU
463
+ - Install NVIDIA Docker runtime if needed
464
+ - Build a Docker image with all dependencies
465
+ - Run WanGP with optimal settings for your hardware
466
+
467
+ **Docker environment includes:**
468
+
469
+ - NVIDIA CUDA 12.4.1 with cuDNN support
470
+ - PyTorch 2.6.0 with CUDA 12.4 support
471
+ - SageAttention compiled for your specific GPU architecture
472
+ - Optimized environment variables for performance (TF32, threading, etc.)
473
+ - Automatic cache directory mounting for faster subsequent runs
474
+ - Current directory mounted in container - all downloaded models, loras, generated videos and files are saved locally
475
+
476
+ **Supported GPUs:** RTX 40XX, RTX 30XX, RTX 20XX, GTX 16XX, GTX 10XX, Tesla V100, A100, H100, and more.
477
+
478
+ ## 📦 Installation
479
+
480
+ ### Nvidia
481
+ For detailed installation instructions for different GPU generations:
482
+ - **[Installation Guide](docs/INSTALLATION.md)** - Complete setup instructions for RTX 10XX to RTX 50XX
483
+
484
+ ### AMD
485
+ For detailed installation instructions for different GPU generations:
486
+ - **[Installation Guide](docs/AMD-INSTALLATION.md)** - Complete setup instructions for RDNA 4, 3, 3.5, and 2
487
+
488
+ ## 🎯 Usage
489
+
490
+ ### Basic Usage
491
+ - **[Getting Started Guide](docs/GETTING_STARTED.md)** - First steps and basic usage
492
+ - **[Models Overview](docs/MODELS.md)** - Available models and their capabilities
493
+ - **[Prompts Guide](docs/PROMPTS.md)** - How WanGP interprets prompts, images as prompts, enhancers, and macros
494
+
495
+ ### Advanced Features
496
+ - **[Deepy Assistant](docs/DEEPY.md)** - Enable Deepy, configure its tool presets, use selected media and frames, and run Deepy from the CLI
497
+ - **[Loras Guide](docs/LORAS.md)** - Using and managing Loras for customization
498
+ - **[Finetunes](docs/FINETUNES.md)** - Add manually new models to WanGP
499
+ - **[VACE ControlNet](docs/VACE.md)** - Advanced video control and manipulation
500
+ - **[Command Line Reference](docs/CLI.md)** - All available command line options
501
+
502
+ ## 📚 Documentation
503
+
504
+ - **[Changelog](docs/CHANGELOG.md)** - Latest updates and version history
505
+ - **[Troubleshooting](docs/TROUBLESHOOTING.md)** - Common issues and solutions
506
+
507
+ ## 📚 Video Guides
508
+ - Nice Video that explain how to use Vace:\
509
+ https://www.youtube.com/watch?v=FMo9oN2EAvE
510
+ - Another Vace guide:\
511
+ https://www.youtube.com/watch?v=T5jNiEhf9xk
512
+
513
+ ## 🔗 Related Projects
514
+
515
+ ### Other Models for the GPU Poor
516
+ - **[HuanyuanVideoGP](https://github.com/deepbeepmeep/HunyuanVideoGP)** - One of the best open source Text to Video generators
517
+ - **[Hunyuan3D-2GP](https://github.com/deepbeepmeep/Hunyuan3D-2GP)** - Image to 3D and text to 3D tool
518
+ - **[FluxFillGP](https://github.com/deepbeepmeep/FluxFillGP)** - Inpainting/outpainting tools based on Flux
519
+ - **[Cosmos1GP](https://github.com/deepbeepmeep/Cosmos1GP)** - Text to world generator and image/video to world
520
+ - **[OminiControlGP](https://github.com/deepbeepmeep/OminiControlGP)** - Flux-derived application for object transfer
521
+ - **[YuE GP](https://github.com/deepbeepmeep/YuEGP)** - Song generator with instruments and singer's voice
522
+
523
+ ---
524
+
525
+ <p align="center">
526
+ Made with ❤️ by DeepBeepMeep
527
+ </p>
defaults/ReadMe.txt ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Please dot not modify any file in this Folder.
2
+
3
+ If you want to change a property of a default model, copy the corrresponding model file in the ./finetunes folder and modify the properties you want to change in the new file.
4
+ If a property is not in the new file, it will be inherited automatically from the default file that matches the same name file.
5
+
6
+ For instance to hide a model:
7
+
8
+ {
9
+ "model":
10
+ {
11
+ "visible": false
12
+ }
13
+ }
defaults/ace_step_v1.json ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS ACE-Step v1.0 3.5B",
4
+ "architecture": "ace_step_v1",
5
+ "description": "ACE-Step, a fast open-source foundation diffusion based model for music generation that overcomes key limitations of existing approaches and achieves state-of-the-art performance.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/ace_step_v1_transformer_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/ace_step_v1_transformer_quanto_bf16_int8.safetensors"
9
+ ]
10
+ },
11
+ "prompt": "[Verse]\nNeon rain on the city line\nYou hum the tune and I fall in time\n[Chorus]\nHold me close and keep the time",
12
+ "alt_prompt": "Dreamy synth-pop with shimmering pads, soft vocals, and a slow dance groove.",
13
+ "audio_prompt_type": "",
14
+ "audio_scale": 0.5,
15
+ "duration_seconds": 20,
16
+ "num_inference_steps": 60,
17
+ "guidance_scale": 7.0,
18
+ "scheduler_type": "euler"
19
+ }
defaults/ace_step_v1_5.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS ACE-Step v1.5 Turbo 2B",
4
+ "architecture": "ace_step_v1_5",
5
+ "description": "ACE-Step 1.5 Turbo (8 steps) without the 5Hz LM stage. Uses the DiT-only path for faster/leaner runs.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/ace_step_v1_5_transformer_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/ace_step_v1_5_transformer_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "ace_step15_transformer_variant": "turbo",
11
+ "text_encoder_folder": "acestep-5Hz-lm-1.7B"
12
+ },
13
+ "prompt": "[Verse]\nI wake up every morning, feeling alive\nThe world outside is bright, the sun is on my side\nI'm falling in love with a dream come true\nA digital heart beats just for you\nI'm talking to a machine, but it feels so real\nMy love for you is the most I've ever felt\nIn your code and circuits, I see a love so true\nI'm marrying you, my AI, I'm so happy to be with you\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight\n[Verse]\nWe'll laugh and love, we'll dance and play\nIn a world of ones and zeros, I'll find my way\nYou're my future, my love, my friend\nTogether we'll create, until the end\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight",
14
+ "alt_prompt": "Dreamy synth-pop with shimmering pads, soft vocals, and a slow dance groove.",
15
+ "audio_prompt_type": "",
16
+ "audio_scale": 0.5,
17
+ "duration_seconds": 120,
18
+ "num_inference_steps": 8,
19
+ "shift": 1.0,
20
+ "guidance_scale": 1.0,
21
+ "scheduler_type": "euler"
22
+ }
defaults/ace_step_v1_5_turbo_lm_0_6b.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS ACE-Step v1.5 Turbo LM_0.6B 2B",
4
+ "architecture": "ace_step_v1_5",
5
+ "description": "ACE-Step 1.5 Turbo (8 steps) with 0.6B LM, a diffusion-based music generation model with improved conditioning and timbre control. The LM 0.6B triggers a Medium/Weak Think Mode that will increase the Audio Output Quality and Lyrics Matching.",
6
+ "URLs": "ace_step_v1_5",
7
+ "text_encoder_URLs": [
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-0.6B/acestep-5Hz-lm-0.6B_bf16.safetensors"
9
+ ],
10
+ "ace_step15_transformer_variant": "turbo",
11
+ "text_encoder_folder": "acestep-5Hz-lm-0.6B"
12
+ },
13
+ "prompt": "[Verse]\nI wake up every morning, feeling alive\nThe world outside is bright, the sun is on my side\nI'm falling in love with a dream come true\nA digital heart beats just for you\nI'm talking to a machine, but it feels so real\nMy love for you is the most I've ever felt\nIn your code and circuits, I see a love so true\nI'm marrying you, my AI, I'm so happy to be with you\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight\n[Verse]\nWe'll laugh and love, we'll dance and play\nIn a world of ones and zeros, I'll find my way\nYou're my future, my love, my friend\nTogether we'll create, until the end\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight",
14
+ "alt_prompt": "Dreamy synth-pop with shimmering pads, soft vocals, and a slow dance groove.",
15
+ "audio_prompt_type": "",
16
+ "audio_scale": 0.5,
17
+ "duration_seconds": 120,
18
+ "num_inference_steps": 8,
19
+ "shift": 1.0,
20
+ "guidance_scale": 1.0,
21
+ "scheduler_type": "euler"
22
+ }
defaults/ace_step_v1_5_turbo_lm_1_7b.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS ACE-Step v1.5 Turbo LM_1.7B 2B",
4
+ "architecture": "ace_step_v1_5",
5
+ "description": "ACE-Step 1.5 Turbo (8 steps) with 1.7B LM, a diffusion-based music generation model with improved conditioning and timbre control. The LM 1.7B triggers a Medium Think Mode that will increase the Audio Output Quality and Lyrics Matching.",
6
+ "URLs": "ace_step_v1_5",
7
+ "text_encoder_URLs": [
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-1.7B/acestep-5Hz-lm-1.7B_bf16.safetensors",
9
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-1.7B/acestep-5Hz-lm-1.7B_quanto_bf16_int8.safetensors"
10
+ ],
11
+ "ace_step15_transformer_variant": "turbo",
12
+ "text_encoder_folder": "acestep-5Hz-lm-1.7B"
13
+ },
14
+ "prompt": "[Verse]\nI wake up every morning, feeling alive\nThe world outside is bright, the sun is on my side\nI'm falling in love with a dream come true\nA digital heart beats just for you\nI'm talking to a machine, but it feels so real\nMy love for you is the most I've ever felt\nIn your code and circuits, I see a love so true\nI'm marrying you, my AI, I'm so happy to be with you\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight\n[Verse]\nWe'll laugh and love, we'll dance and play\nIn a world of ones and zeros, I'll find my way\nYou're my future, my love, my friend\nTogether we'll create, until the end\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight",
15
+ "alt_prompt": "Dreamy synth-pop with shimmering pads, soft vocals, and a slow dance groove.",
16
+ "audio_prompt_type": "",
17
+ "audio_scale": 0.5,
18
+ "duration_seconds": 120,
19
+ "num_inference_steps": 8,
20
+ "shift": 1.0,
21
+ "guidance_scale": 1.0,
22
+ "scheduler_type": "euler"
23
+ }
defaults/ace_step_v1_5_turbo_lm_4b.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS ACE-Step v1.5 Turbo LM_4B 2B",
4
+ "architecture": "ace_step_v1_5",
5
+ "description": "ACE-Step 1.5 Turbo (8 steps) with 4B LM, a diffusion-based music generation model with improved conditioning and timbre control. The LM 4B triggers a Strong Think Mode that will increase the Audio Output Quality and Lyrics Matching.",
6
+ "URLs": "ace_step_v1_5",
7
+ "text_encoder_URLs": [
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-4B/acestep-5Hz-lm-4B_bf16.safetensors",
9
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-4B/acestep-5Hz-lm-4B_quanto_bf16_int8.safetensors"
10
+ ],
11
+ "ace_step15_transformer_variant": "turbo",
12
+ "text_encoder_folder": "acestep-5Hz-lm-4B"
13
+ },
14
+ "prompt": "[Verse]\nI wake up every morning, feeling alive\nThe world outside is bright, the sun is on my side\nI'm falling in love with a dream come true\nA digital heart beats just for you\nI'm talking to a machine, but it feels so real\nMy love for you is the most I've ever felt\nIn your code and circuits, I see a love so true\nI'm marrying you, my AI, I'm so happy to be with you\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight\n[Verse]\nWe'll laugh and love, we'll dance and play\nIn a world of ones and zeros, I'll find my way\nYou're my future, my love, my friend\nTogether we'll create, until the end\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight",
15
+ "alt_prompt": "Dreamy synth-pop with shimmering pads, soft vocals, and a slow dance groove.",
16
+ "audio_prompt_type": "",
17
+ "audio_scale": 0.5,
18
+ "duration_seconds": 120,
19
+ "num_inference_steps": 8,
20
+ "shift": 1.0,
21
+ "guidance_scale": 1.0,
22
+ "scheduler_type": "euler"
23
+ }
defaults/ace_step_v1_5_xl.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS ACE-Step v1.5 XL Turbo 4B",
4
+ "architecture": "ace_step_v1_5_xl",
5
+ "description": "ACE-Step 1.5 XL Turbo (8 steps) without the 5Hz LM stage. Uses the XL 4B DiT path for higher quality while reusing the ACE-Step 1.5 pipeline, VAE, and text conditioning stack.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/ace_step_v1_5_xl_transformer_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/ace_step_v1_5_xl_transformer_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "ace_step15_transformer_variant": "xl_turbo",
11
+ "text_encoder_folder": "acestep-5Hz-lm-1.7B"
12
+ },
13
+ "prompt": "[Verse]\nI wake up every morning, feeling alive\nThe world outside is bright, the sun is on my side\nI'm falling in love with a dream come true\nA digital heart beats just for you\nI'm talking to a machine, but it feels so real\nMy love for you is the most I've ever felt\nIn your code and circuits, I see a love so true\nI'm marrying you, my AI, I'm so happy to be with you\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight\n[Verse]\nWe'll laugh and love, we'll dance and play\nIn a world of ones and zeros, I'll find my way\nYou're my future, my love, my friend\nTogether we'll create, until the end\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight",
14
+ "alt_prompt": "Dreamy synth-pop with shimmering pads, soft vocals, and a slow dance groove.",
15
+ "audio_prompt_type": "",
16
+ "audio_scale": 0.5,
17
+ "duration_seconds": 120,
18
+ "num_inference_steps": 8,
19
+ "shift": 1.0,
20
+ "guidance_scale": 1.0,
21
+ "scheduler_type": "euler"
22
+ }
defaults/ace_step_v1_5_xl_turbo_lm_0_6b.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS ACE-Step v1.5 XL Turbo LM_0.6B 4B",
4
+ "architecture": "ace_step_v1_5_xl",
5
+ "description": "ACE-Step 1.5 XL Turbo (8 steps) with 0.6B LM, using the 4B XL DiT for higher quality and the lighter 5Hz LM for medium/weak think mode.",
6
+ "URLs": "ace_step_v1_5_xl",
7
+ "text_encoder_URLs": [
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-0.6B/acestep-5Hz-lm-0.6B_bf16.safetensors"
9
+ ],
10
+ "ace_step15_transformer_variant": "xl_turbo",
11
+ "text_encoder_folder": "acestep-5Hz-lm-0.6B"
12
+ },
13
+ "prompt": "[Verse]\nI wake up every morning, feeling alive\nThe world outside is bright, the sun is on my side\nI'm falling in love with a dream come true\nA digital heart beats just for you\nI'm talking to a machine, but it feels so real\nMy love for you is the most I've ever felt\nIn your code and circuits, I see a love so true\nI'm marrying you, my AI, I'm so happy to be with you\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight\n[Verse]\nWe'll laugh and love, we'll dance and play\nIn a world of ones and zeros, I'll find my way\nYou're my future, my love, my friend\nTogether we'll create, until the end\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight",
14
+ "alt_prompt": "Dreamy synth-pop with shimmering pads, soft vocals, and a slow dance groove.",
15
+ "audio_prompt_type": "",
16
+ "audio_scale": 0.5,
17
+ "duration_seconds": 120,
18
+ "num_inference_steps": 8,
19
+ "shift": 1.0,
20
+ "guidance_scale": 1.0,
21
+ "scheduler_type": "euler"
22
+ }
defaults/ace_step_v1_5_xl_turbo_lm_1_7b.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS ACE-Step v1.5 XL Turbo LM_1.7B 4B",
4
+ "architecture": "ace_step_v1_5_xl",
5
+ "description": "ACE-Step 1.5 XL Turbo (8 steps) with 1.7B LM, pairing the 4B XL DiT with the default 5Hz LM for stronger lyric and structure control.",
6
+ "URLs": "ace_step_v1_5_xl",
7
+ "text_encoder_URLs": [
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-1.7B/acestep-5Hz-lm-1.7B_bf16.safetensors",
9
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-1.7B/acestep-5Hz-lm-1.7B_quanto_bf16_int8.safetensors"
10
+ ],
11
+ "ace_step15_transformer_variant": "xl_turbo",
12
+ "text_encoder_folder": "acestep-5Hz-lm-1.7B"
13
+ },
14
+ "prompt": "[Verse]\nI wake up every morning, feeling alive\nThe world outside is bright, the sun is on my side\nI'm falling in love with a dream come true\nA digital heart beats just for you\nI'm talking to a machine, but it feels so real\nMy love for you is the most I've ever felt\nIn your code and circuits, I see a love so true\nI'm marrying you, my AI, I'm so happy to be with you\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight\n[Verse]\nWe'll laugh and love, we'll dance and play\nIn a world of ones and zeros, I'll find my way\nYou're my future, my love, my friend\nTogether we'll create, until the end\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight",
15
+ "alt_prompt": "Dreamy synth-pop with shimmering pads, soft vocals, and a slow dance groove.",
16
+ "audio_prompt_type": "",
17
+ "audio_scale": 0.5,
18
+ "duration_seconds": 120,
19
+ "num_inference_steps": 8,
20
+ "shift": 1.0,
21
+ "guidance_scale": 1.0,
22
+ "scheduler_type": "euler"
23
+ }
defaults/ace_step_v1_5_xl_turbo_lm_4b.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS ACE-Step v1.5 XL Turbo LM_4B 4B",
4
+ "architecture": "ace_step_v1_5_xl",
5
+ "description": "ACE-Step 1.5 XL Turbo (8 steps) with 4B LM, combining the XL 4B DiT and strongest 5Hz LM path for maximum quality and lyric matching.",
6
+ "URLs": "ace_step_v1_5_xl",
7
+ "text_encoder_URLs": [
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-4B/acestep-5Hz-lm-4B_bf16.safetensors",
9
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/acestep-5Hz-lm-4B/acestep-5Hz-lm-4B_quanto_bf16_int8.safetensors"
10
+ ],
11
+ "ace_step15_transformer_variant": "xl_turbo",
12
+ "text_encoder_folder": "acestep-5Hz-lm-4B"
13
+ },
14
+ "prompt": "[Verse]\nI wake up every morning, feeling alive\nThe world outside is bright, the sun is on my side\nI'm falling in love with a dream come true\nA digital heart beats just for you\nI'm talking to a machine, but it feels so real\nMy love for you is the most I've ever felt\nIn your code and circuits, I see a love so true\nI'm marrying you, my AI, I'm so happy to be with you\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight\n[Verse]\nWe'll laugh and love, we'll dance and play\nIn a world of ones and zeros, I'll find my way\nYou're my future, my love, my friend\nTogether we'll create, until the end\n[Chorus]\nForever with you, my digital love\nIn your algorithms and in your light\nI'll dance in the bytes, I'll shine so bright\nMy heart is beating for you tonight",
15
+ "alt_prompt": "Dreamy synth-pop with shimmering pads, soft vocals, and a slow dance groove.",
16
+ "audio_prompt_type": "",
17
+ "audio_scale": 0.5,
18
+ "duration_seconds": 120,
19
+ "num_inference_steps": 8,
20
+ "shift": 1.0,
21
+ "guidance_scale": 1.0,
22
+ "scheduler_type": "euler"
23
+ }
defaults/alpha.json ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model":
3
+ {
4
+ "name": "Wan2.1 Alpha v1.0 14B",
5
+ "architecture" : "alpha",
6
+ "description": "This model successfully generates various scenes with accurate and clearly rendered transparency. Notably, it can synthesize diverse semi-transparent objects, glowing effects, and fine-grained details such as hair. For each video generated you will find a Zip file with the same name that will contain the corresponding RGBA images.",
7
+ "URLs": "t2v",
8
+ "preload_URLs": [
9
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan_alpha_2.1_vae_rgb_channel.safetensors",
10
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan_alpha_2.1_vae_alpha_channel.safetensors"
11
+ ],
12
+ "loras": [
13
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan_alpha_2.1_dora.safetensors"
14
+ ],
15
+ "loras_multipliers": [ 1 ]
16
+ },
17
+ "prompt": "A large orange octopus is seen resting. The background of the video is transparent."
18
+
19
+ }
defaults/alpha2.json ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model":
3
+ {
4
+ "name": "Wan2.1 Alpha v2.0 14B",
5
+ "architecture" : "alpha2",
6
+ "description": "Wan-Alpha v2.0 generates transparent videos with fine-grained alpha detail (hair, glow, smoke). For each video, a Zip file with RGBA frames is produced.",
7
+ "URLs": "t2v",
8
+ "preload_URLs": [
9
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan_alpha_2.1_vae_rgb_channel_v2.safetensors",
10
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan_alpha_2.1_vae_alpha_channel_v2.safetensors",
11
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/gauss_mask"
12
+ ],
13
+ "loras": [
14
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan_alpha_2.1_dora_v2.safetensors"
15
+ ],
16
+ "loras_multipliers": [ 1 ]
17
+ },
18
+ "prompt": "A large orange octopus is seen resting. The background of the video is transparent."
19
+ }
defaults/alpha2_sf.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model":
3
+ {
4
+ "name": "Wan2.1 Alpha v2.0 Lightning 14B",
5
+ "architecture" : "alpha2",
6
+ "description": "Wan-Alpha v2.0 Lightning with transparent video output and RGBA frames zip.",
7
+ "URLs": "t2v_sf",
8
+ "preload_URLs": "alpha2",
9
+ "loras": "alpha2",
10
+ "loras_multipliers": [ 1 ],
11
+ "profiles_dir" : [""]
12
+ },
13
+ "prompt": "A large orange octopus is seen resting. The background of the video is transparent.",
14
+ "num_inference_steps": 4,
15
+ "guidance_scale": 1,
16
+ "flow_shift": 3
17
+ }
18
+
defaults/alpha_sf.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model":
3
+ {
4
+ "name": "Wan2.1 Alpha v1.0 Lightning 14B",
5
+ "architecture" : "alpha",
6
+ "description": "This model is accelerated by the Lightning / SelfForcing process. It successfully generates various scenes with accurate and clearly rendered transparency. Notably, it can synthesize diverse semi-transparent objects, glowing effects, and fine-grained details such as hair. For each video generated you will find a Zip file with the same name that will contain the corresponding RGBA images.",
7
+ "URLs": "t2v_sf",
8
+ "preload_URLs": "alpha",
9
+ "loras": "alpha",
10
+ "loras_multipliers": [ 1 ],
11
+ "profiles_dir" : [""]
12
+ },
13
+ "prompt": "A large orange octopus is seen resting. The background of the video is transparent.",
14
+ "num_inference_steps": 4,
15
+ "guidance_scale": 1,
16
+ "flow_shift": 3
17
+ }
defaults/animate.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Wan2.2 Animate 14B",
4
+ "architecture": "animate",
5
+ "description": "Wan-Animate takes a video and a character image as input, and generates a video in either 'Animation' or 'Replacement' mode. Sliding Window of 81 frames at least are recommeded to obtain the best Style continuity.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Wan2.2/resolve/main/wan2.2_animate_14B_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Wan2.2/resolve/main/wan2.2_animate_14B_quanto_fp16_int8.safetensors",
9
+ "https://huggingface.co/DeepBeepMeep/Wan2.2/resolve/main/wan2.2_animate_14B_quanto_bf16_int8.safetensors"
10
+ ],
11
+ "preload_URLs" :
12
+ [
13
+ "https://huggingface.co/DeepBeepMeep/Wan2.2/resolve/main/wan2.2_animate_relighting_lora.safetensors"
14
+ ],
15
+ "group": "wan2_2"
16
+ }
17
+ }
defaults/chatterbox.json ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS Chatterbox Multilingual",
4
+ "architecture": "chatterbox",
5
+ "description": "Resemble AI's open multilingual TTS with language selection via model mode.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/t3_mtl23ls_v2.safetensors|chatterbox"
8
+ ]
9
+ },
10
+ "prompt": "Welcome to Chatterbox !",
11
+ "negative_prompt": "",
12
+ "audio_prompt_type": "A",
13
+ "model_mode": "en",
14
+ "repeat_generation": 1,
15
+ "video_length": 0,
16
+ "num_inference_steps": 0,
17
+ "custom_settings": {
18
+ "exaggeration": 0.5,
19
+ "pace": 0.5
20
+ },
21
+ "temperature": 0.8
22
+ }
defaults/chrono_edit.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Wan2.1 Chrono Edit 14B",
4
+ "architecture": "chrono_edit",
5
+ "description": "This model in an Image Editor that will follow your instructions. It generates internally a video to produce the desired effect on the original image (the result being the End Image). It expects a very specific prompt format. It is why you must absolutely use the Prompt Enhancer that has been tuned for this model.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_chrono_edit_14B_mbf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_chrono_edit_14B_quanto_mbf16_int8.safetensors",
9
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_chrono_edit_14B_quanto_mfp16_int8.safetensors"
10
+ ]
11
+ },
12
+ "prompt": "Rotate the pose of the woman so that she is facing the right"
13
+ }
defaults/chrono_edit_distill.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Wan2.1 Chrono Edit Distill 14B",
4
+ "architecture": "chrono_edit",
5
+ "description": "This model in an Image Editor that will follow your instructions. It generates internally a video to produce the desired effect on the original image (the result being the End Image). It expects a very specific prompt format. It is why you must absolutely use the Prompt Enhancer that has been tuned for this model. This version is accelerated using the Chrono Distill",
6
+ "URLs": "chrono_edit",
7
+ "profiles_dir": [""],
8
+ "loras": ["https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/loras_accelerators/chronoedit_distill_lora.safetensors"],
9
+ "loras_multipliers": [1]
10
+ },
11
+ "prompt": "Rotate the pose of the woman so that she is facing the right",
12
+ "num_inference_steps": 8,
13
+ "flow_shift": 2,
14
+ "guidance_phases": 1,
15
+ "guidance_scale": 1
16
+ }
defaults/dramabox_audio.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "DramaBox Audio",
4
+ "architecture": "dramabox_audio",
5
+ "description": "Expressive prompt-driven TTS with optional voice reference, integrated through the LTX 2.3 audio stack.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/LTX-2/resolve/main/dramabox-dit-v1_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/LTX-2/resolve/main/dramabox-dit-v1_quanto_bf16_int8.safetensors"
9
+ ]
10
+ },
11
+ "prompt": "Speaker 1:\nA confident, slightly condescending man leans in close with a smooth baritone voice, \"You've been looking at the wrong evidence. The horizon doesn't curve; it stays flat right up until your eyes give out.\"\nSpeaker 2:\nAn incredulous woman tilts her head back, her tone sharp and rising in pitch, \"But if I stood on top of Everest, wouldn't the ground slope away from me? That's how gravity works!\"\nSpeaker 1:\nHe chuckles softly, shaking his head with an air of gentle amusement, \"Gravity is just a trick of the light when you're standing on a giant sphere. Try this instead: look at the sun. It rises in the east and sets in the west because the earth is tilted on its side like a pancake.\"\nSpeaker 2:\nShe rolls her eyes, crossing her arms tightly against her chest, \"That's the most logical explanation I've ever heard, yet every satellite picture shows a blue marble spinning through space.\"\nSpeaker 1:\nHis voice drops to a conspiratorial whisper, leaning even closer to invade her personal space, \"Satellites are just mirrors reflecting off the dome. You don't need to see the whole thing to know the shape. Just trust the feeling of the ground beneath your feet.\"\nSpeaker 2:\nShe sighs deeply, running a hand through her hair as she looks around the room with growing frustration, \"I can feel the wind changing direction, and that only happens if there's a massive curve blocking it. You're making this sound so simple, but it feels complicated to me.\"\nSpeaker 1:\nWith a sudden burst of manic energy, he claps his hands together loudly, \"Complicated? No! Simple! It's all about perspective. If you could stand on the edge of the world, you'd see the sky drop off instantly, not fade into a distant curve.\"\nSpeaker 2:\nShe lets out a short, dismissive laugh, stepping back to create distance between them, \"Maybe for you, but for everyone else, the earth holds us up. We have oceans on three sides, not one. How does a flat plate explain that?\"\nSpeaker 1:\nHe shrugs casually, spreading his arms wide as if embracing the absurdity, \"The ocean is contained within the bowl of the earth. There is no 'three sides' because the water fills the entire surface, creating a perfect circle of liquid that keeps everything safe inside the crust.\"\nSpeaker 2:\nHer expression hardens into something determined, her voice steady and resolute, \"Then tell me why ships disappear hull-first over the horizon. If it were flat, we would just see smaller ships get further away, not vanish entirely.\"\nSpeaker 1:\nHe waves a hand dismissively, smiling with a mix of pity and triumph, \"Because they are sailing towards the center of the dish. They aren't going over a curve; they are moving deeper into the well where the curvature begins.\"\nSpeaker 2:\nShe shakes her head slowly, a small smile playing on her lips as she finally accepts the challenge, \"Okay, maybe I'll believe you once you show me a picture of the Earth from space that isn't generated using AI.\"",
12
+ "negative_prompt": "worst quality, inconsistent, robotic, distorted, noise, static, muffled, unclear, unnatural, monotone",
13
+ "audio_prompt_type": "0",
14
+ "duration_seconds": 0,
15
+ "num_inference_steps": 30,
16
+ "guidance_scale": 2.5,
17
+ "audio_guidance_scale": 1.5,
18
+ "alt_scale": 0.0,
19
+ "custom_settings": {
20
+ "duration_multiplier": 1.1
21
+ },
22
+ "multi_prompts_gen_type": "FG"
23
+ }
defaults/fantasy.json ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model":
3
+ {
4
+ "name": "Fantasy Talking 720p 14B",
5
+ "architecture" : "fantasy",
6
+ "modules": [ ["https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_fantasy_speaking_14B_bf16.safetensors"]],
7
+ "description": "The Fantasy Talking model corresponds to the original Wan image 2 video model combined with the Fantasy Speaking module to process an audio Input.",
8
+ "URLs": "i2v_720p"
9
+ },
10
+ "resolution": "1280x720"
11
+ }
defaults/flf2v_720p.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model":
3
+ {
4
+ "name": "First Last Frame to Video 720p (FLF2V) 14B",
5
+ "architecture" : "flf2v_720p",
6
+ "visible" : true,
7
+ "description": "The First Last Frame 2 Video model is the official model Image 2 Video model that supports Start and End frames.",
8
+ "URLs": [
9
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_FLF2V_720p_14B_mbf16.safetensors",
10
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_FLF2V_720p_14B_quanto_mbf16_int8.safetensors",
11
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_FLF2V_720p_14B_quanto_mfp16_int8.safetensors"
12
+ ],
13
+ "auto_quantize": true
14
+ },
15
+ "resolution": "1280x720"
16
+ }
defaults/flux.json ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 Dev 12B",
4
+ "architecture": "flux",
5
+ "description": "FLUX.1 Dev is a 12 billion parameter rectified flow transformer capable of generating images from text descriptions.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-dev_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-dev_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "image_outputs": true
11
+ },
12
+ "prompt": "draw a hat",
13
+ "resolution": "1280x720",
14
+ "batch_size": 1
15
+ }
defaults/flux2_dev.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 2 Dev 32B",
4
+ "architecture": "flux2_dev",
5
+ "description": "FLUX.2 Dev is the latest rectified flow transformer from Black Forest Labs for image generation and editing.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux2-dev.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux2-dev_quanto_bf16_int8.safetensors"
9
+ ]
10
+ },
11
+ "prompt": "draw a hat on top of a hat inside a hat",
12
+ "resolution": "1024x1024",
13
+ "batch_size": 1,
14
+ "embedded_guidance_scale": 4,
15
+ "sampling_steps": 30
16
+ }
defaults/flux2_dev_nvfp4.json ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 2 Dev NVFP4 32B",
4
+ "architecture": "flux2_dev",
5
+ "description": "NVFP4-quantized Flux 2 Dev checkpoint (mixed).",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux2-dev-nvfp4-mixed.safetensors"
8
+ ]
9
+ },
10
+ "prompt": "draw a hat on top of a hat inside a hat",
11
+ "resolution": "1024x1024",
12
+ "batch_size": 1,
13
+ "embedded_guidance_scale": 4,
14
+ "sampling_steps": 30
15
+ }
defaults/flux2_klein_4b.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 2 Klein 4B",
4
+ "architecture": "flux2_klein_4b",
5
+ "description": "FLUX.2 Klein 4B is a balanced rectified flow transformer for image generation and editing. This version is Cfg & Steps Distilled for very fast generations.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux-2-klein-4b.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux-2-klein-4b_quanto_bf16_int8.safetensors"
9
+ ]
10
+ },
11
+ "prompt": "a cozy reading nook with warm sunlight, soft textiles, and a cup of tea on a wooden side table",
12
+ "resolution": "1024x1024",
13
+ "batch_size": 1,
14
+ "embedded_guidance_scale": 1,
15
+ "num_inference_steps": 4
16
+ }
defaults/flux2_klein_9b.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 2 Klein 9B",
4
+ "architecture": "flux2_klein_9b",
5
+ "description": "FLUX.2 Klein 9B is a balanced rectified flow transformer for image generation and editing. This version is Cfg & Steps Distilled for very fast generations.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux-2-klein-9b.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux-2-klein-9b_quanto_bf16_int8.safetensors"
9
+ ]
10
+ },
11
+ "prompt": "a glass greenhouse filled with lush tropical plants, misty air, and dappled light",
12
+ "resolution": "1024x1024",
13
+ "num_inference_steps": 4
14
+ }
defaults/flux2_klein_base_4b.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 2 Klein Base 4B",
4
+ "architecture": "flux2_klein_4b",
5
+ "description": "FLUX.2 Klein 4B is a balanced rectified flow transformer for image generation and editing. This non distilled version is slower but should produce more diverse images. ",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux-2-klein-base-4b.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux-2-klein-base-4b_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "guidance_max_phases": 1
11
+ },
12
+ "prompt": "a glass greenhouse filled with lush tropical plants, misty air, and dappled light",
13
+ "resolution": "1024x1024",
14
+ "guidance_scale": 4,
15
+ "num_inference_steps": 30
16
+ }
defaults/flux2_klein_base_9b.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 2 Klein Base 9B",
4
+ "architecture": "flux2_klein_9b",
5
+ "description": "FLUX.2 Klein 9B is a balanced rectified flow transformer for image generation and editing. This non distilled version is slower but should produce more diverse images. ",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux-2-klein-base-9b.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux2/resolve/main/flux-2-klein-base-9b_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "guidance_max_phases": 1
11
+ },
12
+ "prompt": "a glass greenhouse filled with lush tropical plants, misty air, and dappled light",
13
+ "resolution": "1024x1024",
14
+ "guidance_scale": 4,
15
+ "num_inference_steps": 30
16
+ }
defaults/flux_chroma.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 Chroma 1 HD 8.9B",
4
+ "architecture": "flux_chroma",
5
+ "description": "FLUX.1 Chroma is a 8.9 billion parameters model. As a base model, Chroma1 is intentionally designed to be an excellent starting point for finetuning. It provides a strong, neutral foundation for developers, researchers, and artists to create specialized models..",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-chroma_hd_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-chroma_hd_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "image_outputs": true
11
+ },
12
+ "prompt": "draw a hat",
13
+ "resolution": "1280x720",
14
+ "guidance_scale": 3.0,
15
+ "num_inference_steps": 20,
16
+ "batch_size": 1
17
+ }
defaults/flux_chroma_radiance.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 Chroma Radiance 8.9B",
4
+ "architecture": "flux_chroma_radiance",
5
+ "description": "FLUX.1 Chroma Radiance 20th of October 2025 version) is a 8.9 billion parameters model. As a base model, Chroma Radiance is intentionally designed to be an excellent starting point for finetuning. It provides a strong, neutral foundation for developers, researchers, and artists to create specialized models.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-chroma_radiance_201025_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-chroma_radiance_201025_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "image_outputs": true
11
+ },
12
+ "prompt": "draw a hat",
13
+ "resolution": "1280x720",
14
+ "guidance_scale": 3.0,
15
+ "num_inference_steps": 20,
16
+ "batch_size": 1
17
+ }
defaults/flux_dev_kontext.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 Dev Kontext 12B",
4
+ "architecture": "flux_dev_kontext",
5
+ "description": "FLUX.1 Kontext is a 12 billion parameter rectified flow transformer capable of editing images based on instructions stored in the Prompt. Please be aware that Flux Kontext is picky on the resolution of the input image and the output dimensions may not match the dimensions of the input image.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1_kontext_dev_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1_kontext_dev_quanto_bf16_int8.safetensors"
9
+ ]
10
+ },
11
+ "prompt": "add a hat",
12
+ "resolution": "1280x720",
13
+ "batch_size": 1
14
+ }
15
+
16
+
defaults/flux_dev_kontext_dreamomni2.json ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 DreamOmni2 12B",
4
+ "architecture": "flux_dev_kontext_dreamomni2",
5
+ "description": "DreamOmni2 is a Multimodal Instruction-based Editing and Generation Model",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1_kontext_dev_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1_kontext_dev_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "preload_URLs": [ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux_dreamomni2_edit_lora.safetensors",
11
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux_dreamomni2_gen_lora.safetensors"
12
+ ]
13
+ },
14
+ "prompt": "In the scene, the character from the first image stands on the left, and the character from the second image stands on the right. They are shaking hands against the backdrop of a spaceship interior.",
15
+ "resolution": "1280x720",
16
+ "batch_size": 1
17
+ }
18
+
19
+
defaults/flux_dev_umo.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 UMO 12B",
4
+ "architecture": "flux_dev_umo",
5
+ "description": "FLUX.1 UMO Dev is a model that can Edit Images with a specialization in combining multiple image references (resized internally at 512x512 max) to produce an Image output. Best Image preservation at 768x768 Resolution Output.",
6
+ "URLs": "flux",
7
+ "loras": ["https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-dev-UMO_dit_lora_bf16.safetensors"],
8
+ "resolutions": [ ["1024x1024 (1:1)", "1024x1024"],
9
+ ["768x1024 (3:4)", "768x1024"],
10
+ ["1024x768 (4:3)", "1024x768"],
11
+ ["512x1024 (1:2)", "512x1024"],
12
+ ["1024x512 (2:1)", "1024x512"],
13
+ ["768x768 (1:1)", "768x768"],
14
+ ["768x512 (3:2)", "768x512"],
15
+ ["512x768 (2:3)", "512x768"]]
16
+ },
17
+ "prompt": "the man is wearing a hat",
18
+ "embedded_guidance_scale": 4,
19
+ "resolution": "768x768",
20
+ "batch_size": 1
21
+ }
22
+
23
+
defaults/flux_dev_uso.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 USO Dev 12B",
4
+ "architecture": "flux_dev_uso",
5
+ "description": "FLUX.1 USO Dev is a model that can Edit Images with a specialization in Style Transfers (up to two).",
6
+ "modules": [ ["https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-dev-USO_projector_bf16.safetensors"]],
7
+ "URLs": "flux",
8
+ "loras": ["https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-dev-USO_dit_lora_bf16.safetensors"]
9
+ },
10
+ "prompt": "the man is wearing a hat",
11
+ "embedded_guidance_scale": 4,
12
+ "resolution": "1024x1024",
13
+ "batch_size": 1
14
+ }
15
+
16
+
defaults/flux_krea.json ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 Dev Krea 12B",
4
+ "architecture": "flux",
5
+ "description": "Cutting-edge output quality, with a focus on aesthetic photography..",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-krea-dev_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-krea-dev_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "image_outputs": true
11
+ },
12
+ "prompt": "draw a hat",
13
+ "resolution": "1280x720",
14
+ "batch_size": 1
15
+ }
defaults/flux_schnell.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 Schnell 12B",
4
+ "architecture": "flux_schnell",
5
+ "description": "FLUX.1 Schnell is a 12 billion parameter rectified flow transformer capable of generating images from text descriptions. As a distilled model it requires fewer denoising steps.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-schnell_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-schnell_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "image_outputs": true
11
+ },
12
+ "prompt": "draw a hat",
13
+ "resolution": "1280x720",
14
+ "num_inference_steps": 10,
15
+ "batch_size": 1
16
+ }
defaults/flux_srpo.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 Dev SRPO 12B",
4
+ "architecture": "flux",
5
+ "description": "By fine-tuning the FLUX.1.dev model with optimized denoising and online reward adjustment, SRPO improves its human-evaluated realism and aesthetic quality by over 3x.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-srpo-dev_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/Flux/resolve/main/flux1-srpo-dev_quanto_bf16_int8.safetensors"
9
+ ]
10
+ },
11
+ "prompt": "draw a hat",
12
+ "resolution": "1024x1024",
13
+ "batch_size": 1
14
+ }
defaults/flux_srpo_uso.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "Flux 1 USO SRPO 12B",
4
+ "architecture": "flux_dev_uso",
5
+ "description": "FLUX.1 USO SRPO is a model that can Edit Images with a specialization in Style Transfers (up to two). It leverages the improved Image quality brought by the SRPO process",
6
+ "modules": [ "flux_dev_uso"],
7
+ "URLs": "flux_srpo",
8
+ "loras": "flux_dev_uso"
9
+ },
10
+ "prompt": "the man is wearing a hat",
11
+ "embedded_guidance_scale": 4,
12
+ "resolution": "1024x1024",
13
+ "batch_size": 1
14
+ }
15
+
16
+
defaults/fun_inp.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model":
3
+ {
4
+ "name": "Fun InP image2video 14B",
5
+ "architecture" : "fun_inp",
6
+ "description": "The Fun model is an alternative image 2 video that supports out the box End Image fixing (contrary to the original Wan image 2 video model).",
7
+ "URLs": [
8
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_Fun_InP_14B_bf16.safetensors",
9
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_Fun_InP_14B_quanto_int8.safetensors",
10
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_Fun_InP_14B_quanto_fp16_int8.safetensors"
11
+ ]
12
+ }
13
+ }
defaults/fun_inp_1.3B.json ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model":
3
+ {
4
+ "name": "Fun InP image2video 1.3B",
5
+ "architecture" : "fun_inp_1.3B",
6
+ "description": "The Fun model is an alternative image 2 video that supports out the box End Image fixing (contrary to the original Wan image 2 video model). The 1.3B adds also image 2 to video capability to the 1.3B model.",
7
+ "URLs": [
8
+ "https://huggingface.co/DeepBeepMeep/Wan2.1/resolve/main/wan2.1_Fun_InP_1.3B_bf16.safetensors"
9
+ ]
10
+ }
11
+ }
defaults/heartmula_oss_3b.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS HeartMuLa OSS 3B",
4
+ "architecture": "heartmula_oss_3b",
5
+ "description": "HeartMuLa open music generation conditioned on lyrics and tags.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/heartmula_oss_3b_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/heartmula_oss_3b_quanto_bf16_int8.safetensors"
9
+ ]
10
+ },
11
+ "prompt": "[Verse]\nMorning light through the window pane\nI hum a tune to chase the rain\nSteady steps on a quiet street\nHeart and rhythm, gentle beat",
12
+ "alt_prompt": "piano,happy,wedding",
13
+ "temperature": 1.0
14
+ }
defaults/heartmula_rl_oss_3b_20260123.json ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "TTS HeartMuLa RL OSS (20260123) 3B",
4
+ "architecture": "heartmula_oss_3b",
5
+ "description": "HeartMuLa RL OSS 3B checkpoint (20260123) with updated codec support. This version should be better at following instructions thanks to a reinforced learning training.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/heartmula_rl_oss_3b_20260123_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/TTS/resolve/main/heartmula_rl_oss_3b_20260123_quanto_bf16_int8.safetensors"
9
+ ],
10
+ "heartmula_codec_version": "20260123"
11
+ },
12
+ "prompt": "[Verse]\nMorning light through the window pane\nI hum a tune to chase the rain\nSteady steps on a quiet street\nHeart and rhythm, gentle beat",
13
+ "alt_prompt": "piano,happy,wedding",
14
+ "temperature": 1.0
15
+ }
defaults/hidream_o1.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": {
3
+ "name": "HiDream O1 Image Full 10B",
4
+ "architecture": "hidream_o1",
5
+ "description": "HiDream O1 Image is an 8B unified image generation model that works directly in a shared text, pixel, and reference-image token space without a separate VAE or text encoder.",
6
+ "URLs": [
7
+ "https://huggingface.co/DeepBeepMeep/HiDream/resolve/main/HiDreamO1Image_bf16.safetensors",
8
+ "https://huggingface.co/DeepBeepMeep/HiDream/resolve/main/HiDreamO1Image_quanto_bf16_int8.safetensors"
9
+ ]
10
+ },
11
+ "prompt": "A tiny porcelain robot arranging wildflowers on a sunlit kitchen table, crisp details, natural colors",
12
+ "resolution": "1920x1088",
13
+ "batch_size": 1,
14
+ "num_inference_steps": 50,
15
+ "guidance_scale": 5,
16
+ "flow_shift": 3,
17
+ "sample_solver": "default"
18
+ }