Nekochu commited on
Commit
d28c9a3
·
verified ·
1 Parent(s): 4d444cd

Side-by-side: Nanosaur2 BF16/INT8 + Supra2-IMG

Browse files
Files changed (3) hide show
  1. README.md +10 -6
  2. app.py +179 -80
  3. requirements.txt +1 -0
README.md CHANGED
@@ -1,5 +1,5 @@
1
  ---
2
- title: Nanosaur2 670M
3
  emoji: 🦖
4
  colorFrom: green
5
  colorTo: yellow
@@ -9,17 +9,21 @@ app_file: app.py
9
  pinned: false
10
  license: mit
11
  hardware: zero-a10g
12
- short_description: Tiny 670M anime/furry text-to-image model (ComfyUI nodes)
13
  python_version: "3.12"
14
  startup_duration_timeout: "30m"
15
  models:
16
  - well9472/Nanosaur2-670M
 
 
17
  ---
18
 
19
- # Nanosaur2 670M
20
 
21
- Demo for [well9472/Nanosaur2-670M](https://huggingface.co/well9472/Nanosaur2-670M): a 670M-parameter illustration text-to-image DiT (Gemma3-270M text encoder, DINOv2 semantic VAE), trained on e621/Danbooru. Tags or natural language.
22
 
23
- The model only ships as ComfyUI custom nodes, so this Space runs ComfyUI as a library (cloned at startup) with the author's `nanosaur2_support` nodes. No weights are stored in this repo: everything is downloaded from the Hub.
 
 
24
 
25
- Defaults follow the model's own workflow: Euler / simple, CFG 4, 30 steps, "alternate" guidance (CFG + path-drop).
 
1
  ---
2
+ title: Small T2I Models Side by Side
3
  emoji: 🦖
4
  colorFrom: green
5
  colorTo: yellow
 
9
  pinned: false
10
  license: mit
11
  hardware: zero-a10g
12
+ short_description: Compare Nanosaur2 (BF16/INT8) and Supra2-IMG
13
  python_version: "3.12"
14
  startup_duration_timeout: "30m"
15
  models:
16
  - well9472/Nanosaur2-670M
17
+ - bertbobson/Nanosaur2-670M-INT8-ConvRot
18
+ - SupraLabs/Supra2-IMG
19
  ---
20
 
21
+ # Small text-to-image models, side by side
22
 
23
+ Runs the same prompt and seed through new tiny text-to-image models:
24
 
25
+ - [well9472/Nanosaur2-670M](https://huggingface.co/well9472/Nanosaur2-670M): 670M illustration DiT (ComfyUI nodes, run here through ComfyUI as a library)
26
+ - [bertbobson/Nanosaur2-670M-INT8-ConvRot](https://huggingface.co/bertbobson/Nanosaur2-670M-INT8-ConvRot): INT8 version of the same model (patched nodes.py)
27
+ - [SupraLabs/Supra2-IMG](https://huggingface.co/SupraLabs/Supra2-IMG): 100M DiT, 256x256
28
 
29
+ No weights are stored in this repo: ComfyUI is cloned and all models are downloaded from the Hub at startup.
app.py CHANGED
@@ -1,22 +1,36 @@
 
1
  import os
2
  import random
 
3
  import subprocess
4
  import sys
5
 
6
  import gradio as gr
7
  import spaces
8
- import torch # noqa: F401 (must be imported before CUDA is touched, as required by ZeroGPU)
9
- from huggingface_hub import snapshot_download
 
10
 
11
  # --------------------------------------------------------------------------------------
12
- # Nanosaur2-670M ships only as ComfyUI custom nodes (nanosaur2_support). To run the exact
13
- # same code path, this Space uses ComfyUI as a library. Nothing is stored in this repo:
14
- # ComfyUI is cloned and the model files are downloaded from the Hub at startup.
 
 
15
  # --------------------------------------------------------------------------------------
16
- MODEL_REPO = "well9472/Nanosaur2-670M"
17
- COMFY_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "ComfyUI")
 
 
 
 
18
  MAX_SEED = 2**31 - 1
19
 
 
 
 
 
 
20
  DEFAULT_PROMPT = (
21
  "newest, masterpiece, 1girl, solo, (fennec ears:1.3), long blonde wavy hair, blue eyes, "
22
  "big fluffy tail, smile, forest, sunlight"
@@ -26,130 +40,215 @@ DEFAULT_NEGATIVE = (
26
  "text, bad anatomy, deformed, extra limbs, missing fingers, cropped"
27
  )
28
 
 
29
  if not os.path.isdir(COMFY_DIR):
30
  subprocess.run(
31
  ["git", "clone", "--depth", "1", "https://github.com/comfyanonymous/ComfyUI", COMFY_DIR],
32
  check=True,
33
  )
34
 
35
- MODEL_DIR = snapshot_download(
36
- MODEL_REPO,
37
- allow_patterns=["*.safetensors", "nanosaur2_support/*.py"],
 
 
 
 
 
 
 
 
 
38
  )
39
 
40
- _runtime = {}
 
 
 
41
 
42
 
43
- def get_runtime():
44
- """Import ComfyUI lazily (inside the GPU context) and register the Nanosaur2 nodes."""
45
- if _runtime:
46
- return _runtime
47
  sys.argv = [sys.argv[0]] # ComfyUI parses sys.argv on import
48
- for p in (COMFY_DIR, MODEL_DIR):
49
  if p not in sys.path:
50
  sys.path.insert(0, p)
51
-
52
  import folder_paths
53
  import nodes
54
 
55
  for folder in ("diffusion_models", "text_encoders", "vae"):
56
- folder_paths.add_model_folder_path(folder, MODEL_DIR)
 
57
 
58
  from nanosaur2_support.nodes import Nanosaur2Loader
 
59
 
60
- _runtime.update(nodes=nodes, Loader=Nanosaur2Loader)
61
- return _runtime
62
 
63
 
64
  def snap(v, m=16):
65
  return max(m, int(round(v / m)) * m)
66
 
67
 
68
- @spaces.GPU(duration=90)
69
- def generate(
70
- prompt,
71
- negative_prompt,
72
- width,
73
- height,
74
- steps,
75
- cfg,
76
- guidance_mode,
77
- seed,
78
- randomize_seed,
79
- progress=gr.Progress(track_tqdm=True),
80
- ):
81
- if not prompt or not prompt.strip():
82
- raise gr.Error("Please enter a prompt.")
83
- if randomize_seed:
84
- seed = random.randint(0, MAX_SEED)
85
- seed = int(seed)
86
-
87
- rt = get_runtime()
88
  nodes = rt["nodes"]
89
- from PIL import Image
90
-
91
  with torch.inference_mode():
92
- model, clip, vae = rt["Loader"]().load(
93
- "nanosaur2_diffusion_model.safetensors",
94
- "nanosaur2_text_encoder.safetensors",
95
- "nanosaur2_vae.safetensors",
96
- guidance_mode,
97
  )
98
- positive = nodes.CLIPTextEncode().encode(clip, prompt)[0]
99
- negative = nodes.CLIPTextEncode().encode(clip, negative_prompt or "")[0]
100
  latent = nodes.EmptyLatentImage().generate(snap(width), snap(height), 1)[0]
101
  samples = nodes.KSampler().sample(
102
- model, seed, int(steps), float(cfg), "euler", "simple", positive, negative, latent, 1.0
103
  )[0]
104
  image = nodes.VAEDecode().decode(vae, samples)[0]
105
-
106
  arr = (image[0].clamp(0, 1).cpu().float().numpy() * 255).round().astype("uint8")
107
- return Image.fromarray(arr), seed
108
-
109
-
110
- CSS = """
111
- #page { max-width: 1100px; margin: 0 auto; }
112
- footer { display: none !important; }
113
- """
114
-
115
- with gr.Blocks(title="Nanosaur2 670M") as demo:
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
116
  with gr.Column(elem_id="page"):
117
  gr.Markdown(
118
- "# Nanosaur2 670M\n"
119
- "Tiny (670M DiT + Gemma3-270M + DINOv2 VAE) illustration text-to-image model trained on "
120
- "e621/Danbooru. Accepts tags or natural language. "
121
- "Model: [well9472/Nanosaur2-670M](https://huggingface.co/well9472/Nanosaur2-670M) (MIT, research use). "
122
- "Tip: start the positive prompt with *newest, masterpiece* and the negative with *oldest, low quality*; "
123
- "upweight with *(tag:1.3)*."
124
  )
125
  with gr.Row(equal_height=False):
126
- with gr.Column():
 
127
  prompt = gr.Textbox(label="Prompt", lines=3, value=DEFAULT_PROMPT)
128
- negative_prompt = gr.Textbox(label="Negative prompt", lines=3, value=DEFAULT_NEGATIVE)
129
  run = gr.Button("Generate", variant="primary", size="lg")
130
- with gr.Accordion("Advanced", open=False):
131
  with gr.Row():
132
  width = gr.Slider(512, 1536, value=832, step=16, label="Width")
133
  height = gr.Slider(512, 1536, value=1216, step=16, label="Height")
134
- steps = gr.Slider(1, 80, value=30, step=1, label="Steps", info="Model card: 30-50 (Euler / simple)")
135
  cfg = gr.Slider(1.0, 12.0, value=4.0, step=0.1, label="CFG scale")
136
- guidance_mode = gr.Radio(
137
- ["alternate", "cfg", "path_drop"],
138
- value="alternate",
139
- label="Guidance mode",
140
- info="alternate = CFG on even steps, path-drop on odd steps (recommended)",
141
- )
142
  seed = gr.Slider(0, MAX_SEED, value=42, step=1, label="Seed")
143
  randomize_seed = gr.Checkbox(value=True, label="Randomize seed")
144
- with gr.Column():
145
- result = gr.Image(label="Result", format="png")
146
  used_seed = gr.Number(label="Seed used", interactive=False)
147
 
148
  gr.on(
149
  [run.click, prompt.submit],
150
  generate,
151
- [prompt, negative_prompt, width, height, steps, cfg, guidance_mode, seed, randomize_seed],
152
- [result, used_seed],
 
153
  )
154
 
155
  if __name__ == "__main__":
 
1
+ import importlib.util
2
  import os
3
  import random
4
+ import shutil
5
  import subprocess
6
  import sys
7
 
8
  import gradio as gr
9
  import spaces
10
+ import torch
11
+ from huggingface_hub import hf_hub_download, snapshot_download
12
+ from PIL import Image
13
 
14
  # --------------------------------------------------------------------------------------
15
+ # Side-by-side comparison of small, new text-to-image models:
16
+ # * Nanosaur2-670M (BF16) - ComfyUI custom nodes (author's code)
17
+ # * Nanosaur2-670M INT8 ConvRot - same model, INT8 quant (needs patched nodes.py)
18
+ # * Supra2-IMG (~100M) - standalone DiT, 256x256
19
+ # Nothing is stored in this repo: ComfyUI is cloned and all weights come from the Hub.
20
  # --------------------------------------------------------------------------------------
21
+ NANO_REPO = "well9472/Nanosaur2-670M"
22
+ INT8_REPO = "bertbobson/Nanosaur2-670M-INT8-ConvRot"
23
+ SUPRA_REPO = "SupraLabs/Supra2-IMG"
24
+ HERE = os.path.dirname(os.path.abspath(__file__))
25
+ COMFY_DIR = os.path.join(HERE, "ComfyUI")
26
+ PKG_DIR = os.path.join(HERE, "pkgs")
27
  MAX_SEED = 2**31 - 1
28
 
29
+ NANO_BF16 = "Nanosaur2 670M (BF16)"
30
+ NANO_INT8 = "Nanosaur2 670M (INT8 ConvRot)"
31
+ SUPRA = "Supra2-IMG (100M, 256px)"
32
+ ALL_MODELS = [NANO_BF16, NANO_INT8, SUPRA]
33
+
34
  DEFAULT_PROMPT = (
35
  "newest, masterpiece, 1girl, solo, (fennec ears:1.3), long blonde wavy hair, blue eyes, "
36
  "big fluffy tail, smile, forest, sunlight"
 
40
  "text, bad anatomy, deformed, extra limbs, missing fingers, cropped"
41
  )
42
 
43
+ # ---------------------------------- startup (no CUDA) ----------------------------------
44
  if not os.path.isdir(COMFY_DIR):
45
  subprocess.run(
46
  ["git", "clone", "--depth", "1", "https://github.com/comfyanonymous/ComfyUI", COMFY_DIR],
47
  check=True,
48
  )
49
 
50
+ NANO_DIR = snapshot_download(NANO_REPO, allow_patterns=["*.safetensors", "nanosaur2_support/*.py"])
51
+ INT8_DIR = snapshot_download(INT8_REPO, allow_patterns=["nanosaur2_int8_conv.safetensors", "nodes.py"])
52
+
53
+ # Two copies of the author's node package: the original, and one with the INT8-patched nodes.py.
54
+ os.makedirs(PKG_DIR, exist_ok=True)
55
+ for pkg in ("nanosaur2_support", "nanosaur2_support_int8"):
56
+ dst = os.path.join(PKG_DIR, pkg)
57
+ if not os.path.isdir(dst):
58
+ shutil.copytree(os.path.join(NANO_DIR, "nanosaur2_support"), dst, symlinks=False)
59
+ shutil.copyfile(
60
+ os.path.join(INT8_DIR, "nodes.py"),
61
+ os.path.join(PKG_DIR, "nanosaur2_support_int8", "nodes.py"),
62
  )
63
 
64
+ SUPRA_CODE = hf_hub_download(SUPRA_REPO, "inference.py")
65
+ SUPRA_CKPT = hf_hub_download(SUPRA_REPO, "model_final_ema.pt")
66
+
67
+ _rt = {}
68
 
69
 
70
+ # ------------------------------------- Nanosaur2 ---------------------------------------
71
+ def get_comfy():
72
+ if "nodes" in _rt:
73
+ return _rt
74
  sys.argv = [sys.argv[0]] # ComfyUI parses sys.argv on import
75
+ for p in (COMFY_DIR, PKG_DIR):
76
  if p not in sys.path:
77
  sys.path.insert(0, p)
 
78
  import folder_paths
79
  import nodes
80
 
81
  for folder in ("diffusion_models", "text_encoders", "vae"):
82
+ folder_paths.add_model_folder_path(folder, NANO_DIR)
83
+ folder_paths.add_model_folder_path("diffusion_models", INT8_DIR)
84
 
85
  from nanosaur2_support.nodes import Nanosaur2Loader
86
+ from nanosaur2_support_int8.nodes import Nanosaur2Loader as Nanosaur2LoaderInt8
87
 
88
+ _rt.update(nodes=nodes, Loader=Nanosaur2Loader, LoaderInt8=Nanosaur2LoaderInt8)
89
+ return _rt
90
 
91
 
92
  def snap(v, m=16):
93
  return max(m, int(round(v / m)) * m)
94
 
95
 
96
+ def run_nanosaur(int8, prompt, negative, width, height, steps, cfg, guidance_mode, seed):
97
+ rt = get_comfy()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
98
  nodes = rt["nodes"]
99
+ loader = rt["LoaderInt8"] if int8 else rt["Loader"]
100
+ unet = "nanosaur2_int8_conv.safetensors" if int8 else "nanosaur2_diffusion_model.safetensors"
101
  with torch.inference_mode():
102
+ model, clip, vae = loader().load(
103
+ unet, "nanosaur2_text_encoder.safetensors", "nanosaur2_vae.safetensors", guidance_mode
 
 
 
104
  )
105
+ pos = nodes.CLIPTextEncode().encode(clip, prompt)[0]
106
+ neg = nodes.CLIPTextEncode().encode(clip, negative or "")[0]
107
  latent = nodes.EmptyLatentImage().generate(snap(width), snap(height), 1)[0]
108
  samples = nodes.KSampler().sample(
109
+ model, seed, int(steps), float(cfg), "euler", "simple", pos, neg, latent, 1.0
110
  )[0]
111
  image = nodes.VAEDecode().decode(vae, samples)[0]
 
112
  arr = (image[0].clamp(0, 1).cpu().float().numpy() * 255).round().astype("uint8")
113
+ try:
114
+ import comfy.model_management as mm
115
+
116
+ mm.unload_all_models()
117
+ mm.soft_empty_cache()
118
+ except Exception:
119
+ pass
120
+ return Image.fromarray(arr)
121
+
122
+
123
+ # -------------------------------------- Supra2 -----------------------------------------
124
+ def run_supra(prompt, cfg, steps, seed):
125
+ from diffusers import AutoencoderKL
126
+ from transformers import AutoTokenizer, T5EncoderModel
127
+
128
+ spec = importlib.util.spec_from_file_location("supra_inference", SUPRA_CODE)
129
+ sup = importlib.util.module_from_spec(spec)
130
+ spec.loader.exec_module(sup)
131
+
132
+ dev = torch.device("cuda")
133
+ state = torch.load(SUPRA_CKPT, map_location=dev, weights_only=False)
134
+ scfg = state.get("config", {}) if isinstance(state, dict) else {}
135
+ weights = state["ema"] if isinstance(state, dict) and "ema" in state else state.get("model", state)
136
+ model = sup.SupraDiT().to(dev).eval()
137
+ model.load_state_dict(weights, strict=True)
138
+ ctx_len = int(scfg.get("ctx_len", sup.MAX_CTX_LEN))
139
+
140
+ tokenizer = AutoTokenizer.from_pretrained(sup.T5_NAME)
141
+ text_model = T5EncoderModel.from_pretrained(sup.T5_NAME).to(dev).eval()
142
+ vae = AutoencoderKL.from_pretrained(sup.VAE_NAME).to(dev).eval()
143
+
144
+ def encode(texts):
145
+ tok = tokenizer(texts, padding="max_length", truncation=True, max_length=ctx_len, return_tensors="pt").to(dev)
146
+ with torch.autocast("cuda", dtype=torch.bfloat16):
147
+ ctx = text_model(**tok).last_hidden_state.float()
148
+ return ctx, tok["attention_mask"].float()
149
+
150
+ with torch.no_grad():
151
+ ctx, cmask = encode([prompt])
152
+ use_cfg = cfg > 1.0
153
+ if use_cfg:
154
+ if "uncond_text" in scfg:
155
+ uctx = scfg["uncond_text"].to(dev).float().unsqueeze(0)
156
+ umask = scfg["uncond_mask"].to(dev).float().unsqueeze(0)
157
+ else:
158
+ uctx, umask = encode([""])
159
+ ctx_all, mask_all = torch.cat([ctx, uctx], 0), torch.cat([cmask, umask], 0)
160
+ torch.manual_seed(seed)
161
+ z = torch.randn(1, sup.LATENT_CH, sup.LATENT_SIZE, sup.LATENT_SIZE, device=dev)
162
+ dt = 1.0 / steps
163
+ for i in range(int(steps)):
164
+ t = torch.full((1,), i * dt, device=dev)
165
+ with torch.autocast("cuda", dtype=torch.bfloat16):
166
+ if use_cfg:
167
+ v_both = model(torch.cat([z, z], 0), torch.cat([t, t], 0), ctx_all, mask_all)
168
+ v_c, v_u = v_both.float().chunk(2, 0)
169
+ v = v_u + cfg * (v_c - v_u)
170
+ else:
171
+ v = model(z, t, ctx, cmask).float()
172
+ z = z + dt * v
173
+ with torch.autocast("cuda", dtype=torch.bfloat16):
174
+ img = vae.decode(z / sup.VAE_SCALE).sample
175
+ img = ((img.clamp(-1, 1) + 1) / 2)[0].permute(1, 2, 0).float().cpu().numpy()
176
+ return Image.fromarray((img * 255).round().astype("uint8"))
177
+
178
+
179
+ # --------------------------------------- UI --------------------------------------------
180
+ @spaces.GPU(duration=120)
181
+ def generate(
182
+ models, prompt, negative_prompt, width, height, steps, cfg, guidance_mode,
183
+ supra_steps, supra_cfg, seed, randomize_seed,
184
+ progress=gr.Progress(track_tqdm=True),
185
+ ):
186
+ if not prompt or not prompt.strip():
187
+ raise gr.Error("Please enter a prompt.")
188
+ if not models:
189
+ raise gr.Error("Select at least one model.")
190
+ if randomize_seed:
191
+ seed = random.randint(0, MAX_SEED)
192
+ seed = int(seed)
193
+ results = []
194
+ for name in ALL_MODELS:
195
+ if name not in models:
196
+ continue
197
+ try:
198
+ if name == SUPRA:
199
+ img = run_supra(prompt, float(supra_cfg), int(supra_steps), seed)
200
+ else:
201
+ img = run_nanosaur(name == NANO_INT8, prompt, negative_prompt, width, height,
202
+ steps, cfg, guidance_mode, seed)
203
+ results.append((img, name))
204
+ except Exception as e: # keep the other models' results
205
+ gr.Warning(f"{name} failed: {type(e).__name__}: {e}")
206
+ if not results:
207
+ raise gr.Error("All selected models failed, see the logs.")
208
+ return results, seed
209
+
210
+
211
+ CSS = "#page { max-width: 1200px; margin: 0 auto; } footer { display: none !important; }"
212
+
213
+ with gr.Blocks(title="Small T2I models side by side") as demo:
214
  with gr.Column(elem_id="page"):
215
  gr.Markdown(
216
+ "# Small text-to-image models, side by side\n"
217
+ "Same prompt and seed on new tiny models: "
218
+ "[Nanosaur2-670M](https://huggingface.co/well9472/Nanosaur2-670M) (anime/furry, BF16 and "
219
+ "[INT8](https://huggingface.co/bertbobson/Nanosaur2-670M-INT8-ConvRot)) and "
220
+ "[Supra2-IMG](https://huggingface.co/SupraLabs/Supra2-IMG) (100M, 256px). "
221
+ "Nanosaur2 likes tags: start with *newest, masterpiece*, negative with *oldest, low quality*."
222
  )
223
  with gr.Row(equal_height=False):
224
+ with gr.Column(scale=2):
225
+ models = gr.CheckboxGroup(ALL_MODELS, value=[NANO_BF16, NANO_INT8, SUPRA], label="Models")
226
  prompt = gr.Textbox(label="Prompt", lines=3, value=DEFAULT_PROMPT)
227
+ negative_prompt = gr.Textbox(label="Negative prompt (Nanosaur2 only)", lines=2, value=DEFAULT_NEGATIVE)
228
  run = gr.Button("Generate", variant="primary", size="lg")
229
+ with gr.Accordion("Nanosaur2 settings", open=False):
230
  with gr.Row():
231
  width = gr.Slider(512, 1536, value=832, step=16, label="Width")
232
  height = gr.Slider(512, 1536, value=1216, step=16, label="Height")
233
+ steps = gr.Slider(1, 80, value=30, step=1, label="Steps")
234
  cfg = gr.Slider(1.0, 12.0, value=4.0, step=0.1, label="CFG scale")
235
+ guidance_mode = gr.Radio(["alternate", "cfg", "path_drop"], value="alternate", label="Guidance mode")
236
+ with gr.Accordion("Supra2-IMG settings", open=False):
237
+ supra_steps = gr.Slider(1, 100, value=50, step=1, label="Steps")
238
+ supra_cfg = gr.Slider(1.0, 10.0, value=3.0, step=0.1, label="CFG scale")
239
+ with gr.Accordion("Seed", open=False):
 
240
  seed = gr.Slider(0, MAX_SEED, value=42, step=1, label="Seed")
241
  randomize_seed = gr.Checkbox(value=True, label="Randomize seed")
242
+ with gr.Column(scale=3):
243
+ gallery = gr.Gallery(label="Results", columns=3, height=620, object_fit="contain", format="png")
244
  used_seed = gr.Number(label="Seed used", interactive=False)
245
 
246
  gr.on(
247
  [run.click, prompt.submit],
248
  generate,
249
+ [models, prompt, negative_prompt, width, height, steps, cfg, guidance_mode,
250
+ supra_steps, supra_cfg, seed, randomize_seed],
251
+ [gallery, used_seed],
252
  )
253
 
254
  if __name__ == "__main__":
requirements.txt CHANGED
@@ -1,4 +1,5 @@
1
  torch==2.8.0
2
  torchvision==0.23.0
3
  torchaudio==2.8.0
 
4
  -r https://raw.githubusercontent.com/comfyanonymous/ComfyUI/master/requirements.txt
 
1
  torch==2.8.0
2
  torchvision==0.23.0
3
  torchaudio==2.8.0
4
+ diffusers
5
  -r https://raw.githubusercontent.com/comfyanonymous/ComfyUI/master/requirements.txt