Side-by-side: Nanosaur2 BF16/INT8 + Supra2-IMG
Browse files- README.md +10 -6
- app.py +179 -80
- requirements.txt +1 -0
README.md
CHANGED
|
@@ -1,5 +1,5 @@
|
|
| 1 |
---
|
| 2 |
-
title:
|
| 3 |
emoji: 🦖
|
| 4 |
colorFrom: green
|
| 5 |
colorTo: yellow
|
|
@@ -9,17 +9,21 @@ app_file: app.py
|
|
| 9 |
pinned: false
|
| 10 |
license: mit
|
| 11 |
hardware: zero-a10g
|
| 12 |
-
short_description:
|
| 13 |
python_version: "3.12"
|
| 14 |
startup_duration_timeout: "30m"
|
| 15 |
models:
|
| 16 |
- well9472/Nanosaur2-670M
|
|
|
|
|
|
|
| 17 |
---
|
| 18 |
|
| 19 |
-
#
|
| 20 |
|
| 21 |
-
|
| 22 |
|
| 23 |
-
|
|
|
|
|
|
|
| 24 |
|
| 25 |
-
|
|
|
|
| 1 |
---
|
| 2 |
+
title: Small T2I Models Side by Side
|
| 3 |
emoji: 🦖
|
| 4 |
colorFrom: green
|
| 5 |
colorTo: yellow
|
|
|
|
| 9 |
pinned: false
|
| 10 |
license: mit
|
| 11 |
hardware: zero-a10g
|
| 12 |
+
short_description: Compare Nanosaur2 (BF16/INT8) and Supra2-IMG
|
| 13 |
python_version: "3.12"
|
| 14 |
startup_duration_timeout: "30m"
|
| 15 |
models:
|
| 16 |
- well9472/Nanosaur2-670M
|
| 17 |
+
- bertbobson/Nanosaur2-670M-INT8-ConvRot
|
| 18 |
+
- SupraLabs/Supra2-IMG
|
| 19 |
---
|
| 20 |
|
| 21 |
+
# Small text-to-image models, side by side
|
| 22 |
|
| 23 |
+
Runs the same prompt and seed through new tiny text-to-image models:
|
| 24 |
|
| 25 |
+
- [well9472/Nanosaur2-670M](https://huggingface.co/well9472/Nanosaur2-670M): 670M illustration DiT (ComfyUI nodes, run here through ComfyUI as a library)
|
| 26 |
+
- [bertbobson/Nanosaur2-670M-INT8-ConvRot](https://huggingface.co/bertbobson/Nanosaur2-670M-INT8-ConvRot): INT8 version of the same model (patched nodes.py)
|
| 27 |
+
- [SupraLabs/Supra2-IMG](https://huggingface.co/SupraLabs/Supra2-IMG): 100M DiT, 256x256
|
| 28 |
|
| 29 |
+
No weights are stored in this repo: ComfyUI is cloned and all models are downloaded from the Hub at startup.
|
app.py
CHANGED
|
@@ -1,22 +1,36 @@
|
|
|
|
|
| 1 |
import os
|
| 2 |
import random
|
|
|
|
| 3 |
import subprocess
|
| 4 |
import sys
|
| 5 |
|
| 6 |
import gradio as gr
|
| 7 |
import spaces
|
| 8 |
-
import torch
|
| 9 |
-
from huggingface_hub import snapshot_download
|
|
|
|
| 10 |
|
| 11 |
# --------------------------------------------------------------------------------------
|
| 12 |
-
#
|
| 13 |
-
#
|
| 14 |
-
#
|
|
|
|
|
|
|
| 15 |
# --------------------------------------------------------------------------------------
|
| 16 |
-
|
| 17 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 18 |
MAX_SEED = 2**31 - 1
|
| 19 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 20 |
DEFAULT_PROMPT = (
|
| 21 |
"newest, masterpiece, 1girl, solo, (fennec ears:1.3), long blonde wavy hair, blue eyes, "
|
| 22 |
"big fluffy tail, smile, forest, sunlight"
|
|
@@ -26,130 +40,215 @@ DEFAULT_NEGATIVE = (
|
|
| 26 |
"text, bad anatomy, deformed, extra limbs, missing fingers, cropped"
|
| 27 |
)
|
| 28 |
|
|
|
|
| 29 |
if not os.path.isdir(COMFY_DIR):
|
| 30 |
subprocess.run(
|
| 31 |
["git", "clone", "--depth", "1", "https://github.com/comfyanonymous/ComfyUI", COMFY_DIR],
|
| 32 |
check=True,
|
| 33 |
)
|
| 34 |
|
| 35 |
-
|
| 36 |
-
|
| 37 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 38 |
)
|
| 39 |
|
| 40 |
-
|
|
|
|
|
|
|
|
|
|
| 41 |
|
| 42 |
|
| 43 |
-
|
| 44 |
-
|
| 45 |
-
if
|
| 46 |
-
return
|
| 47 |
sys.argv = [sys.argv[0]] # ComfyUI parses sys.argv on import
|
| 48 |
-
for p in (COMFY_DIR,
|
| 49 |
if p not in sys.path:
|
| 50 |
sys.path.insert(0, p)
|
| 51 |
-
|
| 52 |
import folder_paths
|
| 53 |
import nodes
|
| 54 |
|
| 55 |
for folder in ("diffusion_models", "text_encoders", "vae"):
|
| 56 |
-
folder_paths.add_model_folder_path(folder,
|
|
|
|
| 57 |
|
| 58 |
from nanosaur2_support.nodes import Nanosaur2Loader
|
|
|
|
| 59 |
|
| 60 |
-
|
| 61 |
-
return
|
| 62 |
|
| 63 |
|
| 64 |
def snap(v, m=16):
|
| 65 |
return max(m, int(round(v / m)) * m)
|
| 66 |
|
| 67 |
|
| 68 |
-
|
| 69 |
-
|
| 70 |
-
prompt,
|
| 71 |
-
negative_prompt,
|
| 72 |
-
width,
|
| 73 |
-
height,
|
| 74 |
-
steps,
|
| 75 |
-
cfg,
|
| 76 |
-
guidance_mode,
|
| 77 |
-
seed,
|
| 78 |
-
randomize_seed,
|
| 79 |
-
progress=gr.Progress(track_tqdm=True),
|
| 80 |
-
):
|
| 81 |
-
if not prompt or not prompt.strip():
|
| 82 |
-
raise gr.Error("Please enter a prompt.")
|
| 83 |
-
if randomize_seed:
|
| 84 |
-
seed = random.randint(0, MAX_SEED)
|
| 85 |
-
seed = int(seed)
|
| 86 |
-
|
| 87 |
-
rt = get_runtime()
|
| 88 |
nodes = rt["nodes"]
|
| 89 |
-
|
| 90 |
-
|
| 91 |
with torch.inference_mode():
|
| 92 |
-
model, clip, vae =
|
| 93 |
-
"
|
| 94 |
-
"nanosaur2_text_encoder.safetensors",
|
| 95 |
-
"nanosaur2_vae.safetensors",
|
| 96 |
-
guidance_mode,
|
| 97 |
)
|
| 98 |
-
|
| 99 |
-
|
| 100 |
latent = nodes.EmptyLatentImage().generate(snap(width), snap(height), 1)[0]
|
| 101 |
samples = nodes.KSampler().sample(
|
| 102 |
-
model, seed, int(steps), float(cfg), "euler", "simple",
|
| 103 |
)[0]
|
| 104 |
image = nodes.VAEDecode().decode(vae, samples)[0]
|
| 105 |
-
|
| 106 |
arr = (image[0].clamp(0, 1).cpu().float().numpy() * 255).round().astype("uint8")
|
| 107 |
-
|
| 108 |
-
|
| 109 |
-
|
| 110 |
-
|
| 111 |
-
|
| 112 |
-
|
| 113 |
-
|
| 114 |
-
|
| 115 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 116 |
with gr.Column(elem_id="page"):
|
| 117 |
gr.Markdown(
|
| 118 |
-
"#
|
| 119 |
-
"
|
| 120 |
-
"
|
| 121 |
-
"
|
| 122 |
-
"
|
| 123 |
-
"
|
| 124 |
)
|
| 125 |
with gr.Row(equal_height=False):
|
| 126 |
-
with gr.Column():
|
|
|
|
| 127 |
prompt = gr.Textbox(label="Prompt", lines=3, value=DEFAULT_PROMPT)
|
| 128 |
-
negative_prompt = gr.Textbox(label="Negative prompt", lines=
|
| 129 |
run = gr.Button("Generate", variant="primary", size="lg")
|
| 130 |
-
with gr.Accordion("
|
| 131 |
with gr.Row():
|
| 132 |
width = gr.Slider(512, 1536, value=832, step=16, label="Width")
|
| 133 |
height = gr.Slider(512, 1536, value=1216, step=16, label="Height")
|
| 134 |
-
steps = gr.Slider(1, 80, value=30, step=1, label="Steps"
|
| 135 |
cfg = gr.Slider(1.0, 12.0, value=4.0, step=0.1, label="CFG scale")
|
| 136 |
-
guidance_mode = gr.Radio(
|
| 137 |
-
|
| 138 |
-
|
| 139 |
-
|
| 140 |
-
|
| 141 |
-
)
|
| 142 |
seed = gr.Slider(0, MAX_SEED, value=42, step=1, label="Seed")
|
| 143 |
randomize_seed = gr.Checkbox(value=True, label="Randomize seed")
|
| 144 |
-
with gr.Column():
|
| 145 |
-
|
| 146 |
used_seed = gr.Number(label="Seed used", interactive=False)
|
| 147 |
|
| 148 |
gr.on(
|
| 149 |
[run.click, prompt.submit],
|
| 150 |
generate,
|
| 151 |
-
[prompt, negative_prompt, width, height, steps, cfg, guidance_mode,
|
| 152 |
-
|
|
|
|
| 153 |
)
|
| 154 |
|
| 155 |
if __name__ == "__main__":
|
|
|
|
| 1 |
+
import importlib.util
|
| 2 |
import os
|
| 3 |
import random
|
| 4 |
+
import shutil
|
| 5 |
import subprocess
|
| 6 |
import sys
|
| 7 |
|
| 8 |
import gradio as gr
|
| 9 |
import spaces
|
| 10 |
+
import torch
|
| 11 |
+
from huggingface_hub import hf_hub_download, snapshot_download
|
| 12 |
+
from PIL import Image
|
| 13 |
|
| 14 |
# --------------------------------------------------------------------------------------
|
| 15 |
+
# Side-by-side comparison of small, new text-to-image models:
|
| 16 |
+
# * Nanosaur2-670M (BF16) - ComfyUI custom nodes (author's code)
|
| 17 |
+
# * Nanosaur2-670M INT8 ConvRot - same model, INT8 quant (needs patched nodes.py)
|
| 18 |
+
# * Supra2-IMG (~100M) - standalone DiT, 256x256
|
| 19 |
+
# Nothing is stored in this repo: ComfyUI is cloned and all weights come from the Hub.
|
| 20 |
# --------------------------------------------------------------------------------------
|
| 21 |
+
NANO_REPO = "well9472/Nanosaur2-670M"
|
| 22 |
+
INT8_REPO = "bertbobson/Nanosaur2-670M-INT8-ConvRot"
|
| 23 |
+
SUPRA_REPO = "SupraLabs/Supra2-IMG"
|
| 24 |
+
HERE = os.path.dirname(os.path.abspath(__file__))
|
| 25 |
+
COMFY_DIR = os.path.join(HERE, "ComfyUI")
|
| 26 |
+
PKG_DIR = os.path.join(HERE, "pkgs")
|
| 27 |
MAX_SEED = 2**31 - 1
|
| 28 |
|
| 29 |
+
NANO_BF16 = "Nanosaur2 670M (BF16)"
|
| 30 |
+
NANO_INT8 = "Nanosaur2 670M (INT8 ConvRot)"
|
| 31 |
+
SUPRA = "Supra2-IMG (100M, 256px)"
|
| 32 |
+
ALL_MODELS = [NANO_BF16, NANO_INT8, SUPRA]
|
| 33 |
+
|
| 34 |
DEFAULT_PROMPT = (
|
| 35 |
"newest, masterpiece, 1girl, solo, (fennec ears:1.3), long blonde wavy hair, blue eyes, "
|
| 36 |
"big fluffy tail, smile, forest, sunlight"
|
|
|
|
| 40 |
"text, bad anatomy, deformed, extra limbs, missing fingers, cropped"
|
| 41 |
)
|
| 42 |
|
| 43 |
+
# ---------------------------------- startup (no CUDA) ----------------------------------
|
| 44 |
if not os.path.isdir(COMFY_DIR):
|
| 45 |
subprocess.run(
|
| 46 |
["git", "clone", "--depth", "1", "https://github.com/comfyanonymous/ComfyUI", COMFY_DIR],
|
| 47 |
check=True,
|
| 48 |
)
|
| 49 |
|
| 50 |
+
NANO_DIR = snapshot_download(NANO_REPO, allow_patterns=["*.safetensors", "nanosaur2_support/*.py"])
|
| 51 |
+
INT8_DIR = snapshot_download(INT8_REPO, allow_patterns=["nanosaur2_int8_conv.safetensors", "nodes.py"])
|
| 52 |
+
|
| 53 |
+
# Two copies of the author's node package: the original, and one with the INT8-patched nodes.py.
|
| 54 |
+
os.makedirs(PKG_DIR, exist_ok=True)
|
| 55 |
+
for pkg in ("nanosaur2_support", "nanosaur2_support_int8"):
|
| 56 |
+
dst = os.path.join(PKG_DIR, pkg)
|
| 57 |
+
if not os.path.isdir(dst):
|
| 58 |
+
shutil.copytree(os.path.join(NANO_DIR, "nanosaur2_support"), dst, symlinks=False)
|
| 59 |
+
shutil.copyfile(
|
| 60 |
+
os.path.join(INT8_DIR, "nodes.py"),
|
| 61 |
+
os.path.join(PKG_DIR, "nanosaur2_support_int8", "nodes.py"),
|
| 62 |
)
|
| 63 |
|
| 64 |
+
SUPRA_CODE = hf_hub_download(SUPRA_REPO, "inference.py")
|
| 65 |
+
SUPRA_CKPT = hf_hub_download(SUPRA_REPO, "model_final_ema.pt")
|
| 66 |
+
|
| 67 |
+
_rt = {}
|
| 68 |
|
| 69 |
|
| 70 |
+
# ------------------------------------- Nanosaur2 ---------------------------------------
|
| 71 |
+
def get_comfy():
|
| 72 |
+
if "nodes" in _rt:
|
| 73 |
+
return _rt
|
| 74 |
sys.argv = [sys.argv[0]] # ComfyUI parses sys.argv on import
|
| 75 |
+
for p in (COMFY_DIR, PKG_DIR):
|
| 76 |
if p not in sys.path:
|
| 77 |
sys.path.insert(0, p)
|
|
|
|
| 78 |
import folder_paths
|
| 79 |
import nodes
|
| 80 |
|
| 81 |
for folder in ("diffusion_models", "text_encoders", "vae"):
|
| 82 |
+
folder_paths.add_model_folder_path(folder, NANO_DIR)
|
| 83 |
+
folder_paths.add_model_folder_path("diffusion_models", INT8_DIR)
|
| 84 |
|
| 85 |
from nanosaur2_support.nodes import Nanosaur2Loader
|
| 86 |
+
from nanosaur2_support_int8.nodes import Nanosaur2Loader as Nanosaur2LoaderInt8
|
| 87 |
|
| 88 |
+
_rt.update(nodes=nodes, Loader=Nanosaur2Loader, LoaderInt8=Nanosaur2LoaderInt8)
|
| 89 |
+
return _rt
|
| 90 |
|
| 91 |
|
| 92 |
def snap(v, m=16):
|
| 93 |
return max(m, int(round(v / m)) * m)
|
| 94 |
|
| 95 |
|
| 96 |
+
def run_nanosaur(int8, prompt, negative, width, height, steps, cfg, guidance_mode, seed):
|
| 97 |
+
rt = get_comfy()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 98 |
nodes = rt["nodes"]
|
| 99 |
+
loader = rt["LoaderInt8"] if int8 else rt["Loader"]
|
| 100 |
+
unet = "nanosaur2_int8_conv.safetensors" if int8 else "nanosaur2_diffusion_model.safetensors"
|
| 101 |
with torch.inference_mode():
|
| 102 |
+
model, clip, vae = loader().load(
|
| 103 |
+
unet, "nanosaur2_text_encoder.safetensors", "nanosaur2_vae.safetensors", guidance_mode
|
|
|
|
|
|
|
|
|
|
| 104 |
)
|
| 105 |
+
pos = nodes.CLIPTextEncode().encode(clip, prompt)[0]
|
| 106 |
+
neg = nodes.CLIPTextEncode().encode(clip, negative or "")[0]
|
| 107 |
latent = nodes.EmptyLatentImage().generate(snap(width), snap(height), 1)[0]
|
| 108 |
samples = nodes.KSampler().sample(
|
| 109 |
+
model, seed, int(steps), float(cfg), "euler", "simple", pos, neg, latent, 1.0
|
| 110 |
)[0]
|
| 111 |
image = nodes.VAEDecode().decode(vae, samples)[0]
|
|
|
|
| 112 |
arr = (image[0].clamp(0, 1).cpu().float().numpy() * 255).round().astype("uint8")
|
| 113 |
+
try:
|
| 114 |
+
import comfy.model_management as mm
|
| 115 |
+
|
| 116 |
+
mm.unload_all_models()
|
| 117 |
+
mm.soft_empty_cache()
|
| 118 |
+
except Exception:
|
| 119 |
+
pass
|
| 120 |
+
return Image.fromarray(arr)
|
| 121 |
+
|
| 122 |
+
|
| 123 |
+
# -------------------------------------- Supra2 -----------------------------------------
|
| 124 |
+
def run_supra(prompt, cfg, steps, seed):
|
| 125 |
+
from diffusers import AutoencoderKL
|
| 126 |
+
from transformers import AutoTokenizer, T5EncoderModel
|
| 127 |
+
|
| 128 |
+
spec = importlib.util.spec_from_file_location("supra_inference", SUPRA_CODE)
|
| 129 |
+
sup = importlib.util.module_from_spec(spec)
|
| 130 |
+
spec.loader.exec_module(sup)
|
| 131 |
+
|
| 132 |
+
dev = torch.device("cuda")
|
| 133 |
+
state = torch.load(SUPRA_CKPT, map_location=dev, weights_only=False)
|
| 134 |
+
scfg = state.get("config", {}) if isinstance(state, dict) else {}
|
| 135 |
+
weights = state["ema"] if isinstance(state, dict) and "ema" in state else state.get("model", state)
|
| 136 |
+
model = sup.SupraDiT().to(dev).eval()
|
| 137 |
+
model.load_state_dict(weights, strict=True)
|
| 138 |
+
ctx_len = int(scfg.get("ctx_len", sup.MAX_CTX_LEN))
|
| 139 |
+
|
| 140 |
+
tokenizer = AutoTokenizer.from_pretrained(sup.T5_NAME)
|
| 141 |
+
text_model = T5EncoderModel.from_pretrained(sup.T5_NAME).to(dev).eval()
|
| 142 |
+
vae = AutoencoderKL.from_pretrained(sup.VAE_NAME).to(dev).eval()
|
| 143 |
+
|
| 144 |
+
def encode(texts):
|
| 145 |
+
tok = tokenizer(texts, padding="max_length", truncation=True, max_length=ctx_len, return_tensors="pt").to(dev)
|
| 146 |
+
with torch.autocast("cuda", dtype=torch.bfloat16):
|
| 147 |
+
ctx = text_model(**tok).last_hidden_state.float()
|
| 148 |
+
return ctx, tok["attention_mask"].float()
|
| 149 |
+
|
| 150 |
+
with torch.no_grad():
|
| 151 |
+
ctx, cmask = encode([prompt])
|
| 152 |
+
use_cfg = cfg > 1.0
|
| 153 |
+
if use_cfg:
|
| 154 |
+
if "uncond_text" in scfg:
|
| 155 |
+
uctx = scfg["uncond_text"].to(dev).float().unsqueeze(0)
|
| 156 |
+
umask = scfg["uncond_mask"].to(dev).float().unsqueeze(0)
|
| 157 |
+
else:
|
| 158 |
+
uctx, umask = encode([""])
|
| 159 |
+
ctx_all, mask_all = torch.cat([ctx, uctx], 0), torch.cat([cmask, umask], 0)
|
| 160 |
+
torch.manual_seed(seed)
|
| 161 |
+
z = torch.randn(1, sup.LATENT_CH, sup.LATENT_SIZE, sup.LATENT_SIZE, device=dev)
|
| 162 |
+
dt = 1.0 / steps
|
| 163 |
+
for i in range(int(steps)):
|
| 164 |
+
t = torch.full((1,), i * dt, device=dev)
|
| 165 |
+
with torch.autocast("cuda", dtype=torch.bfloat16):
|
| 166 |
+
if use_cfg:
|
| 167 |
+
v_both = model(torch.cat([z, z], 0), torch.cat([t, t], 0), ctx_all, mask_all)
|
| 168 |
+
v_c, v_u = v_both.float().chunk(2, 0)
|
| 169 |
+
v = v_u + cfg * (v_c - v_u)
|
| 170 |
+
else:
|
| 171 |
+
v = model(z, t, ctx, cmask).float()
|
| 172 |
+
z = z + dt * v
|
| 173 |
+
with torch.autocast("cuda", dtype=torch.bfloat16):
|
| 174 |
+
img = vae.decode(z / sup.VAE_SCALE).sample
|
| 175 |
+
img = ((img.clamp(-1, 1) + 1) / 2)[0].permute(1, 2, 0).float().cpu().numpy()
|
| 176 |
+
return Image.fromarray((img * 255).round().astype("uint8"))
|
| 177 |
+
|
| 178 |
+
|
| 179 |
+
# --------------------------------------- UI --------------------------------------------
|
| 180 |
+
@spaces.GPU(duration=120)
|
| 181 |
+
def generate(
|
| 182 |
+
models, prompt, negative_prompt, width, height, steps, cfg, guidance_mode,
|
| 183 |
+
supra_steps, supra_cfg, seed, randomize_seed,
|
| 184 |
+
progress=gr.Progress(track_tqdm=True),
|
| 185 |
+
):
|
| 186 |
+
if not prompt or not prompt.strip():
|
| 187 |
+
raise gr.Error("Please enter a prompt.")
|
| 188 |
+
if not models:
|
| 189 |
+
raise gr.Error("Select at least one model.")
|
| 190 |
+
if randomize_seed:
|
| 191 |
+
seed = random.randint(0, MAX_SEED)
|
| 192 |
+
seed = int(seed)
|
| 193 |
+
results = []
|
| 194 |
+
for name in ALL_MODELS:
|
| 195 |
+
if name not in models:
|
| 196 |
+
continue
|
| 197 |
+
try:
|
| 198 |
+
if name == SUPRA:
|
| 199 |
+
img = run_supra(prompt, float(supra_cfg), int(supra_steps), seed)
|
| 200 |
+
else:
|
| 201 |
+
img = run_nanosaur(name == NANO_INT8, prompt, negative_prompt, width, height,
|
| 202 |
+
steps, cfg, guidance_mode, seed)
|
| 203 |
+
results.append((img, name))
|
| 204 |
+
except Exception as e: # keep the other models' results
|
| 205 |
+
gr.Warning(f"{name} failed: {type(e).__name__}: {e}")
|
| 206 |
+
if not results:
|
| 207 |
+
raise gr.Error("All selected models failed, see the logs.")
|
| 208 |
+
return results, seed
|
| 209 |
+
|
| 210 |
+
|
| 211 |
+
CSS = "#page { max-width: 1200px; margin: 0 auto; } footer { display: none !important; }"
|
| 212 |
+
|
| 213 |
+
with gr.Blocks(title="Small T2I models side by side") as demo:
|
| 214 |
with gr.Column(elem_id="page"):
|
| 215 |
gr.Markdown(
|
| 216 |
+
"# Small text-to-image models, side by side\n"
|
| 217 |
+
"Same prompt and seed on new tiny models: "
|
| 218 |
+
"[Nanosaur2-670M](https://huggingface.co/well9472/Nanosaur2-670M) (anime/furry, BF16 and "
|
| 219 |
+
"[INT8](https://huggingface.co/bertbobson/Nanosaur2-670M-INT8-ConvRot)) and "
|
| 220 |
+
"[Supra2-IMG](https://huggingface.co/SupraLabs/Supra2-IMG) (100M, 256px). "
|
| 221 |
+
"Nanosaur2 likes tags: start with *newest, masterpiece*, negative with *oldest, low quality*."
|
| 222 |
)
|
| 223 |
with gr.Row(equal_height=False):
|
| 224 |
+
with gr.Column(scale=2):
|
| 225 |
+
models = gr.CheckboxGroup(ALL_MODELS, value=[NANO_BF16, NANO_INT8, SUPRA], label="Models")
|
| 226 |
prompt = gr.Textbox(label="Prompt", lines=3, value=DEFAULT_PROMPT)
|
| 227 |
+
negative_prompt = gr.Textbox(label="Negative prompt (Nanosaur2 only)", lines=2, value=DEFAULT_NEGATIVE)
|
| 228 |
run = gr.Button("Generate", variant="primary", size="lg")
|
| 229 |
+
with gr.Accordion("Nanosaur2 settings", open=False):
|
| 230 |
with gr.Row():
|
| 231 |
width = gr.Slider(512, 1536, value=832, step=16, label="Width")
|
| 232 |
height = gr.Slider(512, 1536, value=1216, step=16, label="Height")
|
| 233 |
+
steps = gr.Slider(1, 80, value=30, step=1, label="Steps")
|
| 234 |
cfg = gr.Slider(1.0, 12.0, value=4.0, step=0.1, label="CFG scale")
|
| 235 |
+
guidance_mode = gr.Radio(["alternate", "cfg", "path_drop"], value="alternate", label="Guidance mode")
|
| 236 |
+
with gr.Accordion("Supra2-IMG settings", open=False):
|
| 237 |
+
supra_steps = gr.Slider(1, 100, value=50, step=1, label="Steps")
|
| 238 |
+
supra_cfg = gr.Slider(1.0, 10.0, value=3.0, step=0.1, label="CFG scale")
|
| 239 |
+
with gr.Accordion("Seed", open=False):
|
|
|
|
| 240 |
seed = gr.Slider(0, MAX_SEED, value=42, step=1, label="Seed")
|
| 241 |
randomize_seed = gr.Checkbox(value=True, label="Randomize seed")
|
| 242 |
+
with gr.Column(scale=3):
|
| 243 |
+
gallery = gr.Gallery(label="Results", columns=3, height=620, object_fit="contain", format="png")
|
| 244 |
used_seed = gr.Number(label="Seed used", interactive=False)
|
| 245 |
|
| 246 |
gr.on(
|
| 247 |
[run.click, prompt.submit],
|
| 248 |
generate,
|
| 249 |
+
[models, prompt, negative_prompt, width, height, steps, cfg, guidance_mode,
|
| 250 |
+
supra_steps, supra_cfg, seed, randomize_seed],
|
| 251 |
+
[gallery, used_seed],
|
| 252 |
)
|
| 253 |
|
| 254 |
if __name__ == "__main__":
|
requirements.txt
CHANGED
|
@@ -1,4 +1,5 @@
|
|
| 1 |
torch==2.8.0
|
| 2 |
torchvision==0.23.0
|
| 3 |
torchaudio==2.8.0
|
|
|
|
| 4 |
-r https://raw.githubusercontent.com/comfyanonymous/ComfyUI/master/requirements.txt
|
|
|
|
| 1 |
torch==2.8.0
|
| 2 |
torchvision==0.23.0
|
| 3 |
torchaudio==2.8.0
|
| 4 |
+
diffusers
|
| 5 |
-r https://raw.githubusercontent.com/comfyanonymous/ComfyUI/master/requirements.txt
|