hello / app.py
aertsimon90's picture
Update app.py
e24ae48 verified
Raw History Blame Contribute Delete
39.3 kB
import os
import sys
import gc
import json
import time
import random
import hashlib
import tempfile
import traceback
import inspect
from pathlib import Path
from datetime import datetime
# ============================================================
# ENVIRONMENT
# ============================================================
os.environ.setdefault(
"PYTORCH_CUDA_ALLOC_CONF",
"expandable_segments:True"
)
# Import spaces BEFORE torch
try:
import spaces
HAS_SPACES = True
except Exception as e:
print(f"[WARN] spaces import failed: {e}", flush=True)
spaces = None
HAS_SPACES = False
import gradio as gr
import torch
from PIL import Image
from huggingface_hub import hf_hub_download
from diffusers import (
QwenImage21Pipeline,
QwenImage21Transformer2DModel,
GGUFQuantizationConfig,
)
# ============================================================
# CONFIG
# ============================================================
GGUF_REPO = os.environ.get(
"GGUF_REPO",
"KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF"
)
GGUF_FILE = os.environ.get(
"GGUF_FILE",
"qwen-image-2.1-UC-Q4_K_M.gguf"
)
MODEL_ID = os.environ.get(
"QWEN_IMAGE_MODEL",
"Qwen/Qwen-Image-2.1"
)
MODEL_REVISION = os.environ.get(
"QWEN_MODEL_REVISION",
"790c92633540aa0cb11d9abf19eb46d861714758"
)
PROMPT_ENHANCER_SPACE = os.environ.get(
"PE_SPACE_ID",
"hugging-apps/qwen-image-2-1-prompt-enhancer"
)
MAX_INPUT_IMAGES = 10
MAX_OUTPUTS = 4
MAX_SEED = 2**31 - 1
MIN_WIDTH = 256
MIN_HEIGHT = 256
MAX_WIDTH = 4096
MAX_HEIGHT = 4096
SIZE_MULTIPLE = 32
DEFAULT_STEPS = 20
LOG_DIR = os.environ.get(
"LOG_DIR",
"./generation_logs"
)
os.makedirs(LOG_DIR, exist_ok=True)
# ============================================================
# CHECKSUMS
# ============================================================
CHECKSUMS = {
"qwen-image-2.1-UC-Q4_K_M.gguf":
os.environ.get("SHA256_Q4_K_M", ""),
"qwen-image-2.1-UC-Q4_0.gguf":
os.environ.get("SHA256_Q4_0", ""),
"qwen-image-2.1-UC-Q5_K_M.gguf":
os.environ.get("SHA256_Q5_K_M", ""),
"qwen-image-2.1-UC-Q6_K.gguf":
os.environ.get("SHA256_Q6_K", ""),
"qwen-image-2.1-UC-Q8_0.gguf":
os.environ.get("SHA256_Q8_0", ""),
"qwen-image-2.1-UC-BF16.gguf":
os.environ.get("SHA256_BF16", ""),
}
# ============================================================
# LOGGING
# ============================================================
def log(message):
timestamp = datetime.now().strftime("%H:%M:%S")
print(
f"[{timestamp}] {message}",
flush=True
)
log("=" * 80)
log("QWEN IMAGE 2.1 GGUF STUDIO")
log("=" * 80)
log(f"Python: {sys.version}")
log(f"PyTorch: {torch.__version__}")
try:
import diffusers
log(f"Diffusers: {diffusers.__version__}")
except Exception:
pass
try:
import transformers
log(f"Transformers: {transformers.__version__}")
except Exception:
pass
log(f"CUDA available: {torch.cuda.is_available()}")
if torch.cuda.is_available():
log(
f"GPU: "
f"{torch.cuda.get_device_name(0)}"
)
log(
f"VRAM: "
f"{torch.cuda.get_device_properties(0).total_memory / 1024**3:.2f} GB"
)
log(
f"CUDA: {torch.version.cuda}"
)
# ============================================================
# SIZE HELPERS
# ============================================================
def round_size(value, multiple=SIZE_MULTIPLE):
value = int(value)
value = max(
MIN_WIDTH,
min(MAX_WIDTH, value)
)
return max(
multiple,
round(value / multiple) * multiple
)
def normalize_size(width, height):
width = max(
MIN_WIDTH,
min(MAX_WIDTH, int(width))
)
height = max(
MIN_HEIGHT,
min(MAX_HEIGHT, int(height))
)
width = round(
width / SIZE_MULTIPLE
) * SIZE_MULTIPLE
height = round(
height / SIZE_MULTIPLE
) * SIZE_MULTIPLE
return int(width), int(height)
# ============================================================
# SHA256
# ============================================================
def sha256_file(path):
log("Calculating SHA256...")
h = hashlib.sha256()
with open(path, "rb") as f:
while True:
chunk = f.read(
16 * 1024 * 1024
)
if not chunk:
break
h.update(chunk)
return h.hexdigest()
# ============================================================
# DOWNLOAD GGUF
# ============================================================
def download_model():
log("")
log("=" * 70)
log("GGUF MODEL")
log("=" * 70)
log(f"Repository: {GGUF_REPO}")
log(f"File: {GGUF_FILE}")
path = hf_hub_download(
repo_id=GGUF_REPO,
filename=GGUF_FILE,
)
log(f"GGUF path: {path}")
expected = CHECKSUMS.get(
GGUF_FILE,
""
)
if expected:
actual = sha256_file(path)
if actual.lower() != expected.lower():
raise RuntimeError(
"GGUF SHA256 mismatch!\n"
f"Expected: {expected}\n"
f"Actual: {actual}"
)
log("SHA256 verification: OK")
else:
log(
"No SHA256 configured. "
"Skipping checksum verification."
)
return path
# ============================================================
# LOAD GGUF TRANSFORMER
# ============================================================
def load_transformer(gguf_path):
log("")
log("=" * 70)
log("LOADING GGUF TRANSFORMER")
log("=" * 70)
log(
"Using modern Diffusers GGUFQuantizationConfig."
)
log(
f"Transformer: {QwenImage21Transformer2DModel.__name__}"
)
log(
f"Config model: {MODEL_ID}"
)
quant_config = GGUFQuantizationConfig(
compute_dtype=torch.bfloat16
)
# --------------------------------------------------------
# Diffusers versions have used both:
# dtype=
# torch_dtype=
#
# Detect supported argument automatically.
# --------------------------------------------------------
signature = inspect.signature(
QwenImage21Transformer2DModel.from_single_file
)
kwargs = {
"quantization_config": quant_config,
"config": MODEL_ID,
"subfolder": "transformer",
}
if "dtype" in signature.parameters:
kwargs["dtype"] = torch.bfloat16
log(
"Using from_single_file(dtype=...)"
)
elif "torch_dtype" in signature.parameters:
kwargs["torch_dtype"] = torch.bfloat16
log(
"Using from_single_file(torch_dtype=...)"
)
else:
log(
"[WARN] Could not detect dtype argument."
)
log("Starting GGUF transformer loading...")
log("This may take several minutes.")
transformer = QwenImage21Transformer2DModel.from_single_file(
gguf_path,
**kwargs
)
log(
"GGUF transformer loaded successfully."
)
return transformer
# ============================================================
# LOAD PIPELINE
# ============================================================
log("")
log("=" * 80)
log("MODEL INITIALIZATION")
log("=" * 80)
gguf_path = download_model()
transformer = load_transformer(
gguf_path
)
log("")
log("Loading Qwen Image 2.1 pipeline...")
pipeline_kwargs = {
"transformer": transformer,
}
pipe_signature = inspect.signature(
QwenImage21Pipeline.from_pretrained
)
if "dtype" in pipe_signature.parameters:
pipeline_kwargs["dtype"] = torch.bfloat16
else:
pipeline_kwargs["torch_dtype"] = torch.bfloat16
pipe = QwenImage21Pipeline.from_pretrained(
MODEL_ID,
**pipeline_kwargs
)
log("Pipeline downloaded/loaded.")
log("Moving pipeline to CUDA...")
pipe = pipe.to("cuda")
log("Pipeline moved to CUDA.")
# ============================================================
# VAE TILING
# ============================================================
try:
pipe.vae.enable_tiling(
tile_sample_min_height=1536,
tile_sample_min_width=1536,
tile_sample_stride_height=1152,
tile_sample_stride_width=1152,
)
log(
"VAE tiling enabled."
)
except Exception as e:
log(
f"[WARN] VAE tiling unavailable: {e}"
)
gc.collect()
if torch.cuda.is_available():
torch.cuda.empty_cache()
log("")
log("=" * 80)
log("MODEL READY")
log("=" * 80)
# ============================================================
# IMAGE HELPERS
# ============================================================
def gallery_to_images(items):
images = []
for item in items or []:
try:
if isinstance(
item,
(tuple, list)
):
if not item:
continue
item = item[0]
if isinstance(
item,
Image.Image
):
image = item
elif isinstance(
item,
str
):
image = Image.open(
item
)
elif hasattr(
item,
"name"
):
image = Image.open(
item.name
)
else:
continue
image = image.convert(
"RGBA"
)
images.append(image)
except Exception as e:
log(
f"[WARN] Could not load image: {e}"
)
return images
def save_temp_image(image):
file = tempfile.NamedTemporaryFile(
suffix=".png",
delete=False
)
path = file.name
file.close()
image.save(path)
return path
# ============================================================
# PROMPT ENHANCER
# ============================================================
_enhancer_client = None
UNIVERSAL_ENHANCER_INSTRUCTION = r"""
You are a professional universal image-prompt compiler.
Convert the user's request into a precise, executable English
prompt for an advanced image generation or image editing model.
The task is NOT limited to manga.
The request may be anything visual, including:
- text-to-image
- image editing
- character design
- character sheet
- character lineup
- multiple characters
- anime
- manga
- comic
- storyboard
- poster
- book cover
- product photography
- portrait
- architecture
- environment design
- cinematic scene
- realistic photography
- illustration
- concept art
- 3D render
- collage
- typography
- advertising
- sequential storytelling
- multi-panel composition
- fantasy
- science fiction
- historical scenes
- or any other visual request.
Your job is to make the request more executable for an image model,
NOT to replace the user's idea with your own.
STRICT RULES:
1. Preserve the user's original intent.
2. Do not replace the requested concept.
3. Do not invent major story events.
4. Do not invent major characters.
5. Do not remove requested characters.
6. Do not remove requested objects.
7. Do not change requested quantities.
8. If the user requests an exact number of panels, preserve it.
9. If the user requests an exact number of characters, preserve it.
10. If the user requests an exact number of objects, preserve it.
11. Preserve character identities.
12. Preserve requested clothing.
13. Preserve requested poses and actions.
14. Preserve requested relationships between characters and objects.
15. Preserve requested dialogue.
16. Preserve the meaning of dialogue exactly.
17. Do not add narration unless requested.
18. Do not add captions unless requested.
19. Do not add watermarks.
20. Translate Turkish or other languages into natural English when
necessary for the image model.
21. Improve visual clarity and structure.
22. Infer reasonable visual details only when necessary to make the
request visually executable.
23. Do not add unrelated artistic concepts.
24. For image editing, clearly distinguish what should change from
what should remain unchanged.
25. For reference images, preserve their identity and intended role.
26. Reference images are provided in upload order:
Image 1, Image 2, Image 3, etc.
27. If the prompt refers to "first image", "second image", etc.,
preserve those references.
28. Maintain consistency when the same character or object appears
multiple times.
29. If the user requests a specific composition, follow it.
30. If the user requests a specific aspect ratio or layout, preserve it.
For complex requests, use relevant visual information such as:
subject,
characters,
identity,
appearance,
face,
hair,
body,
clothing,
accessories,
pose,
expression,
action,
objects,
environment,
spatial relationships,
foreground,
middle ground,
background,
camera angle,
camera distance,
framing,
perspective,
composition,
lighting,
shadows,
atmosphere,
color,
materials,
texture,
style,
rendering,
typography,
text,
dialogue,
layout,
continuity,
editing constraints.
Do NOT blindly include every category.
Only include what is relevant.
For multi-panel or sequential requests:
- Preserve the exact requested panel count.
- Divide the requested story logically.
- Make each panel visually distinct when requested.
- Maintain character consistency across panels.
- Maintain environment continuity where appropriate.
- Preserve dialogue placement and meaning.
For image editing:
- Explicitly describe the requested changes.
- Explicitly preserve important unchanged features.
- Do not redesign the entire image unless requested.
The final response must contain ONLY the finished English image prompt.
Do not explain your reasoning.
Do not mention this instruction.
Do not write "Enhanced Prompt".
Do not add notes.
Do not add commentary.
"""
def get_enhancer_client():
global _enhancer_client
if _enhancer_client is not None:
return _enhancer_client
from gradio_client import Client
log(
f"Connecting to Prompt Enhancer: "
f"{PROMPT_ENHANCER_SPACE}"
)
_enhancer_client = Client(
PROMPT_ENHANCER_SPACE,
token=os.environ.get(
"HF_TOKEN"
),
httpx_kwargs={
"timeout": 900
},
verbose=False,
)
log(
"Prompt Enhancer connected."
)
return _enhancer_client
def enhance_prompt(
original_prompt,
images,
strength,
):
if not original_prompt.strip():
raise gr.Error(
"Prompt cannot be empty."
)
log("")
log("=" * 70)
log("PROMPT ENHANCER")
log("=" * 70)
log(
f"Original prompt length: "
f"{len(original_prompt)} characters"
)
log(
f"Reference images: "
f"{len(images)}"
)
log(
f"Enhancement level: "
f"{strength}"
)
strength_text = {
"Light":
"Only translate and lightly clarify the user's request.",
"Balanced":
"Clarify the request and improve its visual structure.",
"Detailed":
"Create a detailed and highly executable visual prompt.",
"Professional":
"Create a professional production-level image prompt with precise composition and continuity while strictly preserving the user's intent.",
}.get(
strength,
"Create a professional executable image prompt."
)
request = (
UNIVERSAL_ENHANCER_INSTRUCTION
+ "\n\n"
+ "ENHANCEMENT LEVEL:\n"
+ strength_text
+ "\n\n"
+ "USER REQUEST:\n"
+ original_prompt
)
temp_paths = []
try:
from gradio_client import handle_file
client = get_enhancer_client()
for image in images:
path = save_temp_image(
image
)
temp_paths.append(
path
)
image_payload = [
handle_file(path)
for path in temp_paths
]
if images:
max_tokens = 2048
else:
max_tokens = 1536
log(
f"Enhancer max tokens: "
f"{max_tokens}"
)
result = client.predict(
prompt=request,
image_paths=image_payload,
max_new_tokens=max_tokens,
enable_thinking=False,
seed=0,
randomize_seed=True,
api_name="/enhance",
)
if isinstance(
result,
(tuple, list)
):
enhanced = result[0]
else:
enhanced = result
enhanced = str(
enhanced or ""
).strip()
if not enhanced:
raise RuntimeError(
"Prompt enhancer returned an empty prompt."
)
log(
f"Enhanced prompt length: "
f"{len(enhanced)} characters"
)
log(
"Prompt enhancement complete."
)
return enhanced
finally:
for path in temp_paths:
try:
if os.path.exists(path):
os.remove(path)
except Exception:
pass
# ============================================================
# PREPARE REQUEST
# ============================================================
def prepare_request(
mode,
prompt,
input_gallery,
enhancer_enabled,
enhancer_strength,
custom_size,
custom_width,
custom_height,
preset_size,
steps,
seed,
randomize_seed,
transparent,
output_count,
):
log("")
log("=" * 80)
log("NEW REQUEST")
log("=" * 80)
if not prompt or not prompt.strip():
raise gr.Error(
"Prompt cannot be empty."
)
images = gallery_to_images(
input_gallery
)
if len(images) > MAX_INPUT_IMAGES:
raise gr.Error(
f"Maximum {MAX_INPUT_IMAGES} "
"reference images are supported."
)
# --------------------------------------------------------
# MODE
# --------------------------------------------------------
if mode == "Create an image":
if images:
log(
"Create mode selected. "
"Reference images will still be passed to Qwen."
)
elif mode == "Edit an image":
if not images:
raise gr.Error(
"Edit mode requires at least one image."
)
elif mode == "Transparent PNG":
transparent = True
# --------------------------------------------------------
# SEED
# --------------------------------------------------------
if randomize_seed:
seed = random.randint(
0,
MAX_SEED
)
seed = int(seed)
# --------------------------------------------------------
# RESOLUTION
# --------------------------------------------------------
presets = {
"1024 × 1024 — 1:1":
(1024, 1024),
"1344 × 768 — 16:9":
(1344, 768),
"768 × 1344 — 9:16":
(768, 1344),
"1152 × 864 — 4:3":
(1152, 864),
"864 × 1152 — 3:4":
(864, 1152),
"1152 × 768 — 3:2":
(1152, 768),
"768 × 1152 — 2:3":
(768, 1152),
"1536 × 1536 — 1:1":
(1536, 1536),
"2048 × 2048 — 1:1":
(2048, 2048),
"2688 × 1536 — 16:9":
(2688, 1536),
"1536 × 2688 — 9:16":
(1536, 2688),
"1728 × 2368 — 3:4":
(1728, 2368),
}
if custom_size:
width, height = normalize_size(
custom_width,
custom_height
)
else:
width, height = presets.get(
preset_size,
(1024, 1024)
)
log(
f"Resolution: {width}x{height}"
)
# --------------------------------------------------------
# PROMPT
# --------------------------------------------------------
original_prompt = prompt
if enhancer_enabled:
final_prompt = enhance_prompt(
original_prompt,
images,
enhancer_strength,
)
else:
log(
"Prompt Enhancer: OFF"
)
final_prompt = original_prompt
# --------------------------------------------------------
# TRANSPARENT
# --------------------------------------------------------
if transparent:
final_prompt = (
"Create the requested image as an RGBA image "
"with a transparent background. "
"The background must contain true transparency "
"rather than a simulated checkerboard. "
"Preserve the requested subject and composition. "
+ final_prompt
)
# --------------------------------------------------------
# STATE
# --------------------------------------------------------
state = {
"mode":
mode,
"original_prompt":
original_prompt,
"final_prompt":
final_prompt,
"images":
images,
"width":
width,
"height":
height,
"steps":
int(steps),
"seed":
seed,
"transparent":
transparent,
"output_count":
int(output_count),
"enhancer_enabled":
enhancer_enabled,
"enhancer_strength":
enhancer_strength,
}
details = (
"REQUEST PREPARED\n"
"-----------------------------\n"
f"Mode: {mode}\n"
f"Input images: {len(images)}\n"
f"Prompt enhancer: "
f"{'ON' if enhancer_enabled else 'OFF'}\n"
f"Enhancement: {enhancer_strength}\n"
f"Resolution: {width}x{height}\n"
f"Steps: {steps}\n"
f"Seed: {seed}\n"
f"Transparent: {transparent}\n"
f"Outputs: {output_count}\n"
"\n"
f"Original prompt: "
f"{len(original_prompt)} characters\n"
f"Final prompt: "
f"{len(final_prompt)} characters\n"
)
return (
state,
final_prompt,
seed,
details,
)
# ============================================================
# GPU GENERATION
# ============================================================
def _generate_gpu(
state,
):
images = state["images"]
width = state["width"]
height = state["height"]
prompt = state["final_prompt"]
steps = state["steps"]
seed = state["seed"]
output_count = state["output_count"]
log("")
log("=" * 80)
log("GPU GENERATION")
log("=" * 80)
log(
f"Prompt length: "
f"{len(prompt)}"
)
log(
f"Resolution: "
f"{width}x{height}"
)
log(
f"Steps: {steps}"
)
log(
f"Input images: "
f"{len(images)}"
)
log(
f"Outputs: "
f"{output_count}"
)
results = []
# --------------------------------------------------------
# INPUT IMAGES
# --------------------------------------------------------
qwen_images = None
if images:
qwen_images = []
for index, image in enumerate(
images,
start=1
):
log(
f"Preparing reference image "
f"{index}/{len(images)}"
)
qwen_images.append(
image.convert("RGB")
)
# --------------------------------------------------------
# GENERATION
# --------------------------------------------------------
for index in range(
output_count
):
current_seed = (
seed + index
) % (
MAX_SEED + 1
)
log(
f"Generating image "
f"{index + 1}/{output_count}"
)
log(
f"Seed: {current_seed}"
)
generator = torch.Generator(
device="cuda"
).manual_seed(
current_seed
)
kwargs = {
"prompt":
prompt,
"width":
width,
"height":
height,
"num_inference_steps":
steps,
"generator":
generator,
}
if qwen_images:
# Qwen Image 2.1 supports a list of
# condition images in upload order.
kwargs["image"] = qwen_images
log(
"Calling Qwen pipeline..."
)
result = pipe(
**kwargs
)
image = result.images[0]
results.append(
image
)
log(
f"Image {index + 1}/{output_count} complete."
)
del result
return results
if HAS_SPACES:
@spaces.GPU(duration=180)
def generate_gpu(state):
return _generate_gpu(
state
)
else:
def generate_gpu(state):
return _generate_gpu(
state
)
# ============================================================
# SAVE OUTPUT
# ============================================================
def save_image(
image,
seed,
index,
):
timestamp = datetime.now().strftime(
"%Y%m%d_%H%M%S"
)
filename = (
f"qwen_{timestamp}_"
f"seed{seed}_"
f"{index}.png"
)
path = os.path.join(
LOG_DIR,
filename
)
image.save(
path
)
return path
# ============================================================
# FINAL GENERATION STAGE
# ============================================================
def generate_request(
state,
):
if not state:
raise gr.Error(
"No prepared request. "
"Click Generate again."
)
log("")
log("=" * 80)
log("STARTING GPU STAGE")
log("=" * 80)
results = generate_gpu(
state
)
output_paths = []
for index, image in enumerate(
results,
start=1
):
path = save_image(
image,
state["seed"],
index,
)
output_paths.append(
path
)
details = (
"GENERATION COMPLETE\n"
"=============================\n"
f"Mode: {state['mode']}\n"
f"Input images: {len(state['images'])}\n"
f"Prompt Enhancer: "
f"{'ON' if state['enhancer_enabled'] else 'OFF'}\n"
f"Enhancement level: "
f"{state['enhancer_strength']}\n"
f"Resolution: "
f"{state['width']}x{state['height']}\n"
f"Steps: {state['steps']}\n"
f"Seed: {state['seed']}\n"
f"Outputs: {len(results)}\n"
f"Original prompt length: "
f"{len(state['original_prompt'])}\n"
f"Final prompt length: "
f"{len(state['final_prompt'])}\n"
"\n"
"Saved files:\n"
+ "\n".join(output_paths)
)
log(details)
# First image as primary download
primary_file = output_paths[0]
return (
results,
primary_file,
details,
)
# ============================================================
# PREVIEW ENHANCER
# ============================================================
def preview_prompt(
prompt,
input_gallery,
enhancer_enabled,
enhancer_strength,
):
if not prompt or not prompt.strip():
raise gr.Error(
"Enter a prompt first."
)
if not enhancer_enabled:
return prompt
images = gallery_to_images(
input_gallery
)
return enhance_prompt(
prompt,
images,
enhancer_strength,
)
# ============================================================
# MODE UI
# ============================================================
def mode_change(mode):
if mode == "Edit an image":
return gr.update(
visible=True,
label="Reference Images — required for Edit mode"
)
return gr.update(
visible=True,
label="Reference Images — optional, up to 10"
)
# ============================================================
# CSS
# ============================================================
CSS = """
#app-container {
max-width: 1500px;
margin: auto;
}
#prompt textarea {
font-size: 16px !important;
line-height: 1.55 !important;
}
#enhanced-prompt textarea {
font-size: 14px !important;
line-height: 1.5 !important;
}
#status textarea {
font-family: monospace !important;
font-size: 12px !important;
}
#generate-button {
min-height: 58px !important;
font-size: 20px !important;
font-weight: 700 !important;
}
.small-note {
opacity: 0.75;
font-size: 13px;
}
"""
# ============================================================
# UI
# ============================================================
with gr.Blocks(
title="Qwen Image 2.1 GGUF Studio",
css=CSS,
) as demo:
with gr.Column(
elem_id="app-container"
):
gr.Markdown(
"""
# Qwen Image 2.1 — GGUF Studio
Universal image generation and editing interface.
**Text-to-image • Image editing • Multi-reference • Custom resolution • Prompt Enhancer**
"""
)
# ====================================================
# MODE
# ====================================================
mode = gr.Radio(
choices=[
"Create an image",
"Edit an image",
"Transparent PNG",
],
value="Create an image",
label="Mode",
)
# ====================================================
# INPUT IMAGES
# ====================================================
input_gallery = gr.Gallery(
label="Reference Images — up to 10",
type="pil",
columns=5,
height="auto",
allow_preview=True,
interactive=True,
)
gr.Markdown(
"""
**Reference image order matters:** Image 1, Image 2, Image 3, etc.
are passed to Qwen in exactly that order.
""",
elem_classes="small-note"
)
# ====================================================
# PROMPT
# ====================================================
prompt = gr.Textbox(
label="Prompt",
placeholder=(
"Write what you want to generate or edit..."
),
lines=8,
max_lines=40,
elem_id="prompt",
)
gr.Markdown(
"""
**Prompt character limit: none.**
There is no application-level `max_chars` restriction.
""",
elem_classes="small-note"
)
# ====================================================
# PROMPT ENHANCER
# ====================================================
with gr.Accordion(
"Professional Universal Prompt Enhancer",
open=True,
):
with gr.Row():
enhancer_enabled = gr.Checkbox(
label="Enable Prompt Enhancer",
value=True,
)
enhancer_strength = gr.Dropdown(
choices=[
"Light",
"Balanced",
"Detailed",
"Professional",
],
value="Professional",
label="Enhancement Level",
)
preview_enhancer_button = gr.Button(
"Preview Enhanced Prompt"
)
enhanced_prompt = gr.Textbox(
label="Enhanced Prompt",
lines=14,
max_lines=50,
interactive=False,
elem_id="enhanced-prompt",
)
# ====================================================
# RESOLUTION
# ====================================================
with gr.Accordion(
"Resolution",
open=True,
):
custom_size = gr.Checkbox(
label="Use Custom Resolution",
value=False,
)
preset_size = gr.Dropdown(
choices=[
"1024 × 1024 — 1:1",
"1344 × 768 — 16:9",
"768 × 1344 — 9:16",
"1152 × 864 — 4:3",
"864 × 1152 — 3:4",
"1152 × 768 — 3:2",
"768 × 1152 — 2:3",
"1536 × 1536 — 1:1",
"2048 × 2048 — 1:1",
"2688 × 1536 — 16:9",
"1536 × 2688 — 9:16",
"1728 × 2368 — 3:4",
],
value="1024 × 1024 — 1:1",
label="Preset Resolution",
)
with gr.Row():
custom_width = gr.Number(
label="Custom Width",
value=1024,
minimum=MIN_WIDTH,
maximum=MAX_WIDTH,
step=32,
precision=0,
)
custom_height = gr.Number(
label="Custom Height",
value=1024,
minimum=MIN_HEIGHT,
maximum=MAX_HEIGHT,
step=32,
precision=0,
)
gr.Markdown(
f"""
Custom resolution range:
**{MIN_WIDTH}px → {MAX_WIDTH}px**
Width and height are automatically rounded to a multiple of
**{SIZE_MULTIPLE}px**.
""",
elem_classes="small-note"
)
# ====================================================
# ADVANCED
# ====================================================
with gr.Accordion(
"Advanced",
open=False,
):
steps = gr.Slider(
label="Inference Steps",
minimum=4,
maximum=40,
value=DEFAULT_STEPS,
step=1,
)
seed = gr.Number(
label="Seed",
value=0,
minimum=0,
maximum=MAX_SEED,
precision=0,
)
randomize_seed = gr.Checkbox(
label="Randomize Seed",
value=True,
)
output_count = gr.Slider(
label="Number of Outputs",
minimum=1,
maximum=MAX_OUTPUTS,
value=1,
step=1,
)
transparent = gr.Checkbox(
label="Transparent PNG / RGBA",
value=False,
)
# ====================================================
# GENERATE
# ====================================================
generate_button = gr.Button(
"GENERATE",
variant="primary",
elem_id="generate-button",
)
# ====================================================
# STATUS
# ====================================================
status = gr.Textbox(
label="Status / Logs",
value="Ready.",
lines=12,
interactive=False,
elem_id="status",
)
# ====================================================
# OUTPUT
# ====================================================
output_gallery = gr.Gallery(
label="Generated Images",
columns=2,
height="auto",
allow_preview=True,
)
download_file = gr.File(
label="Primary Output",
)
# ====================================================
# FINAL PROMPT
# ====================================================
with gr.Accordion(
"Final Prompt Sent to Qwen",
open=False,
):
final_prompt = gr.Textbox(
label="Final Prompt",
lines=18,
max_lines=60,
interactive=False,
)
# ====================================================
# INTERNAL STATE
# ====================================================
request_state = gr.State(
value=None
)
# ========================================================
# PREVIEW
# ========================================================
preview_enhancer_button.click(
fn=preview_prompt,
inputs=[
prompt,
input_gallery,
enhancer_enabled,
enhancer_strength,
],
outputs=[
enhanced_prompt,
],
queue=True,
)
# ========================================================
# GENERATE STAGE 1
# ========================================================
prepare_event = generate_button.click(
fn=prepare_request,
inputs=[
mode,
prompt,
input_gallery,
enhancer_enabled,
enhancer_strength,
custom_size,
custom_width,
custom_height,
preset_size,
steps,
seed,
randomize_seed,
transparent,
output_count,
],
outputs=[
request_state,
final_prompt,
seed,
status,
],
queue=True,
)
# ========================================================
# GENERATE STAGE 2
# ========================================================
prepare_event.then(
fn=generate_request,
inputs=[
request_state,
],
outputs=[
output_gallery,
download_file,
status,
],
queue=True,
)
# ============================================================
# LAUNCH
# ============================================================
if __name__ == "__main__":
log("")
log("=" * 80)
log("STARTING GRADIO")
log("=" * 80)
demo.queue(
max_size=20,
default_concurrency_limit=1,
)
demo.launch(
server_name="0.0.0.0",
server_port=int(
os.environ.get(
"PORT",
"7860"
)
),
)