Spaces:
Running on Zero
Running on Zero
Download app.py from aertsimon90/hello: direct link, hf CLI and curl.
- Browser
- Download file 39.3 kB
-
https://huggingface.co/spaces/aertsimon90/hello/resolve/main/app.py
- Command line
-
hf download hf://spaces/aertsimon90/hello/app.py
-
curl -L -o app.py https://huggingface.co/spaces/aertsimon90/hello/resolve/main/app.py
39.3 kB
| import os | |
| import sys | |
| import gc | |
| import json | |
| import time | |
| import random | |
| import hashlib | |
| import tempfile | |
| import traceback | |
| import inspect | |
| from pathlib import Path | |
| from datetime import datetime | |
| # ============================================================ | |
| # ENVIRONMENT | |
| # ============================================================ | |
| os.environ.setdefault( | |
| "PYTORCH_CUDA_ALLOC_CONF", | |
| "expandable_segments:True" | |
| ) | |
| # Import spaces BEFORE torch | |
| try: | |
| import spaces | |
| HAS_SPACES = True | |
| except Exception as e: | |
| print(f"[WARN] spaces import failed: {e}", flush=True) | |
| spaces = None | |
| HAS_SPACES = False | |
| import gradio as gr | |
| import torch | |
| from PIL import Image | |
| from huggingface_hub import hf_hub_download | |
| from diffusers import ( | |
| QwenImage21Pipeline, | |
| QwenImage21Transformer2DModel, | |
| GGUFQuantizationConfig, | |
| ) | |
| # ============================================================ | |
| # CONFIG | |
| # ============================================================ | |
| GGUF_REPO = os.environ.get( | |
| "GGUF_REPO", | |
| "KasugaiSakura/Qwen-Image-2.1-Uncensored-Abenzerps-GGUF" | |
| ) | |
| GGUF_FILE = os.environ.get( | |
| "GGUF_FILE", | |
| "qwen-image-2.1-UC-Q4_K_M.gguf" | |
| ) | |
| MODEL_ID = os.environ.get( | |
| "QWEN_IMAGE_MODEL", | |
| "Qwen/Qwen-Image-2.1" | |
| ) | |
| MODEL_REVISION = os.environ.get( | |
| "QWEN_MODEL_REVISION", | |
| "790c92633540aa0cb11d9abf19eb46d861714758" | |
| ) | |
| PROMPT_ENHANCER_SPACE = os.environ.get( | |
| "PE_SPACE_ID", | |
| "hugging-apps/qwen-image-2-1-prompt-enhancer" | |
| ) | |
| MAX_INPUT_IMAGES = 10 | |
| MAX_OUTPUTS = 4 | |
| MAX_SEED = 2**31 - 1 | |
| MIN_WIDTH = 256 | |
| MIN_HEIGHT = 256 | |
| MAX_WIDTH = 4096 | |
| MAX_HEIGHT = 4096 | |
| SIZE_MULTIPLE = 32 | |
| DEFAULT_STEPS = 20 | |
| LOG_DIR = os.environ.get( | |
| "LOG_DIR", | |
| "./generation_logs" | |
| ) | |
| os.makedirs(LOG_DIR, exist_ok=True) | |
| # ============================================================ | |
| # CHECKSUMS | |
| # ============================================================ | |
| CHECKSUMS = { | |
| "qwen-image-2.1-UC-Q4_K_M.gguf": | |
| os.environ.get("SHA256_Q4_K_M", ""), | |
| "qwen-image-2.1-UC-Q4_0.gguf": | |
| os.environ.get("SHA256_Q4_0", ""), | |
| "qwen-image-2.1-UC-Q5_K_M.gguf": | |
| os.environ.get("SHA256_Q5_K_M", ""), | |
| "qwen-image-2.1-UC-Q6_K.gguf": | |
| os.environ.get("SHA256_Q6_K", ""), | |
| "qwen-image-2.1-UC-Q8_0.gguf": | |
| os.environ.get("SHA256_Q8_0", ""), | |
| "qwen-image-2.1-UC-BF16.gguf": | |
| os.environ.get("SHA256_BF16", ""), | |
| } | |
| # ============================================================ | |
| # LOGGING | |
| # ============================================================ | |
| def log(message): | |
| timestamp = datetime.now().strftime("%H:%M:%S") | |
| print( | |
| f"[{timestamp}] {message}", | |
| flush=True | |
| ) | |
| log("=" * 80) | |
| log("QWEN IMAGE 2.1 GGUF STUDIO") | |
| log("=" * 80) | |
| log(f"Python: {sys.version}") | |
| log(f"PyTorch: {torch.__version__}") | |
| try: | |
| import diffusers | |
| log(f"Diffusers: {diffusers.__version__}") | |
| except Exception: | |
| pass | |
| try: | |
| import transformers | |
| log(f"Transformers: {transformers.__version__}") | |
| except Exception: | |
| pass | |
| log(f"CUDA available: {torch.cuda.is_available()}") | |
| if torch.cuda.is_available(): | |
| log( | |
| f"GPU: " | |
| f"{torch.cuda.get_device_name(0)}" | |
| ) | |
| log( | |
| f"VRAM: " | |
| f"{torch.cuda.get_device_properties(0).total_memory / 1024**3:.2f} GB" | |
| ) | |
| log( | |
| f"CUDA: {torch.version.cuda}" | |
| ) | |
| # ============================================================ | |
| # SIZE HELPERS | |
| # ============================================================ | |
| def round_size(value, multiple=SIZE_MULTIPLE): | |
| value = int(value) | |
| value = max( | |
| MIN_WIDTH, | |
| min(MAX_WIDTH, value) | |
| ) | |
| return max( | |
| multiple, | |
| round(value / multiple) * multiple | |
| ) | |
| def normalize_size(width, height): | |
| width = max( | |
| MIN_WIDTH, | |
| min(MAX_WIDTH, int(width)) | |
| ) | |
| height = max( | |
| MIN_HEIGHT, | |
| min(MAX_HEIGHT, int(height)) | |
| ) | |
| width = round( | |
| width / SIZE_MULTIPLE | |
| ) * SIZE_MULTIPLE | |
| height = round( | |
| height / SIZE_MULTIPLE | |
| ) * SIZE_MULTIPLE | |
| return int(width), int(height) | |
| # ============================================================ | |
| # SHA256 | |
| # ============================================================ | |
| def sha256_file(path): | |
| log("Calculating SHA256...") | |
| h = hashlib.sha256() | |
| with open(path, "rb") as f: | |
| while True: | |
| chunk = f.read( | |
| 16 * 1024 * 1024 | |
| ) | |
| if not chunk: | |
| break | |
| h.update(chunk) | |
| return h.hexdigest() | |
| # ============================================================ | |
| # DOWNLOAD GGUF | |
| # ============================================================ | |
| def download_model(): | |
| log("") | |
| log("=" * 70) | |
| log("GGUF MODEL") | |
| log("=" * 70) | |
| log(f"Repository: {GGUF_REPO}") | |
| log(f"File: {GGUF_FILE}") | |
| path = hf_hub_download( | |
| repo_id=GGUF_REPO, | |
| filename=GGUF_FILE, | |
| ) | |
| log(f"GGUF path: {path}") | |
| expected = CHECKSUMS.get( | |
| GGUF_FILE, | |
| "" | |
| ) | |
| if expected: | |
| actual = sha256_file(path) | |
| if actual.lower() != expected.lower(): | |
| raise RuntimeError( | |
| "GGUF SHA256 mismatch!\n" | |
| f"Expected: {expected}\n" | |
| f"Actual: {actual}" | |
| ) | |
| log("SHA256 verification: OK") | |
| else: | |
| log( | |
| "No SHA256 configured. " | |
| "Skipping checksum verification." | |
| ) | |
| return path | |
| # ============================================================ | |
| # LOAD GGUF TRANSFORMER | |
| # ============================================================ | |
| def load_transformer(gguf_path): | |
| log("") | |
| log("=" * 70) | |
| log("LOADING GGUF TRANSFORMER") | |
| log("=" * 70) | |
| log( | |
| "Using modern Diffusers GGUFQuantizationConfig." | |
| ) | |
| log( | |
| f"Transformer: {QwenImage21Transformer2DModel.__name__}" | |
| ) | |
| log( | |
| f"Config model: {MODEL_ID}" | |
| ) | |
| quant_config = GGUFQuantizationConfig( | |
| compute_dtype=torch.bfloat16 | |
| ) | |
| # -------------------------------------------------------- | |
| # Diffusers versions have used both: | |
| # dtype= | |
| # torch_dtype= | |
| # | |
| # Detect supported argument automatically. | |
| # -------------------------------------------------------- | |
| signature = inspect.signature( | |
| QwenImage21Transformer2DModel.from_single_file | |
| ) | |
| kwargs = { | |
| "quantization_config": quant_config, | |
| "config": MODEL_ID, | |
| "subfolder": "transformer", | |
| } | |
| if "dtype" in signature.parameters: | |
| kwargs["dtype"] = torch.bfloat16 | |
| log( | |
| "Using from_single_file(dtype=...)" | |
| ) | |
| elif "torch_dtype" in signature.parameters: | |
| kwargs["torch_dtype"] = torch.bfloat16 | |
| log( | |
| "Using from_single_file(torch_dtype=...)" | |
| ) | |
| else: | |
| log( | |
| "[WARN] Could not detect dtype argument." | |
| ) | |
| log("Starting GGUF transformer loading...") | |
| log("This may take several minutes.") | |
| transformer = QwenImage21Transformer2DModel.from_single_file( | |
| gguf_path, | |
| **kwargs | |
| ) | |
| log( | |
| "GGUF transformer loaded successfully." | |
| ) | |
| return transformer | |
| # ============================================================ | |
| # LOAD PIPELINE | |
| # ============================================================ | |
| log("") | |
| log("=" * 80) | |
| log("MODEL INITIALIZATION") | |
| log("=" * 80) | |
| gguf_path = download_model() | |
| transformer = load_transformer( | |
| gguf_path | |
| ) | |
| log("") | |
| log("Loading Qwen Image 2.1 pipeline...") | |
| pipeline_kwargs = { | |
| "transformer": transformer, | |
| } | |
| pipe_signature = inspect.signature( | |
| QwenImage21Pipeline.from_pretrained | |
| ) | |
| if "dtype" in pipe_signature.parameters: | |
| pipeline_kwargs["dtype"] = torch.bfloat16 | |
| else: | |
| pipeline_kwargs["torch_dtype"] = torch.bfloat16 | |
| pipe = QwenImage21Pipeline.from_pretrained( | |
| MODEL_ID, | |
| **pipeline_kwargs | |
| ) | |
| log("Pipeline downloaded/loaded.") | |
| log("Moving pipeline to CUDA...") | |
| pipe = pipe.to("cuda") | |
| log("Pipeline moved to CUDA.") | |
| # ============================================================ | |
| # VAE TILING | |
| # ============================================================ | |
| try: | |
| pipe.vae.enable_tiling( | |
| tile_sample_min_height=1536, | |
| tile_sample_min_width=1536, | |
| tile_sample_stride_height=1152, | |
| tile_sample_stride_width=1152, | |
| ) | |
| log( | |
| "VAE tiling enabled." | |
| ) | |
| except Exception as e: | |
| log( | |
| f"[WARN] VAE tiling unavailable: {e}" | |
| ) | |
| gc.collect() | |
| if torch.cuda.is_available(): | |
| torch.cuda.empty_cache() | |
| log("") | |
| log("=" * 80) | |
| log("MODEL READY") | |
| log("=" * 80) | |
| # ============================================================ | |
| # IMAGE HELPERS | |
| # ============================================================ | |
| def gallery_to_images(items): | |
| images = [] | |
| for item in items or []: | |
| try: | |
| if isinstance( | |
| item, | |
| (tuple, list) | |
| ): | |
| if not item: | |
| continue | |
| item = item[0] | |
| if isinstance( | |
| item, | |
| Image.Image | |
| ): | |
| image = item | |
| elif isinstance( | |
| item, | |
| str | |
| ): | |
| image = Image.open( | |
| item | |
| ) | |
| elif hasattr( | |
| item, | |
| "name" | |
| ): | |
| image = Image.open( | |
| item.name | |
| ) | |
| else: | |
| continue | |
| image = image.convert( | |
| "RGBA" | |
| ) | |
| images.append(image) | |
| except Exception as e: | |
| log( | |
| f"[WARN] Could not load image: {e}" | |
| ) | |
| return images | |
| def save_temp_image(image): | |
| file = tempfile.NamedTemporaryFile( | |
| suffix=".png", | |
| delete=False | |
| ) | |
| path = file.name | |
| file.close() | |
| image.save(path) | |
| return path | |
| # ============================================================ | |
| # PROMPT ENHANCER | |
| # ============================================================ | |
| _enhancer_client = None | |
| UNIVERSAL_ENHANCER_INSTRUCTION = r""" | |
| You are a professional universal image-prompt compiler. | |
| Convert the user's request into a precise, executable English | |
| prompt for an advanced image generation or image editing model. | |
| The task is NOT limited to manga. | |
| The request may be anything visual, including: | |
| - text-to-image | |
| - image editing | |
| - character design | |
| - character sheet | |
| - character lineup | |
| - multiple characters | |
| - anime | |
| - manga | |
| - comic | |
| - storyboard | |
| - poster | |
| - book cover | |
| - product photography | |
| - portrait | |
| - architecture | |
| - environment design | |
| - cinematic scene | |
| - realistic photography | |
| - illustration | |
| - concept art | |
| - 3D render | |
| - collage | |
| - typography | |
| - advertising | |
| - sequential storytelling | |
| - multi-panel composition | |
| - fantasy | |
| - science fiction | |
| - historical scenes | |
| - or any other visual request. | |
| Your job is to make the request more executable for an image model, | |
| NOT to replace the user's idea with your own. | |
| STRICT RULES: | |
| 1. Preserve the user's original intent. | |
| 2. Do not replace the requested concept. | |
| 3. Do not invent major story events. | |
| 4. Do not invent major characters. | |
| 5. Do not remove requested characters. | |
| 6. Do not remove requested objects. | |
| 7. Do not change requested quantities. | |
| 8. If the user requests an exact number of panels, preserve it. | |
| 9. If the user requests an exact number of characters, preserve it. | |
| 10. If the user requests an exact number of objects, preserve it. | |
| 11. Preserve character identities. | |
| 12. Preserve requested clothing. | |
| 13. Preserve requested poses and actions. | |
| 14. Preserve requested relationships between characters and objects. | |
| 15. Preserve requested dialogue. | |
| 16. Preserve the meaning of dialogue exactly. | |
| 17. Do not add narration unless requested. | |
| 18. Do not add captions unless requested. | |
| 19. Do not add watermarks. | |
| 20. Translate Turkish or other languages into natural English when | |
| necessary for the image model. | |
| 21. Improve visual clarity and structure. | |
| 22. Infer reasonable visual details only when necessary to make the | |
| request visually executable. | |
| 23. Do not add unrelated artistic concepts. | |
| 24. For image editing, clearly distinguish what should change from | |
| what should remain unchanged. | |
| 25. For reference images, preserve their identity and intended role. | |
| 26. Reference images are provided in upload order: | |
| Image 1, Image 2, Image 3, etc. | |
| 27. If the prompt refers to "first image", "second image", etc., | |
| preserve those references. | |
| 28. Maintain consistency when the same character or object appears | |
| multiple times. | |
| 29. If the user requests a specific composition, follow it. | |
| 30. If the user requests a specific aspect ratio or layout, preserve it. | |
| For complex requests, use relevant visual information such as: | |
| subject, | |
| characters, | |
| identity, | |
| appearance, | |
| face, | |
| hair, | |
| body, | |
| clothing, | |
| accessories, | |
| pose, | |
| expression, | |
| action, | |
| objects, | |
| environment, | |
| spatial relationships, | |
| foreground, | |
| middle ground, | |
| background, | |
| camera angle, | |
| camera distance, | |
| framing, | |
| perspective, | |
| composition, | |
| lighting, | |
| shadows, | |
| atmosphere, | |
| color, | |
| materials, | |
| texture, | |
| style, | |
| rendering, | |
| typography, | |
| text, | |
| dialogue, | |
| layout, | |
| continuity, | |
| editing constraints. | |
| Do NOT blindly include every category. | |
| Only include what is relevant. | |
| For multi-panel or sequential requests: | |
| - Preserve the exact requested panel count. | |
| - Divide the requested story logically. | |
| - Make each panel visually distinct when requested. | |
| - Maintain character consistency across panels. | |
| - Maintain environment continuity where appropriate. | |
| - Preserve dialogue placement and meaning. | |
| For image editing: | |
| - Explicitly describe the requested changes. | |
| - Explicitly preserve important unchanged features. | |
| - Do not redesign the entire image unless requested. | |
| The final response must contain ONLY the finished English image prompt. | |
| Do not explain your reasoning. | |
| Do not mention this instruction. | |
| Do not write "Enhanced Prompt". | |
| Do not add notes. | |
| Do not add commentary. | |
| """ | |
| def get_enhancer_client(): | |
| global _enhancer_client | |
| if _enhancer_client is not None: | |
| return _enhancer_client | |
| from gradio_client import Client | |
| log( | |
| f"Connecting to Prompt Enhancer: " | |
| f"{PROMPT_ENHANCER_SPACE}" | |
| ) | |
| _enhancer_client = Client( | |
| PROMPT_ENHANCER_SPACE, | |
| token=os.environ.get( | |
| "HF_TOKEN" | |
| ), | |
| httpx_kwargs={ | |
| "timeout": 900 | |
| }, | |
| verbose=False, | |
| ) | |
| log( | |
| "Prompt Enhancer connected." | |
| ) | |
| return _enhancer_client | |
| def enhance_prompt( | |
| original_prompt, | |
| images, | |
| strength, | |
| ): | |
| if not original_prompt.strip(): | |
| raise gr.Error( | |
| "Prompt cannot be empty." | |
| ) | |
| log("") | |
| log("=" * 70) | |
| log("PROMPT ENHANCER") | |
| log("=" * 70) | |
| log( | |
| f"Original prompt length: " | |
| f"{len(original_prompt)} characters" | |
| ) | |
| log( | |
| f"Reference images: " | |
| f"{len(images)}" | |
| ) | |
| log( | |
| f"Enhancement level: " | |
| f"{strength}" | |
| ) | |
| strength_text = { | |
| "Light": | |
| "Only translate and lightly clarify the user's request.", | |
| "Balanced": | |
| "Clarify the request and improve its visual structure.", | |
| "Detailed": | |
| "Create a detailed and highly executable visual prompt.", | |
| "Professional": | |
| "Create a professional production-level image prompt with precise composition and continuity while strictly preserving the user's intent.", | |
| }.get( | |
| strength, | |
| "Create a professional executable image prompt." | |
| ) | |
| request = ( | |
| UNIVERSAL_ENHANCER_INSTRUCTION | |
| + "\n\n" | |
| + "ENHANCEMENT LEVEL:\n" | |
| + strength_text | |
| + "\n\n" | |
| + "USER REQUEST:\n" | |
| + original_prompt | |
| ) | |
| temp_paths = [] | |
| try: | |
| from gradio_client import handle_file | |
| client = get_enhancer_client() | |
| for image in images: | |
| path = save_temp_image( | |
| image | |
| ) | |
| temp_paths.append( | |
| path | |
| ) | |
| image_payload = [ | |
| handle_file(path) | |
| for path in temp_paths | |
| ] | |
| if images: | |
| max_tokens = 2048 | |
| else: | |
| max_tokens = 1536 | |
| log( | |
| f"Enhancer max tokens: " | |
| f"{max_tokens}" | |
| ) | |
| result = client.predict( | |
| prompt=request, | |
| image_paths=image_payload, | |
| max_new_tokens=max_tokens, | |
| enable_thinking=False, | |
| seed=0, | |
| randomize_seed=True, | |
| api_name="/enhance", | |
| ) | |
| if isinstance( | |
| result, | |
| (tuple, list) | |
| ): | |
| enhanced = result[0] | |
| else: | |
| enhanced = result | |
| enhanced = str( | |
| enhanced or "" | |
| ).strip() | |
| if not enhanced: | |
| raise RuntimeError( | |
| "Prompt enhancer returned an empty prompt." | |
| ) | |
| log( | |
| f"Enhanced prompt length: " | |
| f"{len(enhanced)} characters" | |
| ) | |
| log( | |
| "Prompt enhancement complete." | |
| ) | |
| return enhanced | |
| finally: | |
| for path in temp_paths: | |
| try: | |
| if os.path.exists(path): | |
| os.remove(path) | |
| except Exception: | |
| pass | |
| # ============================================================ | |
| # PREPARE REQUEST | |
| # ============================================================ | |
| def prepare_request( | |
| mode, | |
| prompt, | |
| input_gallery, | |
| enhancer_enabled, | |
| enhancer_strength, | |
| custom_size, | |
| custom_width, | |
| custom_height, | |
| preset_size, | |
| steps, | |
| seed, | |
| randomize_seed, | |
| transparent, | |
| output_count, | |
| ): | |
| log("") | |
| log("=" * 80) | |
| log("NEW REQUEST") | |
| log("=" * 80) | |
| if not prompt or not prompt.strip(): | |
| raise gr.Error( | |
| "Prompt cannot be empty." | |
| ) | |
| images = gallery_to_images( | |
| input_gallery | |
| ) | |
| if len(images) > MAX_INPUT_IMAGES: | |
| raise gr.Error( | |
| f"Maximum {MAX_INPUT_IMAGES} " | |
| "reference images are supported." | |
| ) | |
| # -------------------------------------------------------- | |
| # MODE | |
| # -------------------------------------------------------- | |
| if mode == "Create an image": | |
| if images: | |
| log( | |
| "Create mode selected. " | |
| "Reference images will still be passed to Qwen." | |
| ) | |
| elif mode == "Edit an image": | |
| if not images: | |
| raise gr.Error( | |
| "Edit mode requires at least one image." | |
| ) | |
| elif mode == "Transparent PNG": | |
| transparent = True | |
| # -------------------------------------------------------- | |
| # SEED | |
| # -------------------------------------------------------- | |
| if randomize_seed: | |
| seed = random.randint( | |
| 0, | |
| MAX_SEED | |
| ) | |
| seed = int(seed) | |
| # -------------------------------------------------------- | |
| # RESOLUTION | |
| # -------------------------------------------------------- | |
| presets = { | |
| "1024 × 1024 — 1:1": | |
| (1024, 1024), | |
| "1344 × 768 — 16:9": | |
| (1344, 768), | |
| "768 × 1344 — 9:16": | |
| (768, 1344), | |
| "1152 × 864 — 4:3": | |
| (1152, 864), | |
| "864 × 1152 — 3:4": | |
| (864, 1152), | |
| "1152 × 768 — 3:2": | |
| (1152, 768), | |
| "768 × 1152 — 2:3": | |
| (768, 1152), | |
| "1536 × 1536 — 1:1": | |
| (1536, 1536), | |
| "2048 × 2048 — 1:1": | |
| (2048, 2048), | |
| "2688 × 1536 — 16:9": | |
| (2688, 1536), | |
| "1536 × 2688 — 9:16": | |
| (1536, 2688), | |
| "1728 × 2368 — 3:4": | |
| (1728, 2368), | |
| } | |
| if custom_size: | |
| width, height = normalize_size( | |
| custom_width, | |
| custom_height | |
| ) | |
| else: | |
| width, height = presets.get( | |
| preset_size, | |
| (1024, 1024) | |
| ) | |
| log( | |
| f"Resolution: {width}x{height}" | |
| ) | |
| # -------------------------------------------------------- | |
| # PROMPT | |
| # -------------------------------------------------------- | |
| original_prompt = prompt | |
| if enhancer_enabled: | |
| final_prompt = enhance_prompt( | |
| original_prompt, | |
| images, | |
| enhancer_strength, | |
| ) | |
| else: | |
| log( | |
| "Prompt Enhancer: OFF" | |
| ) | |
| final_prompt = original_prompt | |
| # -------------------------------------------------------- | |
| # TRANSPARENT | |
| # -------------------------------------------------------- | |
| if transparent: | |
| final_prompt = ( | |
| "Create the requested image as an RGBA image " | |
| "with a transparent background. " | |
| "The background must contain true transparency " | |
| "rather than a simulated checkerboard. " | |
| "Preserve the requested subject and composition. " | |
| + final_prompt | |
| ) | |
| # -------------------------------------------------------- | |
| # STATE | |
| # -------------------------------------------------------- | |
| state = { | |
| "mode": | |
| mode, | |
| "original_prompt": | |
| original_prompt, | |
| "final_prompt": | |
| final_prompt, | |
| "images": | |
| images, | |
| "width": | |
| width, | |
| "height": | |
| height, | |
| "steps": | |
| int(steps), | |
| "seed": | |
| seed, | |
| "transparent": | |
| transparent, | |
| "output_count": | |
| int(output_count), | |
| "enhancer_enabled": | |
| enhancer_enabled, | |
| "enhancer_strength": | |
| enhancer_strength, | |
| } | |
| details = ( | |
| "REQUEST PREPARED\n" | |
| "-----------------------------\n" | |
| f"Mode: {mode}\n" | |
| f"Input images: {len(images)}\n" | |
| f"Prompt enhancer: " | |
| f"{'ON' if enhancer_enabled else 'OFF'}\n" | |
| f"Enhancement: {enhancer_strength}\n" | |
| f"Resolution: {width}x{height}\n" | |
| f"Steps: {steps}\n" | |
| f"Seed: {seed}\n" | |
| f"Transparent: {transparent}\n" | |
| f"Outputs: {output_count}\n" | |
| "\n" | |
| f"Original prompt: " | |
| f"{len(original_prompt)} characters\n" | |
| f"Final prompt: " | |
| f"{len(final_prompt)} characters\n" | |
| ) | |
| return ( | |
| state, | |
| final_prompt, | |
| seed, | |
| details, | |
| ) | |
| # ============================================================ | |
| # GPU GENERATION | |
| # ============================================================ | |
| def _generate_gpu( | |
| state, | |
| ): | |
| images = state["images"] | |
| width = state["width"] | |
| height = state["height"] | |
| prompt = state["final_prompt"] | |
| steps = state["steps"] | |
| seed = state["seed"] | |
| output_count = state["output_count"] | |
| log("") | |
| log("=" * 80) | |
| log("GPU GENERATION") | |
| log("=" * 80) | |
| log( | |
| f"Prompt length: " | |
| f"{len(prompt)}" | |
| ) | |
| log( | |
| f"Resolution: " | |
| f"{width}x{height}" | |
| ) | |
| log( | |
| f"Steps: {steps}" | |
| ) | |
| log( | |
| f"Input images: " | |
| f"{len(images)}" | |
| ) | |
| log( | |
| f"Outputs: " | |
| f"{output_count}" | |
| ) | |
| results = [] | |
| # -------------------------------------------------------- | |
| # INPUT IMAGES | |
| # -------------------------------------------------------- | |
| qwen_images = None | |
| if images: | |
| qwen_images = [] | |
| for index, image in enumerate( | |
| images, | |
| start=1 | |
| ): | |
| log( | |
| f"Preparing reference image " | |
| f"{index}/{len(images)}" | |
| ) | |
| qwen_images.append( | |
| image.convert("RGB") | |
| ) | |
| # -------------------------------------------------------- | |
| # GENERATION | |
| # -------------------------------------------------------- | |
| for index in range( | |
| output_count | |
| ): | |
| current_seed = ( | |
| seed + index | |
| ) % ( | |
| MAX_SEED + 1 | |
| ) | |
| log( | |
| f"Generating image " | |
| f"{index + 1}/{output_count}" | |
| ) | |
| log( | |
| f"Seed: {current_seed}" | |
| ) | |
| generator = torch.Generator( | |
| device="cuda" | |
| ).manual_seed( | |
| current_seed | |
| ) | |
| kwargs = { | |
| "prompt": | |
| prompt, | |
| "width": | |
| width, | |
| "height": | |
| height, | |
| "num_inference_steps": | |
| steps, | |
| "generator": | |
| generator, | |
| } | |
| if qwen_images: | |
| # Qwen Image 2.1 supports a list of | |
| # condition images in upload order. | |
| kwargs["image"] = qwen_images | |
| log( | |
| "Calling Qwen pipeline..." | |
| ) | |
| result = pipe( | |
| **kwargs | |
| ) | |
| image = result.images[0] | |
| results.append( | |
| image | |
| ) | |
| log( | |
| f"Image {index + 1}/{output_count} complete." | |
| ) | |
| del result | |
| return results | |
| if HAS_SPACES: | |
| def generate_gpu(state): | |
| return _generate_gpu( | |
| state | |
| ) | |
| else: | |
| def generate_gpu(state): | |
| return _generate_gpu( | |
| state | |
| ) | |
| # ============================================================ | |
| # SAVE OUTPUT | |
| # ============================================================ | |
| def save_image( | |
| image, | |
| seed, | |
| index, | |
| ): | |
| timestamp = datetime.now().strftime( | |
| "%Y%m%d_%H%M%S" | |
| ) | |
| filename = ( | |
| f"qwen_{timestamp}_" | |
| f"seed{seed}_" | |
| f"{index}.png" | |
| ) | |
| path = os.path.join( | |
| LOG_DIR, | |
| filename | |
| ) | |
| image.save( | |
| path | |
| ) | |
| return path | |
| # ============================================================ | |
| # FINAL GENERATION STAGE | |
| # ============================================================ | |
| def generate_request( | |
| state, | |
| ): | |
| if not state: | |
| raise gr.Error( | |
| "No prepared request. " | |
| "Click Generate again." | |
| ) | |
| log("") | |
| log("=" * 80) | |
| log("STARTING GPU STAGE") | |
| log("=" * 80) | |
| results = generate_gpu( | |
| state | |
| ) | |
| output_paths = [] | |
| for index, image in enumerate( | |
| results, | |
| start=1 | |
| ): | |
| path = save_image( | |
| image, | |
| state["seed"], | |
| index, | |
| ) | |
| output_paths.append( | |
| path | |
| ) | |
| details = ( | |
| "GENERATION COMPLETE\n" | |
| "=============================\n" | |
| f"Mode: {state['mode']}\n" | |
| f"Input images: {len(state['images'])}\n" | |
| f"Prompt Enhancer: " | |
| f"{'ON' if state['enhancer_enabled'] else 'OFF'}\n" | |
| f"Enhancement level: " | |
| f"{state['enhancer_strength']}\n" | |
| f"Resolution: " | |
| f"{state['width']}x{state['height']}\n" | |
| f"Steps: {state['steps']}\n" | |
| f"Seed: {state['seed']}\n" | |
| f"Outputs: {len(results)}\n" | |
| f"Original prompt length: " | |
| f"{len(state['original_prompt'])}\n" | |
| f"Final prompt length: " | |
| f"{len(state['final_prompt'])}\n" | |
| "\n" | |
| "Saved files:\n" | |
| + "\n".join(output_paths) | |
| ) | |
| log(details) | |
| # First image as primary download | |
| primary_file = output_paths[0] | |
| return ( | |
| results, | |
| primary_file, | |
| details, | |
| ) | |
| # ============================================================ | |
| # PREVIEW ENHANCER | |
| # ============================================================ | |
| def preview_prompt( | |
| prompt, | |
| input_gallery, | |
| enhancer_enabled, | |
| enhancer_strength, | |
| ): | |
| if not prompt or not prompt.strip(): | |
| raise gr.Error( | |
| "Enter a prompt first." | |
| ) | |
| if not enhancer_enabled: | |
| return prompt | |
| images = gallery_to_images( | |
| input_gallery | |
| ) | |
| return enhance_prompt( | |
| prompt, | |
| images, | |
| enhancer_strength, | |
| ) | |
| # ============================================================ | |
| # MODE UI | |
| # ============================================================ | |
| def mode_change(mode): | |
| if mode == "Edit an image": | |
| return gr.update( | |
| visible=True, | |
| label="Reference Images — required for Edit mode" | |
| ) | |
| return gr.update( | |
| visible=True, | |
| label="Reference Images — optional, up to 10" | |
| ) | |
| # ============================================================ | |
| # CSS | |
| # ============================================================ | |
| CSS = """ | |
| #app-container { | |
| max-width: 1500px; | |
| margin: auto; | |
| } | |
| #prompt textarea { | |
| font-size: 16px !important; | |
| line-height: 1.55 !important; | |
| } | |
| #enhanced-prompt textarea { | |
| font-size: 14px !important; | |
| line-height: 1.5 !important; | |
| } | |
| #status textarea { | |
| font-family: monospace !important; | |
| font-size: 12px !important; | |
| } | |
| #generate-button { | |
| min-height: 58px !important; | |
| font-size: 20px !important; | |
| font-weight: 700 !important; | |
| } | |
| .small-note { | |
| opacity: 0.75; | |
| font-size: 13px; | |
| } | |
| """ | |
| # ============================================================ | |
| # UI | |
| # ============================================================ | |
| with gr.Blocks( | |
| title="Qwen Image 2.1 GGUF Studio", | |
| css=CSS, | |
| ) as demo: | |
| with gr.Column( | |
| elem_id="app-container" | |
| ): | |
| gr.Markdown( | |
| """ | |
| # Qwen Image 2.1 — GGUF Studio | |
| Universal image generation and editing interface. | |
| **Text-to-image • Image editing • Multi-reference • Custom resolution • Prompt Enhancer** | |
| """ | |
| ) | |
| # ==================================================== | |
| # MODE | |
| # ==================================================== | |
| mode = gr.Radio( | |
| choices=[ | |
| "Create an image", | |
| "Edit an image", | |
| "Transparent PNG", | |
| ], | |
| value="Create an image", | |
| label="Mode", | |
| ) | |
| # ==================================================== | |
| # INPUT IMAGES | |
| # ==================================================== | |
| input_gallery = gr.Gallery( | |
| label="Reference Images — up to 10", | |
| type="pil", | |
| columns=5, | |
| height="auto", | |
| allow_preview=True, | |
| interactive=True, | |
| ) | |
| gr.Markdown( | |
| """ | |
| **Reference image order matters:** Image 1, Image 2, Image 3, etc. | |
| are passed to Qwen in exactly that order. | |
| """, | |
| elem_classes="small-note" | |
| ) | |
| # ==================================================== | |
| # PROMPT | |
| # ==================================================== | |
| prompt = gr.Textbox( | |
| label="Prompt", | |
| placeholder=( | |
| "Write what you want to generate or edit..." | |
| ), | |
| lines=8, | |
| max_lines=40, | |
| elem_id="prompt", | |
| ) | |
| gr.Markdown( | |
| """ | |
| **Prompt character limit: none.** | |
| There is no application-level `max_chars` restriction. | |
| """, | |
| elem_classes="small-note" | |
| ) | |
| # ==================================================== | |
| # PROMPT ENHANCER | |
| # ==================================================== | |
| with gr.Accordion( | |
| "Professional Universal Prompt Enhancer", | |
| open=True, | |
| ): | |
| with gr.Row(): | |
| enhancer_enabled = gr.Checkbox( | |
| label="Enable Prompt Enhancer", | |
| value=True, | |
| ) | |
| enhancer_strength = gr.Dropdown( | |
| choices=[ | |
| "Light", | |
| "Balanced", | |
| "Detailed", | |
| "Professional", | |
| ], | |
| value="Professional", | |
| label="Enhancement Level", | |
| ) | |
| preview_enhancer_button = gr.Button( | |
| "Preview Enhanced Prompt" | |
| ) | |
| enhanced_prompt = gr.Textbox( | |
| label="Enhanced Prompt", | |
| lines=14, | |
| max_lines=50, | |
| interactive=False, | |
| elem_id="enhanced-prompt", | |
| ) | |
| # ==================================================== | |
| # RESOLUTION | |
| # ==================================================== | |
| with gr.Accordion( | |
| "Resolution", | |
| open=True, | |
| ): | |
| custom_size = gr.Checkbox( | |
| label="Use Custom Resolution", | |
| value=False, | |
| ) | |
| preset_size = gr.Dropdown( | |
| choices=[ | |
| "1024 × 1024 — 1:1", | |
| "1344 × 768 — 16:9", | |
| "768 × 1344 — 9:16", | |
| "1152 × 864 — 4:3", | |
| "864 × 1152 — 3:4", | |
| "1152 × 768 — 3:2", | |
| "768 × 1152 — 2:3", | |
| "1536 × 1536 — 1:1", | |
| "2048 × 2048 — 1:1", | |
| "2688 × 1536 — 16:9", | |
| "1536 × 2688 — 9:16", | |
| "1728 × 2368 — 3:4", | |
| ], | |
| value="1024 × 1024 — 1:1", | |
| label="Preset Resolution", | |
| ) | |
| with gr.Row(): | |
| custom_width = gr.Number( | |
| label="Custom Width", | |
| value=1024, | |
| minimum=MIN_WIDTH, | |
| maximum=MAX_WIDTH, | |
| step=32, | |
| precision=0, | |
| ) | |
| custom_height = gr.Number( | |
| label="Custom Height", | |
| value=1024, | |
| minimum=MIN_HEIGHT, | |
| maximum=MAX_HEIGHT, | |
| step=32, | |
| precision=0, | |
| ) | |
| gr.Markdown( | |
| f""" | |
| Custom resolution range: | |
| **{MIN_WIDTH}px → {MAX_WIDTH}px** | |
| Width and height are automatically rounded to a multiple of | |
| **{SIZE_MULTIPLE}px**. | |
| """, | |
| elem_classes="small-note" | |
| ) | |
| # ==================================================== | |
| # ADVANCED | |
| # ==================================================== | |
| with gr.Accordion( | |
| "Advanced", | |
| open=False, | |
| ): | |
| steps = gr.Slider( | |
| label="Inference Steps", | |
| minimum=4, | |
| maximum=40, | |
| value=DEFAULT_STEPS, | |
| step=1, | |
| ) | |
| seed = gr.Number( | |
| label="Seed", | |
| value=0, | |
| minimum=0, | |
| maximum=MAX_SEED, | |
| precision=0, | |
| ) | |
| randomize_seed = gr.Checkbox( | |
| label="Randomize Seed", | |
| value=True, | |
| ) | |
| output_count = gr.Slider( | |
| label="Number of Outputs", | |
| minimum=1, | |
| maximum=MAX_OUTPUTS, | |
| value=1, | |
| step=1, | |
| ) | |
| transparent = gr.Checkbox( | |
| label="Transparent PNG / RGBA", | |
| value=False, | |
| ) | |
| # ==================================================== | |
| # GENERATE | |
| # ==================================================== | |
| generate_button = gr.Button( | |
| "GENERATE", | |
| variant="primary", | |
| elem_id="generate-button", | |
| ) | |
| # ==================================================== | |
| # STATUS | |
| # ==================================================== | |
| status = gr.Textbox( | |
| label="Status / Logs", | |
| value="Ready.", | |
| lines=12, | |
| interactive=False, | |
| elem_id="status", | |
| ) | |
| # ==================================================== | |
| # OUTPUT | |
| # ==================================================== | |
| output_gallery = gr.Gallery( | |
| label="Generated Images", | |
| columns=2, | |
| height="auto", | |
| allow_preview=True, | |
| ) | |
| download_file = gr.File( | |
| label="Primary Output", | |
| ) | |
| # ==================================================== | |
| # FINAL PROMPT | |
| # ==================================================== | |
| with gr.Accordion( | |
| "Final Prompt Sent to Qwen", | |
| open=False, | |
| ): | |
| final_prompt = gr.Textbox( | |
| label="Final Prompt", | |
| lines=18, | |
| max_lines=60, | |
| interactive=False, | |
| ) | |
| # ==================================================== | |
| # INTERNAL STATE | |
| # ==================================================== | |
| request_state = gr.State( | |
| value=None | |
| ) | |
| # ======================================================== | |
| # PREVIEW | |
| # ======================================================== | |
| preview_enhancer_button.click( | |
| fn=preview_prompt, | |
| inputs=[ | |
| prompt, | |
| input_gallery, | |
| enhancer_enabled, | |
| enhancer_strength, | |
| ], | |
| outputs=[ | |
| enhanced_prompt, | |
| ], | |
| queue=True, | |
| ) | |
| # ======================================================== | |
| # GENERATE STAGE 1 | |
| # ======================================================== | |
| prepare_event = generate_button.click( | |
| fn=prepare_request, | |
| inputs=[ | |
| mode, | |
| prompt, | |
| input_gallery, | |
| enhancer_enabled, | |
| enhancer_strength, | |
| custom_size, | |
| custom_width, | |
| custom_height, | |
| preset_size, | |
| steps, | |
| seed, | |
| randomize_seed, | |
| transparent, | |
| output_count, | |
| ], | |
| outputs=[ | |
| request_state, | |
| final_prompt, | |
| seed, | |
| status, | |
| ], | |
| queue=True, | |
| ) | |
| # ======================================================== | |
| # GENERATE STAGE 2 | |
| # ======================================================== | |
| prepare_event.then( | |
| fn=generate_request, | |
| inputs=[ | |
| request_state, | |
| ], | |
| outputs=[ | |
| output_gallery, | |
| download_file, | |
| status, | |
| ], | |
| queue=True, | |
| ) | |
| # ============================================================ | |
| # LAUNCH | |
| # ============================================================ | |
| if __name__ == "__main__": | |
| log("") | |
| log("=" * 80) | |
| log("STARTING GRADIO") | |
| log("=" * 80) | |
| demo.queue( | |
| max_size=20, | |
| default_concurrency_limit=1, | |
| ) | |
| demo.launch( | |
| server_name="0.0.0.0", | |
| server_port=int( | |
| os.environ.get( | |
| "PORT", | |
| "7860" | |
| ) | |
| ), | |
| ) |