import os os.environ.setdefault("HF_HOME", "/tmp/huggingface") os.environ.setdefault("HF_MODULES_CACHE", "/tmp/hf_modules") os.environ.setdefault("MPLCONFIGDIR", "/tmp/matplotlib") import html import json import math import re import time from typing import Any import gradio as gr from huggingface_hub import InferenceClient MODEL_ID = "deepseek-ai/DeepSeek-V4-Flash-0731" THREE_VERSION = "0.180.0" MAX_PROMPT_LENGTH = 1_200 MAX_PARTS = 64 ALLOWED_PRIMITIVES = { "box", "sphere", "cylinder", "cone", "torus", "capsule", "dodecahedron", } DETAIL_BUDGETS = { "Iconic · 12–20 parts": 20, "Balanced · 20–36 parts": 36, "Intricate · 36–64 parts": 64, } SYSTEM_PROMPT = """You are an expert procedural 3D art director. Convert a text description into a compact JSON scene specification for a browser Three.js renderer. This is reconstruction-by-code: use named primitive parts, observable proportions, explicit materials, and a readable silhouette. Think through a blockout, structural pass, form pass, material pass, and surface-detail pass silently. Before emitting JSON, silently inventory the identity-defining parts. Spend parts on silhouette and recognition before decoration. Do not claim unseen or unspecified details as factual; make tasteful design inferences instead. Return exactly one valid JSON object and no markdown. The schema is: { "title": "short scene title", "background": "#RRGGBB", "ground": "#RRGGBB", "camera": {"position": [x,y,z], "target": [x,y,z], "fov": 25..65}, "autoRotate": true, "parts": [ { "name": "descriptive unique part name", "primitive": "box|sphere|cylinder|cone|torus|capsule|dodecahedron", "position": [x,y,z], "rotation": [x,y,z], "scale": [x,y,z], "color": "#RRGGBB", "roughness": 0..1, "metalness": 0..1, "emissive": "#RRGGBB", "emissiveIntensity": 0..2 } ] } Coordinate rules: Y is up; the ground is Y=0. Box and cylinder base geometries are unit-sized; scale is the part's final width, height, and depth. Cylinder and cone axes point along Y. Sphere and dodecahedron start at unit diameter. Torus lies in the XY plane before rotation. Capsule's long axis is Y. Positions and rotations are [x,y,z], with rotations in radians. Keep the whole object roughly within a 10 x 10 x 10 volume. Every scale component must be positive and at least 0.05. Use coherent PBR values. Prefer 2–5 well-separated material families. Use repeated parts for wheels, buttons, teeth, rivets, windows, or spokes when they define identity. Avoid coplanar overlap. Ensure the lowest visible geometry meets or slightly enters the ground plane. Do not include comments, JavaScript, HTML, URLs, textures, external assets, or keys outside the schema.""" def _clean_text(value: Any, fallback: str, limit: int = 80) -> str: text = re.sub(r"[^\w .,'&()\-/]", "", str(value or ""), flags=re.UNICODE) text = re.sub(r"\s+", " ", text).strip() return (text[:limit] or fallback).strip() def _color(value: Any, fallback: str) -> str: candidate = str(value or "").strip() return candidate.upper() if re.fullmatch(r"#[0-9a-fA-F]{6}", candidate) else fallback def _number(value: Any, fallback: float, low: float, high: float) -> float: try: number = float(value) except (TypeError, ValueError): return fallback if not math.isfinite(number): return fallback return round(max(low, min(high, number)), 4) def _vector( value: Any, fallback: list[float], low: float, high: float, positive: bool = False, ) -> list[float]: if not isinstance(value, list) or len(value) != 3: value = fallback floor = 0.05 if positive else low return [_number(item, fallback[index], floor, high) for index, item in enumerate(value)] def _extract_json(raw: str) -> dict[str, Any]: candidate = (raw or "").strip() candidate = re.sub(r"^```(?:json)?\s*", "", candidate, flags=re.IGNORECASE) candidate = re.sub(r"\s*```$", "", candidate) try: parsed = json.loads(candidate) except json.JSONDecodeError: start = candidate.find("{") if start < 0: raise ValueError("The endpoint did not return a JSON scene.") from None try: parsed, _ = json.JSONDecoder().raw_decode(candidate[start:]) except json.JSONDecodeError as exc: raise ValueError("The endpoint returned an incomplete JSON scene.") from exc if not isinstance(parsed, dict): raise ValueError("The endpoint returned JSON, but not a scene object.") return parsed def _validate_scene(raw_scene: dict[str, Any], part_budget: int) -> dict[str, Any]: raw_parts = raw_scene.get("parts", raw_scene.get("objects", [])) if not isinstance(raw_parts, list) or not raw_parts: raise ValueError("The generated scene has no renderable parts.") parts: list[dict[str, Any]] = [] for index, raw_part in enumerate(raw_parts[: min(part_budget, MAX_PARTS)]): if not isinstance(raw_part, dict): continue primitive = str(raw_part.get("primitive", "box")).lower().strip() if primitive not in ALLOWED_PRIMITIVES: primitive = "box" parts.append( { "name": _clean_text(raw_part.get("name"), f"part {index + 1}"), "primitive": primitive, "position": _vector(raw_part.get("position"), [0, 0.5, 0], -12, 12), "rotation": _vector(raw_part.get("rotation"), [0, 0, 0], -math.tau, math.tau), "scale": _vector(raw_part.get("scale"), [1, 1, 1], 0.05, 10, positive=True), "color": _color(raw_part.get("color"), "#8A8A8A"), "roughness": _number(raw_part.get("roughness"), 0.55, 0, 1), "metalness": _number(raw_part.get("metalness"), 0.05, 0, 1), "emissive": _color(raw_part.get("emissive"), "#000000"), "emissiveIntensity": _number(raw_part.get("emissiveIntensity"), 0, 0, 2), } ) if not parts: raise ValueError("The generated parts could not be normalized safely.") camera = raw_scene.get("camera") if isinstance(raw_scene.get("camera"), dict) else {} return { "title": _clean_text(raw_scene.get("title"), "Untitled object"), "background": _color(raw_scene.get("background"), "#A9B7C0"), "ground": _color(raw_scene.get("ground"), "#D7D4C8"), "camera": { "position": _vector(camera.get("position"), [5.5, 4.2, 7.5], -20, 20), "target": _vector(camera.get("target"), [0, 1.5, 0], -12, 12), "fov": _number(camera.get("fov"), 38, 25, 65), }, "autoRotate": bool(raw_scene.get("autoRotate", True)), "parts": parts, } def _model_scene(prompt: str, detail: str, finish: str) -> tuple[dict[str, Any], float]: token = os.environ.get("HF_TOKEN") if not token: raise gr.Error("HF_TOKEN is not configured on this Space.") part_budget = DETAIL_BUDGETS.get(detail, 36) request = f"""Build this object or scene: {prompt} Target detail budget: at most {part_budget} parts. Art direction: {finish}. Make one coherent, presentation-ready composition. Return JSON only.""" client = InferenceClient(model=MODEL_ID, provider="auto", token=token, timeout=180) started = time.perf_counter() try: response = client.chat_completion( messages=[ {"role": "system", "content": SYSTEM_PROMPT}, {"role": "user", "content": request}, ], max_tokens=8_000, temperature=0.25, top_p=0.95, response_format={"type": "json_object"}, extra_body={"reasoning_effort": "low"}, ) except Exception as exc: message = re.sub(r"hf_[A-Za-z0-9]+", "[redacted]", str(exc)) raise gr.Error(f"Hugging Face inference failed: {message[:300]}") from exc content = response.choices[0].message.content or "" scene = _validate_scene(_extract_json(content), part_budget) return scene, time.perf_counter() - started RENDER_DOCUMENT = r"""
Drag to orbit · Wheel to zoom · Generated from safe primitives
""" def render_preview(scene: dict[str, Any]) -> str: scene_json = json.dumps(scene, separators=(",", ":"), ensure_ascii=True) scene_json = scene_json.replace("<", "\\u003c").replace(">", "\\u003e").replace("&", "\\u0026") document = RENDER_DOCUMENT.replace("__THREE_VERSION__", THREE_VERSION).replace( "__SCENE_JSON__", scene_json ) return ( '' ) WELCOME_SCENE = { "title": "Scene Machine", "background": "#8E99A8", "ground": "#D9D5C8", "camera": {"position": [5.5, 4.2, 7.5], "target": [0, 1.5, 0], "fov": 38}, "autoRotate": True, "parts": [ {"name": "monitor", "primitive": "box", "position": [0, 1.8, 0], "rotation": [0, 0, 0], "scale": [2.6, 2.1, 1.6], "color": "#D8D3C4", "roughness": .6, "metalness": 0, "emissive": "#000000", "emissiveIntensity": 0}, {"name": "screen", "primitive": "box", "position": [0, 2.05, .83], "rotation": [0, 0, 0], "scale": [1.85, 1.25, .06], "color": "#1C2834", "roughness": .25, "metalness": .1, "emissive": "#24384A", "emissiveIntensity": .45}, {"name": "base", "primitive": "box", "position": [0, .45, .25], "rotation": [-.08, 0, 0], "scale": [2.8, .28, 1.4], "color": "#D8D3C4", "roughness": .65, "metalness": 0, "emissive": "#000000", "emissiveIntensity": 0}, {"name": "power button", "primitive": "sphere", "position": [.95, 1.25, .84], "rotation": [0, 0, 0], "scale": [.16, .16, .08], "color": "#6F7A71", "roughness": .4, "metalness": .05, "emissive": "#78C878", "emissiveIntensity": .35}, ], } def generate_scene(prompt: str, detail: str, finish: str) -> tuple[str, str, str]: prompt = re.sub(r"\s+", " ", str(prompt or "")).strip() if not prompt: raise gr.Error("Describe an object or small scene first.") if len(prompt) > MAX_PROMPT_LENGTH: raise gr.Error(f"Keep the description under {MAX_PROMPT_LENGTH:,} characters.") scene, elapsed = _model_scene(prompt, detail, finish) spec_text = json.dumps(scene, indent=2, ensure_ascii=False) status = ( f"**Built {len(scene['parts'])} safe primitive parts** · " f"{elapsed:.1f}s · `{MODEL_ID}` via Hugging Face Inference Providers" ) return render_preview(scene), spec_text, status CSS = r""" :root { --os9-gray: #c8c8c8; --os9-light: #ffffff; --os9-mid: #999999; --os9-dark: #222222; --os9-blue: #1f4ea8; } body, .gradio-container { background: #777f8c !important; color: #111 !important; font-family: Charcoal, Chicago, Geneva, "Lucida Grande", Arial, sans-serif !important; } .gradio-container { max-width: none !important; padding: 0 !important; } #menu-bar { background: #f4f4f4; border-bottom: 2px solid #111; min-height: 29px; padding: 4px 12px; font-size: 13px; box-shadow: 0 1px 0 #fff; } #menu-bar p { margin: 0 !important; } #menu-bar strong { margin-right: 24px; } .rainbow-mark { display:inline-block; width:13px; height:13px; margin-right:8px; vertical-align:-2px; border:1px solid #111; background:linear-gradient(180deg,#69b34c 0 17%,#f6c445 17% 34%,#f6921e 34% 51%,#e83f36 51% 68%,#8f4aa8 68% 84%,#326db3 84%); } #desktop { max-width: 1260px; margin: 24px auto 28px; padding: 0 18px; } .os-window { border: 2px solid #111 !important; border-radius: 0 !important; background: var(--os9-gray) !important; padding: 5px !important; box-shadow: 3px 3px 0 rgba(0,0,0,.75) !important; } .titlebar { border: 1px solid #111; min-height: 22px; padding: 2px 32px; margin-bottom: 7px; text-align: center; font-size: 12px; line-height: 16px; font-weight: 700; background: repeating-linear-gradient(0deg,#fff 0,#fff 1px,#777 1px,#777 2px); position: relative; } .titlebar:before { content:""; position:absolute; left:3px; top:3px; width:12px; height:12px; background:#d0d0d0; border:1px solid #111; box-shadow:inset 1px 1px #fff; } .titlebar span { background:var(--os9-gray); padding:0 8px; } #hero-copy { padding: 12px 14px 4px; } #hero-copy h1 { font-size: 25px; letter-spacing: -.03em; margin: 0 0 6px; } #hero-copy p { font-size: 13px; line-height: 1.5; margin: 0; max-width: 720px; } .gradio-container .form, .gradio-container .block { border-radius: 0 !important; } .gradio-container label, .gradio-container .label-wrap { font-size: 12px !important; font-weight: 700 !important; } .gradio-container textarea, .gradio-container input, .gradio-container select { border: 2px inset #eee !important; border-radius: 0 !important; background: #fff !important; color: #111 !important; font-family: Geneva, Arial, sans-serif !important; } .gradio-container button { border: 1px solid #111 !important; border-radius: 4px !important; color: #111 !important; background: linear-gradient(#fff,#bcbcbc) !important; box-shadow: inset 1px 1px #fff, 1px 1px #555 !important; font-family: Charcoal, Geneva, Arial, sans-serif !important; font-weight: 700 !important; } .gradio-container button.primary { border: 2px solid #111 !important; background: linear-gradient(#fff,#bfc9dd) !important; } .gradio-container button:active { transform: translate(1px,1px); box-shadow: inset 1px 1px #777 !important; } #preview-frame { border: 2px inset #eee !important; padding: 0 !important; background: #111 !important; } #status-strip { border:1px inset #eee; background:#dedede; padding:6px 8px; min-height:31px; font-size:11px; } #status-strip p { margin:0 !important; } .footer-note { text-align:center; color:#f2f2f2; font-size:11px; text-shadow:1px 1px #222; } .footer-note a { color:#fff !important; text-decoration:underline; } details { border-radius:0 !important; } @media (max-width: 720px) { #desktop { margin-top: 12px; padding: 0 9px; } #menu-bar { font-size: 11px; } #menu-bar strong { margin-right: 10px; } } """ with gr.Blocks(title="Scene Machine 9") as demo: gr.HTML( '" ) with gr.Column(elem_id="desktop"): with gr.Column(elem_classes="os-window"): gr.HTML('
Scene Machine 9
') gr.Markdown( "# Describe it. Build it. Orbit it.\n" "Turn a short description into a procedural Three.js study. DeepSeek V4 plans the " "parts; a strict renderer accepts only safe geometry and PBR material fields.", elem_id="hero-copy", ) with gr.Row(equal_height=False): with gr.Column(scale=4, min_width=300, elem_classes="os-window"): gr.HTML('
New Object
') prompt = gr.Textbox( label="Object description", placeholder="Example: a translucent tangerine desktop computer from 1999…", lines=6, max_lines=9, max_length=MAX_PROMPT_LENGTH, autofocus=True, ) detail = gr.Radio( choices=list(DETAIL_BUDGETS), value="Balanced · 20–36 parts", label="Geometry budget", ) finish = gr.Dropdown( choices=[ "Faithful product visualization", "Playful low-poly diorama", "Polished retro-futurist collectible", "Industrial design maquette", ], value="Faithful product visualization", label="Art direction", ) build = gr.Button("Build 3D Object", variant="primary") clear = gr.ClearButton([prompt], value="Clear") with gr.Column(scale=7, min_width=420, elem_classes="os-window"): gr.HTML('
Untitled 3D View
') preview = gr.HTML(render_preview(WELCOME_SCENE), elem_id="preview-frame") status = gr.Markdown( "**Ready.** The preview runs in a sandboxed iframe; generated code is never executed.", elem_id="status-strip", ) with gr.Accordion("Scene specification · JSON", open=False, elem_classes="os-window"): spec = gr.Code(label="Normalized safe scene", language="json", interactive=False, lines=22) with gr.Column(elem_classes="os-window"): gr.HTML('
Try an Example
') gr.Examples( examples=[ [ "A translucent tangerine all-in-one desktop computer from 1999, with a curved CRT shell, handle, keyboard, round mouse, vents, ports, and softly glowing screen", "Intricate · 36–64 parts", "Faithful product visualization", ], [ "A compact yellow construction excavator with tracked undercarriage, articulated boom, hydraulic cylinders, cab glass, bucket teeth, and safety rails", "Balanced · 20–36 parts", "Industrial design maquette", ], [ "A tabletop brass orrery with the sun, six planets, fine orbital rings, geared supports, and a dark walnut base", "Intricate · 36–64 parts", "Polished retro-futurist collectible", ], [ "A tiny cedar cabin diorama with a steep roof, stone chimney, porch, glowing windows, pine trees, and stepping stones", "Balanced · 20–36 parts", "Playful low-poly diorama", ], ], inputs=[prompt, detail, finish], outputs=[preview, spec, status], fn=generate_scene, cache_examples=True, cache_mode="lazy", ) gr.HTML( '" ) generation = build.click( fn=generate_scene, inputs=[prompt, detail, finish], outputs=[preview, spec, status], api_name="generate", api_description="Generate a normalized procedural Three.js scene from text.", show_progress="full", concurrency_limit=3, ) prompt.submit( fn=generate_scene, inputs=[prompt, detail, finish], outputs=[preview, spec, status], api_name="_submit", api_visibility="private", show_progress="full", concurrency_limit=3, ) if __name__ == "__main__": demo.queue(default_concurrency_limit=3, max_size=24).launch(css=CSS)