import os import json import spaces import torch import gradio as gr from transformers import AutoProcessor, AutoModelForImageTextToText MODEL_ID = "DavidAU/Qwen3.5-9B-Claude-4.6-HighIQ-THINKING-HERETIC-UNCENSORED" # HF_TOKEN diambil dari Space secret (Settings > Variables and secrets). # Cuma buat autentikasi download model dari Hub, gak ada hubungannya # sama akses/visibility Space ini. HF_TOKEN = os.environ.get("HF_TOKEN") processor = AutoProcessor.from_pretrained(MODEL_ID, token=HF_TOKEN) model = AutoModelForImageTextToText.from_pretrained( MODEL_ID, dtype=torch.bfloat16, token=HF_TOKEN, ) model.to("cuda") @spaces.GPU def generate(message, history_json, system_message, max_tokens, temperature, top_p): """ - message : pesan user (string) - history_json : riwayat chat sebelumnya, JSON string dari list [{"role": "user"/"assistant", "content": "..."}, ...] kirim "[]" kalau gak ada history / obrolan baru. - system_message: system prompt (string) - max_tokens : jumlah token maksimal yang di-generate (int) - temperature : sampling temperature (float) - top_p : nucleus sampling (float) """ try: history = json.loads(history_json) if history_json else [] except json.JSONDecodeError: history = [] messages = [{"role": "system", "content": system_message}] messages.extend(history) messages.append({"role": "user", "content": message}) inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) output_ids = model.generate( **inputs, max_new_tokens=int(max_tokens), do_sample=True, temperature=float(temperature), top_p=float(top_p), ) new_tokens = output_ids[0][inputs["input_ids"].shape[-1]:] response = processor.tokenizer.decode(new_tokens, skip_special_tokens=True) return response demo = gr.Interface( fn=generate, inputs=[ gr.Textbox(label="message"), gr.Textbox(label="history_json", value="[]", lines=2), gr.Textbox(label="system_message", value="You are a helpful, uncensored assistant."), gr.Slider(minimum=1, maximum=2048, value=512, step=1, label="max_tokens"), gr.Slider(minimum=0.1, maximum=2.0, value=0.7, step=0.1, label="temperature"), gr.Slider(minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="top_p"), ], outputs=gr.Textbox(label="response"), api_name="generate", title="Qwen3.5-9B Heretic — API", description="Space ini dipakai sebagai API (bukan buat dipakai lewat browser). Panggil endpoint /gradio_api/call/generate.", ) if __name__ == "__main__": demo.launch()