Download app.py from grayd26/Tes: direct link, hf CLI and curl.
- Browser
- Download file 2.84 kB
-
https://huggingface.co/spaces/grayd26/Tes/resolve/main/app.py
- Command line
-
hf download hf://spaces/grayd26/Tes/app.py
-
curl -L -o app.py https://huggingface.co/spaces/grayd26/Tes/resolve/main/app.py
2.84 kB
| import os | |
| import json | |
| import spaces | |
| import torch | |
| import gradio as gr | |
| from transformers import AutoProcessor, AutoModelForImageTextToText | |
| MODEL_ID = "DavidAU/Qwen3.5-9B-Claude-4.6-HighIQ-THINKING-HERETIC-UNCENSORED" | |
| # HF_TOKEN diambil dari Space secret (Settings > Variables and secrets). | |
| # Cuma buat autentikasi download model dari Hub, gak ada hubungannya | |
| # sama akses/visibility Space ini. | |
| HF_TOKEN = os.environ.get("HF_TOKEN") | |
| processor = AutoProcessor.from_pretrained(MODEL_ID, token=HF_TOKEN) | |
| model = AutoModelForImageTextToText.from_pretrained( | |
| MODEL_ID, | |
| dtype=torch.bfloat16, | |
| token=HF_TOKEN, | |
| ) | |
| model.to("cuda") | |
| def generate(message, history_json, system_message, max_tokens, temperature, top_p): | |
| """ | |
| - message : pesan user (string) | |
| - history_json : riwayat chat sebelumnya, JSON string dari list | |
| [{"role": "user"/"assistant", "content": "..."}, ...] | |
| kirim "[]" kalau gak ada history / obrolan baru. | |
| - system_message: system prompt (string) | |
| - max_tokens : jumlah token maksimal yang di-generate (int) | |
| - temperature : sampling temperature (float) | |
| - top_p : nucleus sampling (float) | |
| """ | |
| try: | |
| history = json.loads(history_json) if history_json else [] | |
| except json.JSONDecodeError: | |
| history = [] | |
| messages = [{"role": "system", "content": system_message}] | |
| messages.extend(history) | |
| messages.append({"role": "user", "content": message}) | |
| inputs = processor.apply_chat_template( | |
| messages, | |
| add_generation_prompt=True, | |
| tokenize=True, | |
| return_dict=True, | |
| return_tensors="pt", | |
| ).to(model.device) | |
| output_ids = model.generate( | |
| **inputs, | |
| max_new_tokens=int(max_tokens), | |
| do_sample=True, | |
| temperature=float(temperature), | |
| top_p=float(top_p), | |
| ) | |
| new_tokens = output_ids[0][inputs["input_ids"].shape[-1]:] | |
| response = processor.tokenizer.decode(new_tokens, skip_special_tokens=True) | |
| return response | |
| demo = gr.Interface( | |
| fn=generate, | |
| inputs=[ | |
| gr.Textbox(label="message"), | |
| gr.Textbox(label="history_json", value="[]", lines=2), | |
| gr.Textbox(label="system_message", value="You are a helpful, uncensored assistant."), | |
| gr.Slider(minimum=1, maximum=2048, value=512, step=1, label="max_tokens"), | |
| gr.Slider(minimum=0.1, maximum=2.0, value=0.7, step=0.1, label="temperature"), | |
| gr.Slider(minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="top_p"), | |
| ], | |
| outputs=gr.Textbox(label="response"), | |
| api_name="generate", | |
| title="Qwen3.5-9B Heretic — API", | |
| description="Space ini dipakai sebagai API (bukan buat dipakai lewat browser). Panggil endpoint /gradio_api/call/generate.", | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch() | |