Tes / app.py
grayd26's picture
Update app.py
e9d7c8a verified
Raw History Blame Contribute Delete
2.84 kB
import os
import json
import spaces
import torch
import gradio as gr
from transformers import AutoProcessor, AutoModelForImageTextToText
MODEL_ID = "DavidAU/Qwen3.5-9B-Claude-4.6-HighIQ-THINKING-HERETIC-UNCENSORED"
# HF_TOKEN diambil dari Space secret (Settings > Variables and secrets).
# Cuma buat autentikasi download model dari Hub, gak ada hubungannya
# sama akses/visibility Space ini.
HF_TOKEN = os.environ.get("HF_TOKEN")
processor = AutoProcessor.from_pretrained(MODEL_ID, token=HF_TOKEN)
model = AutoModelForImageTextToText.from_pretrained(
MODEL_ID,
dtype=torch.bfloat16,
token=HF_TOKEN,
)
model.to("cuda")
@spaces.GPU
def generate(message, history_json, system_message, max_tokens, temperature, top_p):
"""
- message : pesan user (string)
- history_json : riwayat chat sebelumnya, JSON string dari list
[{"role": "user"/"assistant", "content": "..."}, ...]
kirim "[]" kalau gak ada history / obrolan baru.
- system_message: system prompt (string)
- max_tokens : jumlah token maksimal yang di-generate (int)
- temperature : sampling temperature (float)
- top_p : nucleus sampling (float)
"""
try:
history = json.loads(history_json) if history_json else []
except json.JSONDecodeError:
history = []
messages = [{"role": "system", "content": system_message}]
messages.extend(history)
messages.append({"role": "user", "content": message})
inputs = processor.apply_chat_template(
messages,
add_generation_prompt=True,
tokenize=True,
return_dict=True,
return_tensors="pt",
).to(model.device)
output_ids = model.generate(
**inputs,
max_new_tokens=int(max_tokens),
do_sample=True,
temperature=float(temperature),
top_p=float(top_p),
)
new_tokens = output_ids[0][inputs["input_ids"].shape[-1]:]
response = processor.tokenizer.decode(new_tokens, skip_special_tokens=True)
return response
demo = gr.Interface(
fn=generate,
inputs=[
gr.Textbox(label="message"),
gr.Textbox(label="history_json", value="[]", lines=2),
gr.Textbox(label="system_message", value="You are a helpful, uncensored assistant."),
gr.Slider(minimum=1, maximum=2048, value=512, step=1, label="max_tokens"),
gr.Slider(minimum=0.1, maximum=2.0, value=0.7, step=0.1, label="temperature"),
gr.Slider(minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="top_p"),
],
outputs=gr.Textbox(label="response"),
api_name="generate",
title="Qwen3.5-9B Heretic — API",
description="Space ini dipakai sebagai API (bukan buat dipakai lewat browser). Panggil endpoint /gradio_api/call/generate.",
)
if __name__ == "__main__":
demo.launch()