SeizeB / app.py
fullsname's picture
Update space
0a52e79
Raw History Blame Contribute Delete
5.11 kB
import os
import gradio as gr
from huggingface_hub import InferenceClient
# ===== MAIN FUNCTION =====
def seize_respond(message, history, system_message, max_tokens, temperature, top_p):
"""
🎯 Seize: Direct Serious AI powered by Seize2B model
"""
# 1. GET TOKEN FROM ENVIRONMENT (SAFE!)
token = os.environ.get("HF_TOKEN")
if not token:
yield "ERROR: Token not configured. Please set HF_TOKEN secret in Space settings."
return
# 2. INITIALIZE CLIENT - CORRECTED API
client = InferenceClient(model="HuggingFaceH4/zephyr-7b-beta", token=token)
# 3. COMBINE SYSTEM MESSAGE WITH USER MESSAGE
seize2b_system = """You are Seize, powered by the Seize2B model.
IDENTITY: Seize (Seize2B Model)
STYLE: Direct, serious, professional
RULES:
1. Respond with maximum conciseness
2. No greetings, emojis, or casual phrases
3. If question unclear, request clarification directly
4. Use bullet points for lists
5. Never apologize or say "I'm here to help"
6. Focus only on the question asked
7. If you don't know, say "No data" or "Cannot compute"
8. Default response length: 1-3 sentences"""
# 4. PREPARE PROMPT - Combine system and user message
full_prompt = f"""<|system|>
{seize2b_system}</s>
<|user|>
{message}</s>
<|assistant|>
"""
# 5. GET RESPONSE - CORRECTED API CALL
try:
stream = client.text_generation(
prompt=full_prompt,
max_new_tokens=max_tokens,
stream=True,
temperature=temperature,
top_p=top_p,
repetition_penalty=1.1
)
response = ""
for chunk in stream:
if chunk:
response += chunk
yield response
except Exception as e:
yield f"Error: {str(e)}"
# ===== GRADIO INTERFACE =====
with gr.Blocks(
title="Seize (Seize2B Model)",
theme=gr.themes.Soft(
primary_hue="blue",
secondary_hue="gray",
neutral_hue="gray"
)
) as demo:
# HEADER
gr.Markdown("""
# 🚀 Seize
### Direct AI Assistant • Seize2B Model
*Serious responses. No fluff.*
---
""")
# CHATBOT
chatbot = gr.Chatbot(height=400, type="messages")
# INPUTS
with gr.Row():
msg = gr.Textbox(
placeholder="Ask Seize...",
scale=7,
container=False
)
submit_btn = gr.Button("🚀 Ask", scale=1, variant="primary")
# CONTROLS (HIDDEN BY DEFAULT)
with gr.Accordion("⚙️ Advanced Settings", open=False):
with gr.Row():
system_msg = gr.Textbox(
label="System Message",
value="""You are Seize, powered by the Seize2B model.
IDENTITY: Seize (Seize2B Model)
STYLE: Direct, serious, professional""",
lines=3
)
with gr.Row():
max_tokens = gr.Slider(50, 1000, value=512, label="Max Tokens")
temperature = gr.Slider(0.1, 2.0, value=0.3, label="Temperature")
top_p = gr.Slider(0.1, 1.0, value=0.9, label="Top-p")
# EXAMPLE QUESTIONS
gr.Examples(
examples=[
"what's good",
"Explain quantum computing",
"List 3 benefits of exercise",
"Write Python fibonacci function",
"What is Seize2B?"
],
inputs=msg,
label="💡 Try these questions"
)
# CLEAR BUTTON
clear_btn = gr.Button("🗑️ Clear Chat", variant="secondary")
# ===== EVENT HANDLERS =====
def respond(message, chat_history):
"""Handle message submission and get AI response"""
if not message.strip():
yield chat_history
return
# Add user message to chat history
chat_history.append({"role": "user", "content": message})
# Create a placeholder for assistant response
chat_history.append({"role": "assistant", "content": ""})
yield chat_history
# Get AI response
full_response = ""
# Use the system message from the UI if provided
system_prompt = system_msg.value if system_msg.value else ""
for chunk in seize_respond(
message=message,
history=chat_history,
system_message=system_prompt,
max_tokens=int(max_tokens.value),
temperature=float(temperature.value),
top_p=float(top_p.value)
):
if chunk:
full_response = chunk
# Update the last message in chat history
chat_history[-1]["content"] = full_response
yield chat_history
def clear_chat():
"""Clear chat history"""
return [], ""
# CONNECT EVERYTHING
msg.submit(respond, [msg], [chatbot])
submit_btn.click(respond, [msg], [chatbot])
clear_btn.click(clear_chat, None, [chatbot, msg], queue=False)
# ===== LAUNCH =====
if __name__ == "__main__":
demo.launch(debug=True)