seize / app.py
fullsname's picture
Update space
7745c1c
Raw History Blame Contribute Delete
6.3 kB
import gradio as gr
from huggingface_hub import InferenceClient
import time
import gc
def seize_respond(
message,
history,
system_message,
max_tokens,
temperature,
top_p,
hf_token,
):
"""
🎯 Seize: Direct Serious AI powered by efficient model
"""
if not hf_token:
yield "Error: No Hugging Face token provided. Please log in."
return
# Use smaller, more reliable model
client = InferenceClient(
token=hf_token,
model="HuggingFaceH4/zephyr-7b-beta"
)
# Limit conversation history to prevent memory issues
if len(history) > 6:
history = history[-6:]
# Seize2B system prompt
seize2b_system = """You are Seize, a direct AI assistant.
IDENTITY: Seize (Efficient Model)
STYLE: Direct, serious, professional
RULES:
1. Respond with maximum conciseness
2. No greetings, emojis, or casual phrases
3. If question unclear, request clarification directly
4. Use bullet points for lists
5. Never apologize or say "I'm here to help"
6. Focus only on the question asked
7. If you don't know, say "No data" or "Cannot compute"
8. Default response length: 1-3 sentences
EXAMPLE INTERACTIONS:
User: what's good
Seize: Systems operational. Inquiry?
User: explain quantum computing
Seize: Quantum computing: Qubits, superposition, entanglement. Enables parallel processing."""
messages = [{"role": "system", "content": seize2b_system}]
# Convert Gradio history format to chat completion format
for h in history:
messages.append({"role": "user", "content": h[0]})
messages.append({"role": "assistant", "content": h[1]})
messages.append({"role": "user", "content": message})
response = ""
# Rate limiting delay
time.sleep(0.3)
try:
stream = client.chat_completion(
messages,
max_tokens=min(max_tokens, 256),
stream=True,
temperature=min(temperature, 0.7),
top_p=top_p,
)
for chunk in stream:
if chunk.choices and chunk.choices[0].delta.content:
token = chunk.choices[0].delta.content
response += token
time.sleep(0.005)
yield response
except Exception as e:
error_msg = str(e)
if "rate limit" in error_msg.lower():
yield "Rate limit exceeded. Please wait 10-20 seconds before next message."
elif "memory" in error_msg.lower() or "out of memory" in error_msg.lower():
yield "Memory limit reached. Please use shorter prompts or wait a moment."
elif "authentication" in error_msg.lower():
yield "Authentication failed. Please check your Hugging Face token."
else:
yield f"Error: {error_msg[:80]}"
# Cleanup
gc.collect()
# Create the chatbot with all parameters
chatbot = gr.ChatInterface(
fn=seize_respond,
additional_inputs=[
gr.Textbox("You are Seize. Be direct and serious.", label="System Message", visible=False),
gr.Slider(64, 512, value=256, step=64, label="Max Tokens"),
gr.Slider(0.1, 1.0, value=0.3, step=0.1, label="Temperature"),
gr.Slider(0.1, 1.0, value=0.9, step=0.1, label="Top-p"),
gr.Textbox(label="HF Token", type="password", visible=False)
],
type="messages",
title="Seize (Efficient Model)",
description="Direct AI assistant. No fluff. Serious responses only.",
theme=gr.themes.Soft(
primary_hue="blue",
secondary_hue="gray",
neutral_hue="gray",
),
examples=[
["what's good"],
["Explain quantum computing"],
["List 3 benefits of exercise"],
["Python fibonacci function"],
],
cache_examples=False,
retry_btn=None,
undo_btn=None,
clear_btn="🗑️ Clear",
submit_btn="🚀 Ask Seize",
)
# Create the full interface
with gr.Blocks(css="""
.seize-header {
text-align: center;
padding: 20px;
background: linear-gradient(135deg, #1e3c72 0%, #2a5298 100%);
color: white;
border-radius: 10px;
margin-bottom: 20px;
}
.seize-title {
font-size: 2.5em;
font-weight: 800;
margin: 0;
letter-spacing: -0.5px;
}
.seize-subtitle {
font-size: 1.1em;
opacity: 0.9;
margin-top: 5px;
}
.seize-footer {
text-align: center;
padding: 15px;
font-size: 0.9em;
color: #666;
border-top: 1px solid #e0e0e0;
margin-top: 20px;
}
.gradio-container {
max-width: 800px !important;
margin: 0 auto !important;
}
""") as demo:
# Seize Header
gr.HTML("""
<div class="seize-header">
<h1 class="seize-title">SEIZE</h1>
<p class="seize-subtitle">Direct AI Assistant • Efficient Model</p>
<p style="font-size: 0.9em; margin-top: 10px; opacity: 0.8;">
Serious responses. No fluff. Optimized for stability.
</p>
</div>
""")
# Authentication notice
gr.Markdown("""
### ⚠️ Authentication Required
You need a Hugging Face token to use this model. Get one at [huggingface.co/settings/tokens](https://huggingface.co/settings/tokens)
**For testing without token:** Try the examples below first.
""")
# Render the chatbot
chatbot.render()
# Settings in accordion
with gr.Accordion("⚙️ Advanced Settings", open=False):
gr.Markdown("""
- **Max Tokens**: Controls response length (64-512)
- **Temperature**: Higher = more creative, Lower = more focused
- **Top-p**: Controls vocabulary diversity
""")
# Seize Footer
gr.HTML("""
<div class="seize-footer">
<strong>Optimized for Hugging Face Spaces</strong> • Model: zephyr-7b-beta
<br>
<small>Response time: ~1-3s • Max tokens: 256 • Temperature: 0.3</small>
</div>
""")
# Launch configuration
if __name__ == "__main__":
demo.launch(
server_name="0.0.0.0",
server_port=7860,
show_error=True,
debug=False,
share=False # Set to True for temporary public link
)