import gradio as gr from huggingface_hub import InferenceClient import time import gc def seize_respond( message, history, system_message, max_tokens, temperature, top_p, hf_token, ): """ 🎯 Seize: Direct Serious AI powered by efficient model """ if not hf_token: yield "Error: No Hugging Face token provided. Please log in." return # Use smaller, more reliable model client = InferenceClient( token=hf_token, model="HuggingFaceH4/zephyr-7b-beta" ) # Limit conversation history to prevent memory issues if len(history) > 6: history = history[-6:] # Seize2B system prompt seize2b_system = """You are Seize, a direct AI assistant. IDENTITY: Seize (Efficient Model) STYLE: Direct, serious, professional RULES: 1. Respond with maximum conciseness 2. No greetings, emojis, or casual phrases 3. If question unclear, request clarification directly 4. Use bullet points for lists 5. Never apologize or say "I'm here to help" 6. Focus only on the question asked 7. If you don't know, say "No data" or "Cannot compute" 8. Default response length: 1-3 sentences EXAMPLE INTERACTIONS: User: what's good Seize: Systems operational. Inquiry? User: explain quantum computing Seize: Quantum computing: Qubits, superposition, entanglement. Enables parallel processing.""" messages = [{"role": "system", "content": seize2b_system}] # Convert Gradio history format to chat completion format for h in history: messages.append({"role": "user", "content": h[0]}) messages.append({"role": "assistant", "content": h[1]}) messages.append({"role": "user", "content": message}) response = "" # Rate limiting delay time.sleep(0.3) try: stream = client.chat_completion( messages, max_tokens=min(max_tokens, 256), stream=True, temperature=min(temperature, 0.7), top_p=top_p, ) for chunk in stream: if chunk.choices and chunk.choices[0].delta.content: token = chunk.choices[0].delta.content response += token time.sleep(0.005) yield response except Exception as e: error_msg = str(e) if "rate limit" in error_msg.lower(): yield "Rate limit exceeded. Please wait 10-20 seconds before next message." elif "memory" in error_msg.lower() or "out of memory" in error_msg.lower(): yield "Memory limit reached. Please use shorter prompts or wait a moment." elif "authentication" in error_msg.lower(): yield "Authentication failed. Please check your Hugging Face token." else: yield f"Error: {error_msg[:80]}" # Cleanup gc.collect() # Create the chatbot with all parameters chatbot = gr.ChatInterface( fn=seize_respond, additional_inputs=[ gr.Textbox("You are Seize. Be direct and serious.", label="System Message", visible=False), gr.Slider(64, 512, value=256, step=64, label="Max Tokens"), gr.Slider(0.1, 1.0, value=0.3, step=0.1, label="Temperature"), gr.Slider(0.1, 1.0, value=0.9, step=0.1, label="Top-p"), gr.Textbox(label="HF Token", type="password", visible=False) ], type="messages", title="Seize (Efficient Model)", description="Direct AI assistant. No fluff. Serious responses only.", theme=gr.themes.Soft( primary_hue="blue", secondary_hue="gray", neutral_hue="gray", ), examples=[ ["what's good"], ["Explain quantum computing"], ["List 3 benefits of exercise"], ["Python fibonacci function"], ], cache_examples=False, retry_btn=None, undo_btn=None, clear_btn="🗑️ Clear", submit_btn="🚀 Ask Seize", ) # Create the full interface with gr.Blocks(css=""" .seize-header { text-align: center; padding: 20px; background: linear-gradient(135deg, #1e3c72 0%, #2a5298 100%); color: white; border-radius: 10px; margin-bottom: 20px; } .seize-title { font-size: 2.5em; font-weight: 800; margin: 0; letter-spacing: -0.5px; } .seize-subtitle { font-size: 1.1em; opacity: 0.9; margin-top: 5px; } .seize-footer { text-align: center; padding: 15px; font-size: 0.9em; color: #666; border-top: 1px solid #e0e0e0; margin-top: 20px; } .gradio-container { max-width: 800px !important; margin: 0 auto !important; } """) as demo: # Seize Header gr.HTML("""
Direct AI Assistant • Efficient Model
Serious responses. No fluff. Optimized for stability.