import gradio as gr from transformers import AutoModelForCausalLM, AutoTokenizer # Load model directly model_name = "microsoft/phi-2" tokenizer = AutoTokenizer.from_pretrained(model_name) model = AutoModelForCausalLM.from_pretrained( model_name, trust_remote_code=True, device_map="auto" ) # Simple chat history without streaming complexity chat_history = [] def chat(message): global chat_history # Build the prompt from history prompt = "" for user_msg, bot_msg in chat_history: prompt += f"Human: {user_msg}\nAssistant: {bot_msg}\n" prompt += f"Human: {message}\nAssistant:" # Generate response (simple version) inputs = tokenizer(prompt, return_tensors="pt").to(model.device) outputs = model.generate( inputs.input_ids, max_new_tokens=256, temperature=0.7, do_sample=True ) # Get response text response = tokenizer.decode(outputs[0], skip_special_tokens=True) assistant_response = response[len(prompt):].strip() # Update history and return chat_history.append((message, assistant_response)) return chat_history # Create a simple interface with gr.Blocks() as demo: chatbot = gr.Chatbot() msg = gr.Textbox(placeholder="Type your message here...") clear = gr.Button("Clear") msg.submit(chat, msg, chatbot).then(lambda: "", None, msg) clear.click(lambda: [], None, chatbot) clear.click(lambda: [], None, msg) if __name__ == "__main__": demo.launch()