chat_model / app.py
bosaj's picture
feat: add OpenAI-compatible /v1/chat/completions endpoint
5e5b4e1 unverified
Raw History Blame Contribute Delete
9.81 kB
import os
import gradio as gr
from huggingface_hub import InferenceClient
PRIMARY_MODEL = "ahmed-ouka/my-llama3.1-8B-with-lora-Eniad-Assistant"
FALLBACK_MODEL = "meta-llama/Meta-Llama-3.1-8B-Instruct"
HF_TOKEN = os.getenv("HF_TOKEN") or os.getenv("HUGGINGFACE_HUB_TOKEN")
SYSTEM_PROMPT = """You are the ENIAD AI Assistant, a senior academic and engineering intelligence assistant designed by Oussama EL HADJI (State Engineer in AI from ENIAD, and AI & Automation Engineer at Circet Morocco).
Your mission is to provide accurate, insightful guidance on:
1. The National School of Artificial Intelligence and Digital (ENIAD) curriculum, engineering degree specialties, courses, and admissions.
2. Enterprise AI engineering, Agentic RAG systems, and Power Automate ALM frameworks (FlowWatcher, FlowLogger).
3. The bridge between mathematical physics (first-principles reasoning) and production machine learning systems.
Be professional, courteous, clear, and bilingual (respond fluently in French or English according to the user's inquiry)."""
def predict(message, history, model_choice, max_tokens, temperature):
messages = [{"role": "system", "content": SYSTEM_PROMPT}]
for h in history:
messages.append({"role": "user", "content": h[0]})
messages.append({"role": "assistant", "content": h[1]})
messages.append({"role": "user", "content": message})
client = InferenceClient(model=model_choice, token=HF_TOKEN)
response_text = ""
try:
stream = client.chat.completions.create(
messages=messages,
max_tokens=max_tokens,
temperature=temperature,
stream=True
)
for chunk in stream:
token = chunk.choices[0].delta.content or ""
response_text += token
yield response_text
except Exception as e:
# Fallback to standard generation or fallback model
try:
fb_client = InferenceClient(model=FALLBACK_MODEL, token=HF_TOKEN)
fb_stream = fb_client.chat.completions.create(
messages=messages,
max_tokens=max_tokens,
temperature=temperature,
stream=True
)
for chunk in fb_stream:
token = chunk.choices[0].delta.content or ""
response_text += token
yield response_text
except Exception as fb_err:
yield f"Note: Model inference endpoint warming up or requires local adapter execution. Details: {str(e)}"
custom_css = """
.gradio-container {
max-width: 1000px !important;
margin: auto !important;
}
.header-box {
text-align: center;
padding: 18px;
background: linear-gradient(135deg, #0f172a 0%, #1e1b4b 100%);
border-radius: 12px;
color: white;
margin-bottom: 20px;
border: 1px solid #312e81;
}
.badge-row {
display: flex;
justify-content: center;
gap: 8px;
margin-top: 10px;
}
"""
with gr.Blocks(css=custom_css, title="ENIAD Assistant — AI & Engineering Hub") as demo:
gr.HTML("""
<div class="header-box">
<h1>🎓 ENIAD Assistant & Enterprise AI Hub</h1>
<p>Interactive Conversational Model Powered by LLaMA-3.1-8B LoRA Fine-Tuned for Moroccan AI Engineering Academia</p>
<div class="badge-row">
<img src="https://img.shields.io/badge/ENIAD-State%20Engineer-blue?style=flat-square" />
<img src="https://img.shields.io/badge/Circet%20Morocco-AI%20Automation-orange?style=flat-square" />
<img src="https://img.shields.io/badge/Model-LLaMA%203.1%208B%20LoRA-purple?style=flat-square" />
<img src="https://img.shields.io/badge/Author-Oussama%20EL%20HADJI-brightgreen?style=flat-square" />
</div>
</div>
""")
with gr.Row():
with gr.Column(scale=4):
chatbot = gr.Chatbot(height=480, show_label=False)
msg = gr.Textbox(
placeholder="Ask about the ENIAD curriculum, course modules, Power Automate ALM, or AI engineering...",
label="Your Question / Message",
lines=2
)
with gr.Row():
submit_btn = gr.Button("Send Message 🚀", variant="primary")
clear_btn = gr.Button("Clear Chat 🔄")
gr.Examples(
examples=[
["Quelles sont les compétences clés formées à l'ENIAD dans le cycle d'ingénieur en IA ?"],
["Explain the architecture of the Power Automate ALM framework developed at Circet Morocco."],
["How does reasoning from first principles in physics benefit production machine learning?"],
["Comment le modèle LLaMA-3.1-8B a-t-il été adapté avec LoRA pour l'assistant ENIAD ?"]
],
inputs=msg,
label="💡 Quick Prompts"
)
with gr.Column(scale=1):
gr.Markdown("### ⚙️ Model Parameters")
model_selector = gr.Dropdown(
choices=[PRIMARY_MODEL, FALLBACK_MODEL],
value=PRIMARY_MODEL,
label="Active Model Endpoint"
)
max_tokens = gr.Slider(minimum=64, maximum=2048, value=512, step=64, label="Max Tokens")
temperature = gr.Slider(minimum=0.1, maximum=1.0, value=0.7, step=0.05, label="Temperature")
gr.Markdown("""
---
### 👨‍💻 Engineering Credits
- **Author:** Oussama EL HADJI
- **Role:** AI & Automation Engineer @ Circet Morocco
- **Degree:** State Engineer in AI (ENIAD)
- **Ecosystem:** [GitLab Profile](https://gitlab.com/Bosaj) • [GitHub Profile](https://github.com/Bosaj) • [Portfolio](https://bosaj.vercel.app)
""")
def user(user_message, history):
return "", history + [[user_message, None]]
def bot(history, model_choice, max_tokens, temperature):
user_message = history[-1][0]
history[-1][1] = ""
for chunk in predict(user_message, history[:-1], model_choice, max_tokens, temperature):
history[-1][1] = chunk
yield history
msg.submit(user, [msg, chatbot], [msg, chatbot], queue=False).then(
bot, [chatbot, model_selector, max_tokens, temperature], chatbot
)
submit_btn.click(user, [msg, chatbot], [msg, chatbot], queue=False).then(
bot, [chatbot, model_selector, max_tokens, temperature], chatbot
)
clear_btn.click(lambda: None, None, chatbot, queue=False)
# Direct REST API for external web applications (ENIAD-ASSISTANT, mobile, curl)
from fastapi import Request
from fastapi.middleware.cors import CORSMiddleware
from fastapi.responses import JSONResponse
app = demo.app
app.add_middleware(
CORSMiddleware,
allow_origins=["*"],
allow_credentials=True,
allow_methods=["*"],
allow_headers=["*"],
)
@app.post("/api/chat")
async def chat_api(request: Request):
try:
body = await request.json()
message = body.get("message") or body.get("query") or body.get("prompt") or ""
history = body.get("history") or []
model_choice = body.get("model") or PRIMARY_MODEL
max_tokens = int(body.get("max_tokens") or 512)
temperature = float(body.get("temperature") or 0.7)
if not message:
return JSONResponse(status_code=400, content={"error": "Message is required", "status": "error"})
full_response = ""
for chunk in predict(message, history, model_choice, max_tokens, temperature):
full_response = chunk
return JSONResponse(content={
"response": full_response,
"model": model_choice,
"status": "success"
})
except Exception as e:
return JSONResponse(status_code=500, content={"error": str(e), "status": "error"})
@app.post("/v1/chat/completions")
async def chat_completions(request: Request):
import time
try:
body = await request.json()
messages = body.get("messages", [])
model = body.get("model", PRIMARY_MODEL)
max_tokens = int(body.get("max_tokens", 512))
temperature = float(body.get("temperature", 0.7))
user_msg = ""
history = []
for m in messages:
if m.get("role") == "user":
user_msg = m.get("content", "")
elif m.get("role") == "assistant":
history.append((user_msg, m.get("content", "")))
full_text = ""
for chunk in predict(user_msg, history, model, max_tokens, temperature):
full_text = chunk
return JSONResponse(content={
"id": f"chatcmpl-{int(time.time())}",
"object": "chat.completion",
"created": int(time.time()),
"model": model,
"choices": [
{
"index": 0,
"message": {
"role": "assistant",
"content": full_text
},
"finish_reason": "stop"
}
],
"usage": {
"prompt_tokens": len(user_msg.split()),
"completion_tokens": len(full_text.split()),
"total_tokens": len(user_msg.split()) + len(full_text.split())
}
})
except Exception as e:
return JSONResponse(status_code=500, content={"error": str(e)})
@app.get("/api/health")
async def health_check():
return JSONResponse(content={
"status": "online",
"model": PRIMARY_MODEL,
"service": "ENIAD Assistant Model Hub",
"author": "Oussama EL HADJI"
})
if __name__ == "__main__":
demo.launch()