PathogenAgent / app.py
Sepideh2027's picture
Update app.py
0084dce verified
Raw History Blame Contribute Delete
4.05 kB
# ==========================================================
# PATHOGENAGENT - DEMO WITH MODULES 1 & 2
# ==========================================================
import gradio as gr
import json
import random
from transformers import AutoTokenizer, AutoModelForCausalLM
# ✅ Import ماژول‌ها
from modules.module_1_intent_router import IntentRouter
from modules.module_2_tool_executor import ToolExecutor
# ==========================================================
# ۱. بارگذاری دیتاست
# ==========================================================
def load_dataset():
try:
with open("biomedical_10k_dataset.json", "r", encoding="utf-8") as f:
data = json.load(f)
return data.get("questions", [])
except:
return []
QUESTIONS = load_dataset()
print(f"✅ {len(QUESTIONS)} questions loaded")
# ==========================================================
# ۲. راه‌اندازی ماژول‌ها و مدل
# ==========================================================
intent_router = IntentRouter()
tool_executor = ToolExecutor()
print("🔄 Loading BioGPT...")
MODEL_NAME = "Sepideh2027/biogpt-clinvar-finetuned"
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
model = AutoModelForCausalLM.from_pretrained(MODEL_NAME)
print("✅ Model loaded!")
# ==========================================================
# ۳. تابع اصلی
# ==========================================================
def get_random_question():
if QUESTIONS:
return random.choice(QUESTIONS).get("question", "")
return "What is the clinical significance of CFTR F508del?"
def run_agent(query, use_random=False):
if not query or use_random:
query = get_random_question()
# Step 1: Intent Detection (ماژول ۱)
intent_result = intent_router.detect_intent(query)
intent = intent_result["intent"]
tools = intent_result["required_tools"]
confidence = intent_result["confidence"]
# Step 2: Evidence Retrieval (ماژول ۲)
evidence = tool_executor.execute(query, tools)
# Step 3: BioGPT Generation
inputs = tokenizer(query, return_tensors="pt", truncation=True, max_length=512)
outputs = model.generate(**inputs, max_new_tokens=100)
response = tokenizer.decode(outputs[0], skip_special_tokens=True)
if query in response:
response = response.split(query)[-1].strip()
# Step 4: ساخت خروجی
evidence_text = ""
for source, items in evidence.items():
if items:
evidence_text += f"\n- **{source}:** " + ", ".join([i.get("title", "") for i in items[:3]])
else:
evidence_text += f"\n- **{source}:** No results"
return f"""## 🧬 PathogenAgent
**Question:** {query}
**Intent:** {intent} (confidence: {confidence:.2f})
**Tools Used:** {', '.join(tools)}
**Evidence:** {evidence_text}
**Answer:** {response}
---
*Powered by BioGPT + RAG*
"""
# ==========================================================
# ۴. رابط Gradio
# ==========================================================
with gr.Blocks(title="PathogenAgent", theme=gr.themes.Soft()) as demo:
gr.Markdown("""
# 🧬 PathogenAgent
### Evidence-Grounded AI Agent for Pathogen Genomics
**BioGPT + RAG + PubMed/ClinVar/GenBank**
""")
with gr.Row():
query_input = gr.Textbox(
label="🔬 Enter your question",
placeholder="e.g., What is the clinical significance of CFTR F508del?",
lines=3,
value="What is the clinical significance of CFTR F508del?"
)
with gr.Row():
submit_btn = gr.Button("🚀 Run", variant="primary")
random_btn = gr.Button("🎲 Random", variant="secondary")
output = gr.Markdown(label="📝 Response")
submit_btn.click(fn=run_agent, inputs=[query_input], outputs=[output])
random_btn.click(fn=lambda: run_agent("", True), inputs=[], outputs=[output])
if __name__ == "__main__":
demo.launch()