# ========================================================== # PATHOGENAGENT - DEMO WITH MODULES 1 & 2 # ========================================================== import gradio as gr import json import random from transformers import AutoTokenizer, AutoModelForCausalLM # ✅ Import ماژول‌ها from modules.module_1_intent_router import IntentRouter from modules.module_2_tool_executor import ToolExecutor # ========================================================== # ۱. بارگذاری دیتاست # ========================================================== def load_dataset(): try: with open("biomedical_10k_dataset.json", "r", encoding="utf-8") as f: data = json.load(f) return data.get("questions", []) except: return [] QUESTIONS = load_dataset() print(f"✅ {len(QUESTIONS)} questions loaded") # ========================================================== # ۲. راه‌اندازی ماژول‌ها و مدل # ========================================================== intent_router = IntentRouter() tool_executor = ToolExecutor() print("🔄 Loading BioGPT...") MODEL_NAME = "Sepideh2027/biogpt-clinvar-finetuned" tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME) model = AutoModelForCausalLM.from_pretrained(MODEL_NAME) print("✅ Model loaded!") # ========================================================== # ۳. تابع اصلی # ========================================================== def get_random_question(): if QUESTIONS: return random.choice(QUESTIONS).get("question", "") return "What is the clinical significance of CFTR F508del?" def run_agent(query, use_random=False): if not query or use_random: query = get_random_question() # Step 1: Intent Detection (ماژول ۱) intent_result = intent_router.detect_intent(query) intent = intent_result["intent"] tools = intent_result["required_tools"] confidence = intent_result["confidence"] # Step 2: Evidence Retrieval (ماژول ۲) evidence = tool_executor.execute(query, tools) # Step 3: BioGPT Generation inputs = tokenizer(query, return_tensors="pt", truncation=True, max_length=512) outputs = model.generate(**inputs, max_new_tokens=100) response = tokenizer.decode(outputs[0], skip_special_tokens=True) if query in response: response = response.split(query)[-1].strip() # Step 4: ساخت خروجی evidence_text = "" for source, items in evidence.items(): if items: evidence_text += f"\n- **{source}:** " + ", ".join([i.get("title", "") for i in items[:3]]) else: evidence_text += f"\n- **{source}:** No results" return f"""## 🧬 PathogenAgent **Question:** {query} **Intent:** {intent} (confidence: {confidence:.2f}) **Tools Used:** {', '.join(tools)} **Evidence:** {evidence_text} **Answer:** {response} --- *Powered by BioGPT + RAG* """ # ========================================================== # ۴. رابط Gradio # ========================================================== with gr.Blocks(title="PathogenAgent", theme=gr.themes.Soft()) as demo: gr.Markdown(""" # 🧬 PathogenAgent ### Evidence-Grounded AI Agent for Pathogen Genomics **BioGPT + RAG + PubMed/ClinVar/GenBank** """) with gr.Row(): query_input = gr.Textbox( label="🔬 Enter your question", placeholder="e.g., What is the clinical significance of CFTR F508del?", lines=3, value="What is the clinical significance of CFTR F508del?" ) with gr.Row(): submit_btn = gr.Button("🚀 Run", variant="primary") random_btn = gr.Button("🎲 Random", variant="secondary") output = gr.Markdown(label="📝 Response") submit_btn.click(fn=run_agent, inputs=[query_input], outputs=[output]) random_btn.click(fn=lambda: run_agent("", True), inputs=[], outputs=[output]) if __name__ == "__main__": demo.launch()