import os import re from typing import TypedDict, Dict, Any # ChatGroq is imported lazily in node_xai_report to allow offline/mock testing # TavilyClient and heavy media libs are imported lazily in functions to allow offline/text-only runs from core.tools import hf_image_to_text, tavily_search_tool # Import your local forensic engine and database tools # (Adjust these imports based on where you saved them in your project) from core.media_agent import MediaIntelligenceAgent import json # ========================================== # 1. DEFINE THE STATE DICTIONARY # ========================================== # This acts as the "Memory" passed between every agent class AgentState(TypedDict): media_path: str media_context: str # What is happening in the media? audio_transcript: str # Extracted speech transcript audio_forensic_data: Dict[str, Any] # Voice authenticity analysis (verdict, confidence) forensic_data: Dict[str, Any] # Holds CNN scores and verdicts identity_data: Dict[str, str] # Holds Name and Wikipedia Bio osint_context: str # Holds the live news from Tavily osint_sources: list # URL references gathered during OSINT final_report: str # Holds the final Markdown output # ========================================== # 2. DEFINE THE NODES (The Agents) # ========================================== captioner = None captioner_model = "Salesforce/blip-image-captioning-base" print("āš™ļø Booting Media Intelligence Agent (Audio)...") try: media_agent = MediaIntelligenceAgent() except Exception as e: print(f"🚨 Could not initialize Media Agent: {e}") media_agent = None def _extract_urls(text: str) -> list: if not text: return [] urls = re.findall(r"https?://[^\s\]\[\)\}\"\'<>]+", text) cleaned = [] seen = set() for u in urls: url = u.rstrip(".,;:!?") if url and url not in seen: seen.add(url) cleaned.append(url) return cleaned def _build_osint_queries(state: Dict[str, Any]) -> list: identity = state.get("identity_data", {}) or {} subject_name = identity.get("name", "Unknown") media_context = (state.get("media_context") or "").strip() transcript = (state.get("audio_transcript") or "").strip() queries = [] if subject_name and subject_name != "Unknown": queries.append(f"Latest fact checks, misinformation, deepfake scams, and fraud targeting {subject_name}") queries.append(f"{subject_name} rumor debunked fake news") if transcript and len(transcript) > 15: snippet = transcript[:180] queries.append(f"Fact check transcript claim: {snippet}") if media_context and len(media_context) > 15: queries.append(f"Fact check media claim: {media_context[:180]}") if not queries: queries.append("Latest verified misinformation and media manipulation trends") deduped = [] seen = set() for q in queries: key = q.lower() if key not in seen: seen.add(key) deduped.append(q) return deduped[:3] def _invoke_search_tool(tool_obj, query: str) -> str: """Invoke either plain callables or CrewAI Tool wrappers safely.""" try: if callable(tool_obj): return tool_obj(query) except Exception: pass for attr in ("run", "_run", "func", "invoke"): try: fn = getattr(tool_obj, attr, None) if callable(fn): out = fn(query) if isinstance(out, dict): return json.dumps(out) return str(out) except Exception: continue return "Search tool unavailable." def _truncate_text(text: str, limit: int = 1800) -> str: """Trim text to a safe size for model prompts while preserving the beginning context.""" if not text: return "" text = str(text).strip() if len(text) <= limit: return text return text[:limit].rstrip() + "…" def _compact_lines(text: str, max_lines: int = 18, max_chars: int = 2200) -> str: """Keep only the first useful lines of a large multi-line OSINT block.""" if not text: return "" lines = [ln.rstrip() for ln in str(text).splitlines() if ln.strip()] compact = "\n".join(lines[:max_lines]) return _truncate_text(compact, max_chars) def _build_fallback_report(state: Dict[str, Any], text_only: bool) -> str: """Create a concise deterministic report if the LLM request is too large or unavailable.""" media_context = str(state.get("media_context", "Unknown visual context.")).strip() audio_transcript = str(state.get("audio_transcript", "No audio detected.")).strip() forensic_data = state.get("forensic_data", {}) or {} audio_forensic_data = state.get("audio_forensic_data", {}) or {} osint_context = str(state.get("osint_context", "No OSINT data available.")).strip() sources = state.get("osint_sources", []) or [] refs_md = "\n".join([f"- {u}" for u in sources[:8]]) if sources else "- No verifiable sources found." if text_only: return ( "# THREAT INTELLIGENCE BRIEFING: CLAIM REVIEW\n\n" "## Executive Summary\n" "This is a claim-review fallback generated because the full LLM prompt exceeded service limits. The claim should be treated as unverified until corroborated by the listed sources.\n\n" "## Research Summary\n" f"- Claim: {media_context}\n" f"- OSINT summary: {_truncate_text(osint_context, 700)}\n\n" "## Verification Steps\n" "- Check the listed source URLs directly.\n" "- Compare the claim against reputable outlets and original uploads.\n" "- Verify dates, context, and any manipulated captions or edits.\n\n" "## Threat Implications\n" "If false, the claim could be part of a misinformation or rumor amplification pattern.\n\n" "## References\n" f"{refs_md}\n" ) return ( "# THREAT INTELLIGENCE BRIEFING: MEDIA AUTHENTICITY\n\n" "## 1. Executive Summary\n" "The media appears authentic based on available visual and audio checks, but this fallback report was generated because the LLM prompt exceeded service limits.\n\n" "## 2. Media Context & Claim\n" f"{_truncate_text(media_context, 500)}\n\n" "## 3. Visual Forensics Analysis\n" f"{_truncate_text(str(forensic_data), 700)}\n\n" "## 4. Voice & Audio Forensics\n" f"Transcript: {_truncate_text(audio_transcript, 500)}\n" f"Audio lab: {_truncate_text(str(audio_forensic_data), 500)}\n\n" "## 5. Threat Intelligence & Motive\n" f"{_truncate_text(osint_context, 900)}\n\n" "## 6. References\n" f"{refs_md}\n" ) def _derive_final_verdict(input_type: str, final_report: str, forensic_data: Dict[str, Any], audio_forensic_data: Dict[str, Any]): """Derive the final verdict from actual forensic signals, not report wording.""" if input_type == "text": verdict = "šŸ“„ CLAIM REVIEW" confidence = 100.0 reasoning = final_report.split("\n\n")[0][:300] if final_report else "See report for details" xai_breakdown = "See full report for verification steps and evidence" return verdict, confidence, reasoning, xai_breakdown visual_fake = bool((forensic_data or {}).get("is_fake", False)) visual_conf = float((forensic_data or {}).get("confidence", 0.0) or 0.0) audio_verdict = str((audio_forensic_data or {}).get("verdict", "N/A")).strip().lower() audio_conf = float((audio_forensic_data or {}).get("confidence", 0.0) or 0.0) audio_fake = False if audio_verdict and audio_verdict not in {"n/a", "unknown", "inconclusive"}: suspicious_labels = ("fake", "synthetic", "clone", "ai-generated", "ai generated", "deepfake") audio_fake = any(lbl in audio_verdict for lbl in suspicious_labels) and audio_conf >= 60.0 is_fake = visual_fake or audio_fake if is_fake: verdict = "āš ļø DEEPFAKE DETECTED" confidence = max(visual_conf, audio_conf, 60.0) else: verdict = "āœ… AUTHENTIC" confidence = max(visual_conf, audio_conf, 85.0) confidence = round(min(max(confidence, 0.0), 100.0), 2) reasoning = final_report.split("##")[1][:300] if "##" in final_report else final_report[:300] xai_breakdown = final_report.split("## āš–ļø")[1][:500] if "## āš–ļø" in final_report else "See full report for details" return verdict, confidence, reasoning, xai_breakdown def extract_middle_frame(media_path: str): """Extracts the middle frame of a video and converts it for Hugging Face.""" try: import cv2 from PIL import Image except Exception: raise RuntimeError("OpenCV/Pillow required for frame extraction") cap = cv2.VideoCapture(media_path) total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) cap.set(cv2.CAP_PROP_POS_FRAMES, max(0, total_frames // 2)) ret, frame = cap.read() cap.release() if ret: rgb_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) return Image.fromarray(rgb_frame) return None def node_vision_context(state: AgentState) -> AgentState: """Node 0: Hugging Face BLIP watches the media to understand the plot.""" print("\n[LANGGRAPH] 🟢 Entering Node: Vision Context Extraction (Local)") media_path = state["media_path"] is_video = media_path.lower().endswith((".mp4", ".avi", ".mov")) if not media_path or not os.path.exists(media_path): return {"media_context": "Could not extract visual frame from media."} if is_video: pil_image = extract_middle_frame(media_path) else: try: from PIL import Image pil_image = Image.open(media_path).convert("RGB") except Exception: pil_image = None if not pil_image: return {"media_context": "Could not extract visual frame from media."} # If we have a local pipeline (not created here) it would be used; otherwise prefer hosted HF inference print(" -> šŸ‘ļø Extracting visual context...") try: if captioner is not None: result = captioner(pil_image, max_new_tokens=50) caption = result[0]["generated_text"] else: if os.getenv("HUGGINGFACE_API_KEY"): caption = hf_image_to_text(captioner_model, pil_image, max_new_tokens=50) else: return {"media_context": "No captioner available (set HUGGINGFACE_API_KEY to enable hosted captions)."} formatted_caption = f"The media shows: {caption.capitalize()}." print(f" -> šŸ“ Context Extracted: {formatted_caption}") return {"media_context": formatted_caption} except Exception as e: print(f" -> āš ļø Vision Error: {e}") return {"media_context": "Failed to extract scene context."} def node_audio_analysis(state: AgentState) -> AgentState: """Node 1: Extract transcript and run voice deepfake detection.""" print("\n[LANGGRAPH] 🟢 Entering Node: Audio Analysis") media_path = state.get("media_path", "") if not media_path or not os.path.exists(media_path): return { "audio_transcript": "No audio detected.", "audio_forensic_data": {"verdict": "N/A", "confidence": 0.0} } if media_agent is None: return { "audio_transcript": "Audio agent unavailable.", "audio_forensic_data": {"verdict": "N/A", "confidence": 0.0} } try: print(" -> šŸ”Š Running ASR + voice forensics...") audio_result = media_agent.extract_audio(media_path) # Extract transcript and forensic data separately transcript = audio_result.get("transcript", "No audio detected.") forensic_data = { "verdict": audio_result.get("verdict", "N/A"), "confidence": audio_result.get("confidence", 0.0) } return { "audio_transcript": transcript, "audio_forensic_data": forensic_data } except Exception as e: print(f" -> āš ļø Audio Analysis Error: {e}") return { "audio_transcript": "Audio analysis failed.", "audio_forensic_data": {"verdict": "N/A", "confidence": 0.0} } def node_process_media(state: AgentState) -> AgentState: """Node 2: Runs DeepFace and the local CNN Ensemble.""" print("\n[LANGGRAPH] 🟢 Entering Node: Pixel Forensics & Identity Extraction") media_path = state["media_path"] # 1. Run the Local Ensemble CNN try: from core.forensics import EnsembleForensicsEngine engine = EnsembleForensicsEngine() forensic_results = engine.analyze(media_path) except Exception as e: print(f" āš ļø Forensics Engine not available: {e}") forensic_results = {"note": "Forensics engine unavailable in this environment."} # 2. Extract Identity (Simulated here: wire your DeepFace function in) # Ideally, call your `test_identity.py` logic here. For safety if no match: identity_results = { "name": "Unknown", "description": "No known public figure detected or matched in database." } # Example of how you would load your metadata if DeepFace found a match: # metadata = json.load(open('known_faces/metadata.json')) # if match_key in metadata: identity_results = metadata[match_key] return {"forensic_data": forensic_results, "identity_data": identity_results} def node_osint_investigation(state: AgentState) -> AgentState: """Node 3: Scours the web for threat intelligence using Tavily with explicit sources.""" print("\n[LANGGRAPH] 🟢 Entering Node: OSINT Threat Intel Gathering") try: queries = _build_osint_queries(state) sections = [] all_sources = [] for idx, query in enumerate(queries, start=1): print(f" -> šŸ”Ž OSINT Query {idx}/{len(queries)}: '{query}'") tavily_text = _invoke_search_tool(tavily_search_tool, query) section = ( f"[Query {idx}] {query}\n" f"Tavily Results:\n{tavily_text}\n\n" ) sections.append(section) all_sources.extend(_extract_urls(tavily_text)) deduped_sources = [] seen = set() for src in all_sources: if src not in seen: seen.add(src) deduped_sources.append(src) source_block = "\n".join([f"- {u}" for u in deduped_sources[:10]]) if deduped_sources else "- No verifiable sources found." context = "\n\n".join(sections) context += "\n\nOSINT SOURCE REFERENCES:\n" + source_block return { "osint_context": context if context.strip() else "No significant recent threat activity found.", "osint_sources": deduped_sources[:10], } except Exception as e: print(f" -> āš ļø OSINT Error: {e}") return { "osint_context": "OSINT Search failed. Rely purely on pixel forensics.", "osint_sources": [], } def node_xai_report(state: AgentState) -> AgentState: """Node 4: The Mastermind LLM writes the final executive brief including Audio.""" print("\n[LANGGRAPH] 🟢 Entering Node: XAI Report Generation") # Initialize Groq (Requires GROQ_API_KEY in your .env) — import lazily and provide a mock fallback try: from langchain_groq import ChatGroq llm = ChatGroq( model="llama-3.3-70b-versatile", temperature=0.2, # Low temperature for an analytical, factual tone max_tokens=1024 ) except Exception: # Provide a simple mock that formats a basic report when no LLM is available. class _MockLLM: def invoke(self, prompt_text): # Create a short deterministic summary from inputs embedded in the prompt content = ( "# THREAT INTELLIGENCE BRIEFING: MEDIA AUTHENTICITY\n\n" "## 1. Executive Summary\n" "This is a text-only claim evaluation. The assistant evaluated the claim for sourcing and context.\n\n" "## 2. Media Context & Claim\n" "(Text-only) Claim summary and recommended verification steps.\n\n" "## 3. Visual Forensics Analysis\n" "Not applicable for text-only inputs.\n\n" "## 4. Voice & Audio Forensics\n" "Not applicable for text-only inputs.\n\n" "## 5. Threat Intelligence & Motive\n" "Basic OSINT summary included where available." ) return type("Resp", (), {"content": content})() llm = _MockLLM() # If this is a text-only analysis, instruct the LLM to avoid visual/audio authenticity labels forensic_note = "" try: forensic_note = json.dumps(state.get("forensic_data", {})) except Exception: forensic_note = str(state.get("forensic_data", "")) text_only = False if isinstance(state.get("forensic_data", {}), dict): note = state.get("forensic_data", {}).get("note", "") if "text-only" in note.lower() or "text only" in note.lower(): text_only = True # Build prompt with conditional instructions if text_only: preface = ( "THIS IS A TEXT-ONLY INPUT. Do NOT label the content as a 'deepfake' or attempt visual/audio authenticity judgements. " "Instead, evaluate the factual accuracy, reliability, and sourcing of the claim; provide a concise research summary, suggested next steps for verification, and cite any sources." ) else: preface = "" if text_only: # Dedicated prompt for text-only claim verification template_str = ( preface + "\n" + "You are an expert fact-checker and threat intelligence analyst. This input contains ONLY text (a claim or statement). Do NOT discuss media authenticity, visual forensics, audio, or deepfakes.\n\n" "### THE CLAIM ###\n" "- Claim: {media_context}\n\n" "### TASK ###\n" "1) Provide a short (1-2 sentence) Executive Summary stating the claim credibility (e.g., Supported / Unsupported / Ambiguous) and the main reason.\n" "2) Provide a concise Research Summary (2-4 bullets) listing any corroborating or contradicting evidence, and cite sources if available.\n" "3) Suggest pragmatic verification steps (3 items) a human investigator should take to validate the claim.\n" "4) Note potential motives or threat implications if the claim were spread maliciously.\n\n" "Return the result as a Markdown document with headings: Executive Summary, Research Summary, Verification Steps, Threat Implications." ) else: template_str = preface + "\n" + ( "You are an elite Threat Intelligence Analyst. Your job is to write a formal, non-technical Intelligence Briefing regarding a suspicious piece of media.\n\n" "DO NOT use any machine learning jargon. Do not mention \"CNNs\", \"pixels\", \"models\", or specific percentage scores. Translate the technical data into plain-English forensic findings.\n\n" "### THE RAW DATA ###\n" "- Visual Claim in Media: {media_context}\n" "- Spoken Transcript: \"{audio_transcript}\"\n" "- Subject Identified: {identity_data}\n" "- Visual Lab Results: {forensic_data}\n" "- Audio Lab Results: {audio_forensic_data}\n" "- Live News / Threat Intel: {osint_context}\n\n" "- Source References (URLs): {osint_sources}\n\n" "### REPORT FORMAT ###\n" "Generate a professional report using the following Markdown structure exactly:\n\n" "# THREAT INTELLIGENCE BRIEFING: MEDIA AUTHENTICITY\n\n" "## 1. Executive Summary\n" "State clearly whether this media is authentic or artificially generated (a deepfake). Summarize the main visual and audio reasons for this verdict in one to two sentences.\n\n" "## 2. Media Context & Claim\n" "Describe what is happening in the media. Explain who the identified subject is, what they are doing visually, and what they are explicitly saying based on the transcript.\n\n" "## 3. Visual Forensics Analysis\n" "Explain the visual tampering in plain English based on the visual lab results.\n\n" "## 4. Voice & Audio Forensics\n" "Analyze the audio lab results. State clearly if the voice was determined to be an AI-generated voice clone. Mention the transcript if the spoken words are suspicious or manipulative (like asking for money).\n\n" "## 5. Threat Intelligence & Motive\n" "Connect the media's claim and transcript to the live news data. If the media shows the subject promoting a scam, and the news shows active scams targeting this subject, explicitly state that this media is part of a known, active misinformation campaign.\n" "\n## 6. References\n" "List 3-8 concrete URLs used in your analysis. Use only URLs present in Source References (URLs). If there are none, write: 'No verifiable sources found.'\n" ) # Keep the prompt safely under token limits. safe_media_context = _truncate_text(str(state.get("media_context", "Unknown visual context.")), 700) safe_audio_transcript = _truncate_text(str(state.get("audio_transcript", "No audio detected.")), 700) safe_forensic_data = _truncate_text(json.dumps(state.get("forensic_data", {})), 1200) safe_audio_forensic_data = _truncate_text(json.dumps(state.get("audio_forensic_data", {})), 800) safe_identity_data = _truncate_text(json.dumps(state.get("identity_data", {})), 500) safe_osint_context = _compact_lines(state.get("osint_context", "No OSINT data available."), max_lines=14, max_chars=1800) safe_osint_sources = _truncate_text(json.dumps((state.get("osint_sources", []) or [])[:8]), 600) # Format the prompt with our entire state dictionary formatted_prompt = template_str.format( media_context=safe_media_context, audio_transcript=safe_audio_transcript, forensic_data=safe_forensic_data, audio_forensic_data=safe_audio_forensic_data, identity_data=safe_identity_data, osint_context=safe_osint_context, osint_sources=safe_osint_sources ) # Generate the report print(" -> 🧠 Groq is drafting the XAI Brief with Audio Insights...") try: response = llm.invoke(formatted_prompt) report_content = response.content except Exception as e: print(f" -> āš ļø Groq API Error: {e}") report_content = _build_fallback_report(state, text_only) # Ensure references are always present for auditability refs = state.get("osint_sources", []) or [] refs_md = "\n".join([f"- {u}" for u in refs[:8]]) if refs else "- No verifiable sources found." if text_only and "## References" not in report_content: report_content = report_content.rstrip() + "\n\n## References\n" + refs_md + "\n" if (not text_only) and "## 6. References" not in report_content: report_content = report_content.rstrip() + "\n\n## 6. References\n" + refs_md + "\n" return {"final_report": report_content} # ========================================== # 3. BUILD THE GRAPH PIPELINE # ========================================== def build_aispy_workflow(): from langgraph.graph import StateGraph, END workflow = StateGraph(AgentState) # Add all agents workflow.add_node("vision_context", node_vision_context) workflow.add_node("audio_analysis", node_audio_analysis) workflow.add_node("process_media", node_process_media) workflow.add_node("osint_investigation", node_osint_investigation) workflow.add_node("xai_report", node_xai_report) # Define the strict sequence (Vision -> Audio -> Media -> OSINT -> XAI) workflow.set_entry_point("vision_context") workflow.add_edge("vision_context", "audio_analysis") workflow.add_edge("audio_analysis", "process_media") workflow.add_edge("process_media", "osint_investigation") workflow.add_edge("osint_investigation", "xai_report") workflow.add_edge("xai_report", END) return workflow.compile() # ========================================== # 5. COMPATIBILITY WRAPPER FOR FLASK APP # ========================================== # Pydantic models for structured output (required by app.py) from pydantic import BaseModel from typing import Optional class ForensicsReport(BaseModel): is_manipulated: bool fake_probability: float visual_evidence: str extracted_caption: str class OSINTReport(BaseModel): claim_verified: bool debunked: bool sources_used: list = [] key_findings: str = "" class FinalVerdict(BaseModel): verdict: str confidence: float reasoning: str xai_breakdown: str def run_aispy_pipeline( input_type: str, media_path: str = None, text_claim: str = None, identity_data: dict = None, request_id: str = None ): """ Entry point for Flask app.py - wraps the LangGraph workflow and returns data in the format expected by the frontend. Returns a dict with: - errors: List of error strings - final_result: FinalVerdict Pydantic object - forensics_data: ForensicsReport Pydantic object - osint_data: OSINTReport Pydantic object """ print("==================================================") print("šŸš€ STARTING AI-SPY MASTERMIND WORKFLOW") print("==================================================") errors = [] # Initialize state with defaults initial_state = { "media_path": media_path or "", "media_context": text_claim or "No contextual claim provided.", "audio_transcript": "No audio detected.", "audio_forensic_data": {"verdict": "N/A", "confidence": 0.0}, "forensic_data": {}, "identity_data": identity_data or {"name": "Unknown", "description": ""}, "osint_context": "", "osint_sources": [], "final_report": "" } try: # Orchestrate deterministically by input type so we only run relevant analysis. media_context = initial_state.get("media_context", "") forensic_data = {} osint_context = "" osint_sources = [] final_report = "" if input_type == "text": # Research and summarize the provided text claim. Do not perform visual/audio forensics. media_context = text_claim or "Text-only claim provided." forensic_data = {"note": "Text-only analysis — visual/audio forensics not applicable."} osint_out = node_osint_investigation({ "identity_data": initial_state.get("identity_data", {"name": "Unknown"}), "media_context": media_context, "audio_transcript": "", }) osint_context = osint_out.get("osint_context", "") osint_sources = osint_out.get("osint_sources", []) xai_out = node_xai_report({ "media_context": media_context, "audio_transcript": "", "forensic_data": forensic_data, "audio_forensic_data": {"verdict": "N/A", "confidence": 0.0}, "identity_data": initial_state.get("identity_data", {"name": "Unknown"}), "osint_context": osint_context, "osint_sources": osint_sources, }) final_report = xai_out.get("final_report", "") elif input_type == "image": # Image-only: caption, run local forensics, OSINT, then XAI report v_out = node_vision_context({"media_path": media_path}) media_context = v_out.get("media_context", media_context) p_out = node_process_media({"media_path": media_path}) forensic_data = p_out.get("forensic_data", {}) if isinstance(p_out, dict) else {} identity_data = p_out.get("identity_data", initial_state.get("identity_data", {"name": "Unknown"})) try: osint_out = node_osint_investigation({ "identity_data": identity_data, "media_context": media_context, "audio_transcript": "", }) osint_context = osint_out.get("osint_context", "") osint_sources = osint_out.get("osint_sources", []) except Exception: osint_context = "" osint_sources = [] xai_out = node_xai_report({ "media_context": media_context, "audio_transcript": "", "forensic_data": forensic_data, "audio_forensic_data": {"verdict": "N/A", "confidence": 0.0}, "identity_data": identity_data, "osint_context": osint_context, "osint_sources": osint_sources, }) final_report = xai_out.get("final_report", "") elif input_type == "audio": # Audio-only: ASR + audio forensics, then XAI focusing on audio findings a_out = node_audio_analysis({"media_path": media_path}) audio_transcript = a_out.get("audio_transcript", "") audio_forensic = a_out.get("audio_forensic_data", {"verdict": "N/A", "confidence": 0.0}) osint_out = node_osint_investigation({ "identity_data": initial_state.get("identity_data", {"name": "Unknown"}), "media_context": text_claim or "Audio-only analysis.", "audio_transcript": audio_transcript, }) osint_context = osint_out.get("osint_context", "") osint_sources = osint_out.get("osint_sources", []) xai_out = node_xai_report({ "media_context": text_claim or "Audio-only analysis.", "audio_transcript": audio_transcript, "forensic_data": {"note": "Audio-only analysis — visual forensics not applied."}, "audio_forensic_data": audio_forensic, "identity_data": initial_state.get("identity_data", {"name": "Unknown"}), "osint_context": osint_context, "osint_sources": osint_sources, }) final_report = xai_out.get("final_report", "") else: # Full media workflow (video or mixed inputs): Vision -> Audio -> Pixel Forensics -> OSINT -> XAI v_out = node_vision_context({"media_path": media_path}) media_context = v_out.get("media_context", media_context) a_out = node_audio_analysis({"media_path": media_path}) audio_transcript = a_out.get("audio_transcript", "") audio_forensic = a_out.get("audio_forensic_data", {"verdict": "N/A", "confidence": 0.0}) p_out = node_process_media({"media_path": media_path}) forensic_data = p_out.get("forensic_data", {}) if isinstance(p_out, dict) else {} identity_data = p_out.get("identity_data", initial_state.get("identity_data", {"name": "Unknown"})) try: osint_out = node_osint_investigation({ "identity_data": identity_data, "media_context": media_context, "audio_transcript": audio_transcript, }) osint_context = osint_out.get("osint_context", "") osint_sources = osint_out.get("osint_sources", []) except Exception: osint_context = "" osint_sources = [] xai_out = node_xai_report({ "media_context": media_context, "audio_transcript": audio_transcript, "forensic_data": forensic_data, "audio_forensic_data": audio_forensic, "identity_data": identity_data, "osint_context": osint_context, "osint_sources": osint_sources, }) final_report = xai_out.get("final_report", "") # Convert forensic data to ForensicsReport if isinstance(forensic_data, dict): forensics_report = ForensicsReport( is_manipulated=forensic_data.get('is_fake', False), fake_probability=float(forensic_data.get('confidence', 0)) / 100.0, visual_evidence=forensic_data.get('reason', 'No evidence available'), extracted_caption=str(forensic_data.get('metrics', {})) ) else: forensics_report = ForensicsReport( is_manipulated=False, fake_probability=0.0, visual_evidence="Could not analyze media", extracted_caption="N/A" ) # Parse OSINT context (simple extraction from text) osint_report = OSINTReport( claim_verified=("verified" in osint_context.lower()) or (len(osint_sources) > 0), debunked=("no significant recent threat activity" in osint_context.lower()) or ("debunk" in osint_context.lower()), sources_used=osint_sources, key_findings=osint_context[:200] if osint_context else "No threat intelligence" ) # Generate final verdict from XAI report verdict, confidence, reasoning, xai_breakdown = _derive_final_verdict( input_type, final_report, forensic_data, audio_forensic if input_type == "audio" or input_type not in {"text", "image"} else {}, ) final_verdict = FinalVerdict( verdict=verdict, confidence=confidence, reasoning=reasoning, xai_breakdown=xai_breakdown ) # Convert pydantic models to plain dicts for safe serialization out_final = final_verdict.dict() if hasattr(final_verdict, "dict") else final_verdict out_forensics = forensics_report.dict() if hasattr(forensics_report, "dict") else forensics_report out_osint = osint_report.dict() if hasattr(osint_report, "dict") else osint_report return { "errors": errors, "final_result": out_final, "forensics_data": out_forensics, "osint_data": out_osint, "final_report": final_report, "media_context": media_context } except Exception as e: errors.append(f"Pipeline error: {str(e)}") print(f"āŒ Pipeline Error: {e}") # Return safe defaults on error return { "errors": errors, "final_result": None, "forensics_data": ForensicsReport( is_manipulated=False, fake_probability=0.0, visual_evidence="Error during analysis", extracted_caption="N/A" ), "osint_data": OSINTReport(claim_verified=False, debunked=False, sources_used=[]) } # ========================================== # 4. RUNNER FOR TESTING # ========================================== if __name__ == "__main__": from dotenv import load_dotenv load_dotenv() # Make sure TAVILY_API_KEY and GROQ_API_KEY are loaded print("==================================================") print("šŸš€ BOOTING AI-SPY MASTERMIND WORKFLOW") print("==================================================") # Compile the graph app = build_aispy_workflow() # Get input from the user test_target = input("\nšŸ“ Enter the path to the media: ").strip().strip("\"'") # Run the state machine inputs = {"media_path": test_target} final_state = app.invoke(inputs) print("\n" + "="*80) print("šŸ“„ AI-SPY EXECUTIVE INTELLIGENCE BRIEF") print("="*80) print(final_state["final_report"])