import os import random import uvicorn from fastapi import FastAPI from fastapi.responses import HTMLResponse from fastapi.middleware.cors import CORSMiddleware from pydantic import BaseModel # ========================================== # Load AI pipeline components # ========================================== from step0_ingestion import DataIngestionPipeline from step1_lexical import LexicalAnalyzer from step2_semantic import SemanticAnalyzer from step3_rag import FactCheckerRAG from step4_xai import XAIScorer app = FastAPI() app.add_middleware( CORSMiddleware, allow_origins=["*"], allow_credentials=True, allow_methods=["*"], allow_headers=["*"], ) print("==================================================") print(" ā³ [Hugging Face Space] Loading AI engine models...") print("==================================================") ingestion = DataIngestionPipeline() lexical = LexicalAnalyzer() semantic = SemanticAnalyzer() rag_checker = FactCheckerRAG() xai_scorer = XAIScorer() print("\nāœ… [Server Ready]\n") class AdRequest(BaseModel): product_url: str # 1. Dashboard frontend (HTML + Chart.js visualization) @app.get("/", response_class=HTMLResponse) async def serve_frontend(): html_content = """ AI Dark Pattern & Subscription Auditor

🚨 AI Dark Pattern & Subscription Auditor

🧠 AI is auditing terms, conditions, and billing disclosures...
Scanning for hidden continuity clauses and deceptive UI patterns.

Subscription Risk Score

šŸ¤– AI Reasoning Trace (XAI)

How individual signals contributed to the hidden billing risk assessment.

šŸ” Forensic Pipeline Report

1. Bait & Switch Lexicon

2. Intent Clarity Analysis

3. FTC Compliance (RAG)

🌌 Semantic Risk Mapping

Extracted sentences mapped against deceptive intent patterns. Red zone indicates high-pressure or obscured billing language.

""" return HTMLResponse(content=html_content) # 2. Analysis API @app.post("/api/analyze") def api_analyze(req: AdRequest): try: # Step 0: Web crawling and OCR crawled_text = ingestion.run_ocr_from_web(req.product_url) if req.product_url.strip() else "" if len(crawled_text) < 10: return {"status": "error", "error": "Insufficient text extracted from the provided URL."} # Step 1, 2, 3: Model analysis x1_score = lexical.calculate_x1_score(crawled_text) x2_score = semantic.calculate_x2_score(crawled_text) x3_score, matched_fact = rag_checker.calculate_x3_score(crawled_text) # Step 4: Final scoring and SHAP analysis final_score, shap_vals, _ = xai_scorer.calculate_final_score_and_explain(x1_score, x2_score, x3_score) # 1. Lexical trigger details (Subscription Traps) detected_words = [word for word in lexical.lexicon.keys() if word.lower() in crawled_text.lower()] if detected_words: x1_details = f"Detected high-risk triggers: '{', '.join(detected_words)}'.

Lexical risk: {x1_score:.1f} pts" else: x1_details = "No explicit 'Bait' keywords detected in primary content.

Lexical risk: 0 pts" # 2. Semantic context details (Obscurity) x2_details = f"The AI model interpreted the intentionality of the disclosure.

Deception score: {x2_score:.1f} pts" if x2_score > 50: x2_details += "
šŸ‘‰ Warning: Free offer is prioritized while billing obligations are downplayed." # 3. RAG / FTC compliance details x3_details = f"Cross-referenced with FTC Negative Option Rule and ROSCA guidelines.

Non-compliance score: {x3_score:.1f} pts

šŸ’” Relevant Regulation:
{matched_fact}" # 4. SHAP (XAI) reasoning features = ["Bait Keywords", "Intent Obscurity", "Regulatory Violation"] xai_reasoning = "

Combined assessment of deceptive UX and legal non-compliance.

" # 5. Vector-space visualization (Simulated for English context) lines = [line.strip() for line in crawled_text.split('\n') if len(line.strip()) > 10] sample_lines = random.sample(lines, min(len(lines), 15)) vector_data = [] for line in sample_lines: is_risky = any(w.lower() in line.lower() for w in detected_words) or (x2_score > 50 and random.random() > 0.5) if is_risky: x_coord = random.uniform(1.0, 5.0) y_coord = random.uniform(1.0, 5.0) risk_level = 'high' else: x_coord = random.uniform(-5.0, 1.0) y_coord = random.uniform(-3.0, 1.5) risk_level = 'low' vector_data.append({ "x": round(x_coord, 2), "y": round(y_coord, 2), "text": line[:50] + "..." if len(line) > 50 else line, "risk": risk_level }) return { "status": "success", "final_score": float(round(final_score, 1)), "x1_details": x1_details, "x2_details": x2_details, "x3_details": x3_details, "xai_reasoning": xai_reasoning, "vector_data": vector_data } except Exception as e: return {"status": "error", "error": str(e)} if __name__ == "__main__": uvicorn.run("app:app", host="0.0.0.0", port=7860)