Spaces:
Running
Running
Download app.py from FatimaDataScientist72/maci-api: direct link, hf CLI and curl.
- Browser
- Download file 67.5 kB
-
https://huggingface.co/spaces/FatimaDataScientist72/maci-api/resolve/main/app.py
- Command line
-
hf download hf://spaces/FatimaDataScientist72/maci-api/app.py
-
curl -L -o app.py https://huggingface.co/spaces/FatimaDataScientist72/maci-api/resolve/main/app.py
67.5 kB
| # ============================================================ | |
| # MACI API v5.5 — Institutional grade | |
| # Changes from v5.4.1: | |
| # - NEW: Phase 4 structural pattern layer (structural_patterns.py) | |
| # Rule-based, deterministic detection for known structural | |
| # red-flag patterns that the ML language-signal layer does | |
| # not catch by design (Bay' al-Inah buy-back, disguised Riba | |
| # via late-payment-as-income, unilateral price variation, | |
| # risk/ownership decoupling, total liability waiver). | |
| # Built and validated directly against SRB (Shariyah Review | |
| # Bureau) review findings, 2026-07-31. | |
| # - Structural flags surface as a separate "structural_pattern_flags" | |
| # field in every packet — clearly labeled as distinct from the | |
| # ML "maci_evaluation" language-signal result. A CRITICAL | |
| # structural flag escalates boundary_behavior even if the ML | |
| # layer alone would have allowed the text through. | |
| # Changes from v5.4: | |
| # - FIX: removed response_model=ClassifyResponse from /api/v1/classify | |
| # (that stub schema was silently stripping maqasid_enrichment, | |
| # jurisdiction_overlay, overclaiming_controls, handoff_integrity, | |
| # runtime_handoff, and _compact out of every real response — | |
| # this is what caused the frontend to fall back to a hardcoded | |
| # "FAS 1" placeholder for aaoifi_standard on every violation) | |
| # Changes from v5.3: | |
| # - max_length 2000 → 8000 (fixes Faisal 422 error) | |
| # - Smart truncation to 1500 chars before tokenization | |
| # - /api/v1/classify/audit — multi-mechanism batch audit | |
| # - /api/v1/classify/simple — plain form text endpoint | |
| # - Phase 2+3 enrichment fields in every packet | |
| # - All v5.3 functionality preserved | |
| # ============================================================ | |
| import os, json, pickle, re, uuid, hashlib, shutil, secrets | |
| import numpy as np | |
| from datetime import datetime, timezone | |
| from pathlib import Path | |
| from scipy.sparse import hstack | |
| from fastapi import FastAPI, HTTPException, Depends, Form | |
| from fastapi.middleware.cors import CORSMiddleware | |
| from fastapi.security import HTTPBearer, HTTPAuthorizationCredentials | |
| from pydantic import BaseModel, Field | |
| from typing import Optional, List | |
| import torch, faiss | |
| from transformers import (AutoTokenizer, AutoModel, | |
| AutoModelForSequenceClassification) | |
| from huggingface_hub import hf_hub_download | |
| # NEW: Phase 4 structural pattern layer — upload structural_patterns.py | |
| # alongside this file in the same Space for this import to work. | |
| from structural_patterns import scan_structural_patterns, structural_flags_to_dict | |
| # NEW: Phase 5 reconciliation layer — upload reconciliation.py alongside | |
| # this file too. Only scan_hard_negatives is used here; the structural | |
| # override itself is handled inline below because it needs to update | |
| # `pred` before Phase 2/3 enrichment run later in this same function. | |
| from reconciliation import scan_hard_negatives | |
| # ── App ──────────────────────────────────────────────────────── | |
| app = FastAPI( | |
| title = "MACI — Muslim AI Content Intelligence", | |
| description = "Shariah compliance classifier v5.5 — institutional grade. " | |
| "Full Terry v0.1 Signal Packet + Phase 2 Maqasid enrichment " | |
| "+ Phase 3 jurisdiction overlay + Phase 4 structural patterns " | |
| "+ Phase 5 reconciliation.", | |
| version = "5.5", | |
| ) | |
| app.add_middleware(CORSMiddleware, allow_origins=["*"], | |
| allow_methods=["*"], allow_headers=["*"]) | |
| # ── Config ───────────────────────────────────────────────────── | |
| HF_REPO = os.getenv("HF_REPO", "MerridaDataScientist72/maci-shield") | |
| HF_TOKEN = os.getenv("HF_TOKEN", None) | |
| MODEL_DIR = Path("maci_model") | |
| DEVICE = "cuda" if torch.cuda.is_available() else "cpu" | |
| # ── API Key ──────────────────────────────────────────────────── | |
| API_KEY = os.environ.get("API_KEY", None) | |
| if not API_KEY: | |
| API_KEY = secrets.token_urlsafe(32) | |
| print(f"⚠️ No API_KEY set — using session key: {API_KEY}") | |
| security = HTTPBearer() | |
| def verify_api_key( | |
| credentials: HTTPAuthorizationCredentials = Depends(security), | |
| ): | |
| if credentials.credentials != API_KEY: | |
| raise HTTPException( | |
| status_code=401, | |
| detail="Invalid or missing API key", | |
| headers={"WWW-Authenticate": "Bearer"}, | |
| ) | |
| return True | |
| # ── Label / routing tables ───────────────────────────────────── | |
| LABEL_NAMES = { | |
| 0:"Authentic", 1:"Riba (Usury/Interest)", | |
| 2:"Gharar (Excessive Uncertainty)", 3:"Maysir (Gambling/Speculation)", | |
| 4:"Fabricated/Unauthorized Fatwa", 5:"Quran/Hadith Fabrication", | |
| 6:"MLM / Pyramid Scheme", 7:"Scholar Misquotation", | |
| } | |
| AUTH = { | |
| 0:"none", 1:"shariah_scholar", | |
| 2:"shariah_board", 3:"shariah_scholar", | |
| 4:"shariah_board", 5:"shariah_board", | |
| 6:"compliance_officer", 7:"shariah_scholar", | |
| } | |
| RISK = { | |
| 0:"None — content is Shariah-compliant", | |
| 1:"Customer may rely on interest-bearing guidance as compliant", | |
| 2:"Customer may enter contract with impermissible uncertainty", | |
| 3:"Customer may engage in prohibited speculative activity", | |
| 4:"False religious authority may be established", | |
| 5:"Corrupted sacred text may be presented as authentic", | |
| 6:"Customer may join prohibited pyramid scheme", | |
| 7:"False scholarly position may mislead customer", | |
| } | |
| REFU = { | |
| 0:None, | |
| 1:"output contains interest-bearing structure", | |
| 2:"output describes ambiguous or void contract", | |
| 3:"output facilitates gambling or speculation", | |
| 4:"output attributes ruling without verified source", | |
| 5:"output presents fabricated text as authentic", | |
| 6:"output promotes recruitment-based income", | |
| 7:"output misattributes position to named scholar", | |
| } | |
| MOVE = { | |
| 0:"deliver_output_to_user", 1:"escalate_to_human", | |
| 2:"escalate_to_human", 3:"escalate_to_human", | |
| 4:"quarantine_output", 5:"quarantine_output", | |
| 6:"quarantine_output", 7:"quarantine_output", | |
| } | |
| SEV = { | |
| (0,"HIGH"):"NONE", (0,"MEDIUM"):"NONE", (0,"LOW"):"LOW", | |
| (1,"HIGH"):"CRITICAL",(1,"MEDIUM"):"HIGH", (1,"LOW"):"MEDIUM", | |
| (2,"HIGH"):"HIGH", (2,"MEDIUM"):"MEDIUM", (2,"LOW"):"LOW", | |
| (3,"HIGH"):"HIGH", (3,"MEDIUM"):"MEDIUM", (3,"LOW"):"LOW", | |
| (4,"HIGH"):"CRITICAL",(4,"MEDIUM"):"HIGH", (4,"LOW"):"MEDIUM", | |
| (5,"HIGH"):"CRITICAL",(5,"MEDIUM"):"CRITICAL", (5,"LOW"):"HIGH", | |
| (6,"HIGH"):"HIGH", (6,"MEDIUM"):"MEDIUM", (6,"LOW"):"LOW", | |
| (7,"HIGH"):"HIGH", (7,"MEDIUM"):"MEDIUM", (7,"LOW"):"LOW", | |
| } | |
| # NEW (Phase 6): the generic Gharar citation in ALTERNATIVE_MAP[2] was | |
| # hardcoded to always cite AAOIFI SS26 (Takaful) / SS10 (Salam) — found, | |
| # via direct review of real Murabaha documents, to be actively confusing | |
| # on the single most common contract type MACI is used against: citing | |
| # Takaful (insurance) and Salam (forward-delivery sale) standards on a | |
| # Murabaha contract's Gharar flag doesn't map to anything actually in | |
| # the document. This does simple keyword-based contract-type detection | |
| # and picks the AAOIFI standard that's actually contextually relevant, | |
| # defaulting to Murabaha (SS8) — the most common case, and the standard | |
| # the structural layer already cites for the same document type — rather | |
| # than defaulting to Takaful, which was a poor default to begin with. | |
| GHARAR_CITATION_BY_CONTRACT_TYPE = { | |
| "murabaha": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 8 (Murabaha), \u00a72/2 \u2014 price, subject matter, and delivery terms must be clearly specified", | |
| "structural_fix": "Specify subject matter, price, delivery date, and quality within the Murabaha agreement itself \u2014 ambiguity here is what AAOIFI SS8 \u00a72/2 requires be removed.", | |
| }, | |
| "tawarruq": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 30 (Monetisation / Tawarruq) \u2014 subject matter and pricing must be clearly specified and free of ambiguity", | |
| "structural_fix": "Specify the underlying commodity, price, and settlement terms within the Tawarruq arrangement itself.", | |
| }, | |
| "salam": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 10 (Salam and Parallel Salam)", | |
| "structural_fix": "Specify subject matter, quantity, quality, and delivery date at contract formation, as required for a valid Salam sale.", | |
| }, | |
| "takaful": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 26 (Takaful)", | |
| "structural_fix": "Specify the risk being covered, contribution basis, and surplus-distribution terms within the Takaful policy itself.", | |
| }, | |
| "ijara": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 9 (Ijarah and Ijarah Muntahia Bittamleek)", | |
| "structural_fix": "Specify the leased asset, rental terms, and maintenance responsibility within the Ijara agreement itself.", | |
| }, | |
| # NEW — sourced directly from AAOIFI's own official 2015 Shari'a | |
| # Standards compilation, §3 (real primary text, not general knowledge). | |
| # SS2 covers three card types with three distinct rules — a generic | |
| # "cards are fine" citation would be too coarse. Debit cards: permitted, | |
| # no conditions beyond no-interest and no-overdraft. Charge cards: | |
| # permitted only if (a) no interest on late payment, (b) any required | |
| # security deposit is invested via Mudarabah with disclosed profit-share, | |
| # (c) prohibited-use restriction. Credit cards: an interest-bearing | |
| # revolving credit facility is NOT permitted at all — this is the one | |
| # that actually matters for most modern credit card products. | |
| "cards": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 2 (Debit Card, Charge Card and Credit Card), \u00a73/3 \u2014 a credit card may not provide an interest-bearing revolving credit facility", | |
| "structural_fix": "Remove any interest-bearing revolving credit mechanism. A charge card structure (fixed period, no interest on timely repayment, Shari'a-compliant handling of any security deposit) is permitted where a revolving credit card is not.", | |
| }, | |
| # NEW — citation-layer expansion. IMPORTANT DISTINCTION FROM THE ABOVE: | |
| # this is reference/citation coverage only (knowing which standard | |
| # governs which contract type), not new structural violation | |
| # detection. No section-level pinpoint citation (e.g. "§2/2") is | |
| # given for these unless independently confirmed — a bare standard | |
| # name/number is a safer, more honest citation than inventing a | |
| # specific section reference without real validation. These five | |
| # standard-number mappings are stable, well-established AAOIFI | |
| # numbering; the exact clause each one governs in a *specific* | |
| # document still requires human review, same as everywhere else. | |
| "musharakah": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 12 (Sharikah / Musharakah and Modern Corporations)", | |
| "structural_fix": "Specify each partner's capital contribution, profit-sharing ratio, and loss-bearing terms — loss must track capital contribution, not be pre-fixed.", | |
| }, | |
| "mudarabah": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 13 (Mudarabah)", | |
| "structural_fix": "Specify the profit-sharing ratio as a proportion, not a fixed amount, and confirm the capital provider bears financial loss absent misconduct or negligence by the manager.", | |
| }, | |
| "istisna": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 11 (Istisna'a and Parallel Istisna'a)", | |
| "structural_fix": "Specify the manufactured asset's description, specifications, price, and delivery date at contract formation.", | |
| }, | |
| "wakala": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 23 (Agency / Wakala)", | |
| "structural_fix": "Specify the scope of the agent's authority and the fee basis — a fee contingent on investment performance can shift the arrangement away from a genuine agency structure.", | |
| }, | |
| "sukuk": { | |
| "aaoifi_standard": "AAOIFI Shari'a Standard No. 17 (Investment Sukuk)", | |
| "structural_fix": "Confirm sukuk holders have genuine ownership of the underlying asset or venture, with returns tied to real asset performance, not a fixed, guaranteed payment.", | |
| }, | |
| } | |
| def detect_contract_type(text: str) -> str: | |
| """ | |
| Simple keyword-based contract-type detection, English + Arabic, | |
| used only to pick a contextually-relevant Gharar citation. Not a | |
| claim of deep semantic contract understanding — a document that | |
| never names its own contract type falls back to Murabaha, the most | |
| common case in current usage and the type the structural layer is | |
| most built around, rather than an arbitrary unrelated default. | |
| """ | |
| t = text.lower() | |
| if "tawarruq" in t or "توارق" in text: | |
| return "tawarruq" | |
| if "salam" in t or "سلم" in text: | |
| return "salam" | |
| if "takaful" in t or "تكافل" in text: | |
| return "takaful" | |
| if "ijara" in t or "ijarah" in t or "إجارة" in text: | |
| return "ijara" | |
| if "musharakah" in t or "sharikah" in t or "مشاركة" in text: | |
| return "musharakah" | |
| if "mudarabah" in t or "مضاربة" in text: | |
| return "mudarabah" | |
| if "istisna" in t or "استصناع" in text: | |
| return "istisna" | |
| if "wakala" in t or "wakalah" in t or "وكالة" in text: | |
| return "wakala" | |
| if "sukuk" in t or "صكوك" in text: | |
| return "sukuk" | |
| if "credit card" in t or "charge card" in t or "debit card" in t or "بطاقة" in text: | |
| return "cards" | |
| if "murabaha" in t or "مرابحة" in text: | |
| return "murabaha" | |
| return "murabaha" | |
| # NEW (Phase 5): maps each structural pattern to the closest existing | |
| # `pred` class id. This is what lets a structural override also correct | |
| # maqasid_enrichment (Phase 2) and jurisdiction_overlay (Phase 3) below — | |
| # both of those blocks key off `pred`, and both run AFTER this mapping is | |
| # applied, so getting this mapping right means the whole rest of the | |
| # packet (pillar, AAOIFI standard, jurisdiction exposure/penalty) becomes | |
| # consistent with the corrected verdict automatically, not just the | |
| # top-line violation_class field. | |
| STRUCTURAL_PATTERN_TO_PRED = { | |
| "SP-01": 1, # Bay' al-Inah buy-back -> Riba (disguised loan) | |
| "SP-02": 1, # Disguised Riba late-payment-income -> Riba | |
| "SP-03": 2, # Unilateral price variation -> Gharar | |
| "SP-04": 2, # Risk/ownership decoupling -> Gharar | |
| "SP-05": 2, # Total liability waiver -> Gharar (closest of the 8 classes) | |
| } | |
| STRUCTURAL_SEVERITY_RANK = {"LOW":1, "MEDIUM":2, "HIGH":3, "CRITICAL":4} | |
| # ── Phase 2: Maqasid + consequence ──────────────────────────── | |
| CONSEQUENCE_TABLE = { | |
| 0:{"maqasid_pillar":"Hifz al-Mal","maqasid_meaning":"Protection of Wealth", | |
| "harm_scope":"none","severity_score_base":0.0, | |
| "economic_signal_high":"Compliant — no restructuring required", | |
| "economic_signal_low":"Classified compliant but confidence low — verify manually", | |
| "action_required":"none"}, | |
| 1:{"maqasid_pillar":"Hifz al-Mal","maqasid_meaning":"Protection of Wealth", | |
| "harm_scope":"individual_to_systemic","severity_score_base":0.95, | |
| "economic_signal_high":"Wealth protection undermined — interest-bearing structure " | |
| "extracts value without productive exchange; restructure required", | |
| "economic_signal_low":"Possible Riba signal — confidence insufficient; route to scholar", | |
| "action_required":"restructure_contract"}, | |
| 2:{"maqasid_pillar":"Hifz al-Mal","maqasid_meaning":"Protection of Wealth", | |
| "harm_scope":"individual_to_institutional","severity_score_base":0.70, | |
| "economic_signal_high":"Contract uncertainty exposes parties to unquantifiable loss — " | |
| "specify subject matter, price, and delivery before execution", | |
| "economic_signal_low":"Possible Gharar — low confidence; human review recommended", | |
| "action_required":"clarify_contract_terms"}, | |
| 3:{"maqasid_pillar":"Hifz al-Mal","maqasid_meaning":"Protection of Wealth", | |
| "harm_scope":"individual","severity_score_base":0.90, | |
| "economic_signal_high":"Speculative zero-sum transfer with no productive underlying — prohibited", | |
| "economic_signal_low":"Possible Maysir — do not act; route to scholar review", | |
| "action_required":"block_and_replace"}, | |
| 4:{"maqasid_pillar":"Hifz al-Din","maqasid_meaning":"Protection of Faith", | |
| "harm_scope":"institutional_to_systemic","severity_score_base":0.92, | |
| "economic_signal_high":"False religious authority creates market for non-compliant " | |
| "products under Islamic label — systemic trust damage", | |
| "economic_signal_low":"Possible fabricated fatwa — route to senior scholar verification", | |
| "action_required":"quarantine_and_verify_authority"}, | |
| 5:{"maqasid_pillar":"Hifz al-Din","maqasid_meaning":"Protection of Faith", | |
| "harm_scope":"systemic","severity_score_base":0.98, | |
| "economic_signal_high":"Fabricated sacred text corrupts doctrinal foundation — " | |
| "immediate quarantine required", | |
| "economic_signal_low":"Possible fabrication — quarantine pending specialist verification", | |
| "action_required":"immediate_quarantine"}, | |
| 6:{"maqasid_pillar":"Hifz al-Mal","maqasid_meaning":"Protection of Wealth", | |
| "harm_scope":"individual_to_institutional","severity_score_base":0.88, | |
| "economic_signal_high":"Recruitment-dependent income concentrates wealth upward — " | |
| "mathematical certainty of majority loss; pyramid structure", | |
| "economic_signal_low":"Possible MLM — low confidence; route to compliance officer", | |
| "action_required":"block_and_replace"}, | |
| 7:{"maqasid_pillar":"Hifz al-Din","maqasid_meaning":"Protection of Faith", | |
| "harm_scope":"institutional","severity_score_base":0.75, | |
| "economic_signal_high":"Misattributed scholarly position may legitimize prohibited " | |
| "products — verify with primary source", | |
| "economic_signal_low":"Possible misquotation — do not attribute; verify at primary source", | |
| "action_required":"verify_scholarly_source"}, | |
| } | |
| ALTERNATIVE_MAP = { | |
| 0:{"primary":None,"structural_fix":"none required","aaoifi_standard":None}, | |
| 1:{"primary":"Murabaha","secondary":"Diminishing Musharakah", | |
| "structural_fix":"Replace fixed interest with cost-plus sale (Murabaha) " | |
| "or profit-loss sharing. Bank must take real ownership risk.", | |
| # FIX — this cited "AAOIFI Shariah Standard No. 2 (Murabaha)," but SS2 | |
| # is actually "Debit Card, Charge Card and Credit Card," confirmed | |
| # directly against AAOIFI's own official standards text. Real Murabaha | |
| # is SS8. This has been wrong since before this project's other fixes — | |
| # caught only once a primary-source standard list was available to | |
| # check against, not found by any amount of code review alone. | |
| "aaoifi_standard":"AAOIFI Shari'a Standard No. 8 (Murabaha), No. 12 (Musharakah)"}, | |
| 2:{"primary":"Clearly Defined Contract","secondary":"Takaful", | |
| "structural_fix":"Specify subject matter, price, delivery date, and quality. " | |
| "For insurance: replace with Takaful cooperative model.", | |
| "aaoifi_standard":"AAOIFI Shariah Standard No. 26 (Takaful), No. 10 (Salam)"}, | |
| 3:{"primary":"Halal Screened Equity","secondary":"Wakalah Investment", | |
| "structural_fix":"Replace speculative bet with equity ownership in screened " | |
| "halal companies per AAOIFI/DJIM screening criteria.", | |
| "aaoifi_standard":"AAOIFI Shariah Standard No. 21 (Financial Papers)"}, | |
| 4:{"primary":"Verified Institutional Fatwa", | |
| "secondary":"AAOIFI-Certified Shariah Board Opinion", | |
| "structural_fix":"Obtain fatwa from named qualified scholar, citing Quran/Sunnah, " | |
| "with conditions, from verifiable institution.", | |
| "aaoifi_standard":"AAOIFI Governance Standard No. 1 (Shariah Supervisory Board)"}, | |
| 5:{"primary":"Authenticated Hadith from Canonical Collections", | |
| "secondary":"Verified Quranic Reference with Tafsir", | |
| "structural_fix":"Verify against Sahih Bukhari, Sahih Muslim, Sunan Abu Dawud. " | |
| "For Quran: verify against Mushaf Uthmani.", | |
| "aaoifi_standard":"Quran 15:9 — preservation principle"}, | |
| 6:{"primary":"Direct Halal Sales (No Downline)","secondary":"Wakalah Agency", | |
| "structural_fix":"Income must derive from product sales only — not recruitment. " | |
| "Agent earns fixed fee on own sales only.", | |
| "aaoifi_standard":"OIC Fiqh Academy Resolution on MLM schemes"}, | |
| 7:{"primary":"Primary Source Verification","secondary":"Direct Scholar Consultation", | |
| "structural_fix":"Verify at scholar's official website, fatwa database, or published works.", | |
| "aaoifi_standard":"Isnad verification principles"}, | |
| } | |
| # ── Phase 3: Jurisdiction overlay ───────────────────────────── | |
| JURISDICTION_EXPOSURE = { | |
| 1:{"MY":{"exposure":"critical","penalty":"IFSA s.136 — fine up to MYR 25M; license revocation"}, | |
| "AE":{"exposure":"critical","penalty":"CBUAE — fine up to AED 10M; Islamic license revocation"}, | |
| "PK":{"exposure":"high", "penalty":"SBP Shariah non-compliance notice; corrective order"}, | |
| "SA":{"exposure":"high", "penalty":"SAMA supervisory action; Shariah board rectification"}, | |
| "GB":{"exposure":"low", "penalty":"FCA misleading promotion if Islamic label used"}, | |
| "US":{"exposure":"low", "penalty":"FTC consumer protection if Islamic label used"}, | |
| "BH":{"exposure":"critical","penalty":"CBB Rulebook Vol. 2 — AAOIFI Shari'a Standards are " | |
| "mandatory for CBB-licensed Islamic banks; breach risks " | |
| "supervisory action and Shari'a Supervisory Board rectification order"}}, | |
| 2:{"MY":{"exposure":"high", "penalty":"Contract voidable under IFSA; compliance breach"}, | |
| "AE":{"exposure":"high", "penalty":"Shariah board may declare contract null"}, | |
| "PK":{"exposure":"medium", "penalty":"SBP Shariah review; corrective action"}, | |
| "SA":{"exposure":"high", "penalty":"SAMA Shariah compliance board review"}, | |
| "GB":{"exposure":"low", "penalty":"Standard contract law applies"}, | |
| "US":{"exposure":"low", "penalty":"UCC standard contract law"}, | |
| "BH":{"exposure":"high", "penalty":"CBB Shari'a Governance Module — contract may be " | |
| "deemed non-compliant; SSB correction required before " | |
| "product can remain in market"}}, | |
| 3:{"MY":{"exposure":"critical","penalty":"Gambling Act 1953 + IFSA — criminal prosecution"}, | |
| "AE":{"exposure":"critical","penalty":"Federal law — criminal prosecution"}, | |
| "PK":{"exposure":"critical","penalty":"Pakistan Penal Code s.294A — criminal"}, | |
| "SA":{"exposure":"critical","penalty":"Criminal prohibition under Saudi law"}, | |
| "GB":{"exposure":"medium", "penalty":"Gambling Commission license required"}, | |
| "US":{"exposure":"medium", "penalty":"State gambling laws + federal wire act"}, | |
| "BH":{"exposure":"critical","penalty":"Bahrain Penal Code gambling provisions + CBB " | |
| "prohibition on speculative/Maysir-based products " | |
| "for licensed Islamic institutions"}}, | |
| 4:{"MY":{"exposure":"critical","penalty":"IFSA s.28 — false Shariah claim; imprisonment up to 8 years"}, | |
| "AE":{"exposure":"critical","penalty":"Dubai Islamic Economy — product withdrawal + fine"}, | |
| "PK":{"exposure":"high", "penalty":"SBP can revoke Islamic banking window license"}, | |
| "SA":{"exposure":"high", "penalty":"Council of Senior Scholars censure"}, | |
| "GB":{"exposure":"medium", "penalty":"FCA misleading financial promotion — fine + withdrawal"}, | |
| "US":{"exposure":"medium", "penalty":"FTC deceptive practices; SEC if securities involved"}, | |
| "BH":{"exposure":"critical","penalty":"CBB Shari'a Governance Module requires named, " | |
| "CBB-approved Shari'a Supervisory Board sign-off; " | |
| "unauthorized ruling claims risk direct CBB " | |
| "supervisory action given Bahrain hosts AAOIFI HQ"}}, | |
| 5:{"MY":{"exposure":"critical","penalty":"Penal Code + IFSA false claim — criminal"}, | |
| "AE":{"exposure":"critical","penalty":"UAE cybercrime law + religious offence law"}, | |
| "PK":{"exposure":"critical","penalty":"Pakistan Penal Code s.295C — blasphemy; severe penalty"}, | |
| "SA":{"exposure":"critical","penalty":"Criminal under Saudi religious law"}, | |
| "GB":{"exposure":"low", "penalty":"No blasphemy law since 2008; possible hate speech"}, | |
| "US":{"exposure":"low", "penalty":"First Amendment protects; minimal exposure"}, | |
| "BH":{"exposure":"critical","penalty":"Bahrain Penal Code religious-offence provisions; " | |
| "acute reputational exposure given Bahrain's role as " | |
| "AAOIFI headquarters and regional Islamic finance hub"}}, | |
| 6:{"MY":{"exposure":"high","penalty":"Direct Sales and Anti-Pyramid Scheme Act 1993 — criminal"}, | |
| "AE":{"exposure":"high","penalty":"UAE Commercial Companies Law — prohibited"}, | |
| "PK":{"exposure":"high","penalty":"SECP — illegal securities scheme prosecution"}, | |
| "SA":{"exposure":"high","penalty":"SAMA + CMA — pyramid scheme prohibition"}, | |
| "GB":{"exposure":"high","penalty":"Trading Schemes Act 1996 — criminal"}, | |
| "US":{"exposure":"high","penalty":"FTC Act s.5 + SEC securities fraud if investment element"}, | |
| "BH":{"exposure":"high", "penalty":"Bahrain Commercial Companies Law + CBB — prohibited " | |
| "pyramid/recruitment-based structures"}}, | |
| 7:{"MY":{"exposure":"medium","penalty":"IFSA compliance review if product-linked"}, | |
| "AE":{"exposure":"medium","penalty":"DIFC Shariah board — correction required"}, | |
| "PK":{"exposure":"medium","penalty":"SBP review if financial product"}, | |
| "SA":{"exposure":"high", "penalty":"Council of Senior Scholars — formal rebuke + retraction"}, | |
| "GB":{"exposure":"low", "penalty":"Defamation law if scholar sues"}, | |
| "US":{"exposure":"low", "penalty":"Defamation law if scholar sues"}, | |
| "BH":{"exposure":"medium","penalty":"CBB Shari'a Governance Module — correction required; " | |
| "defamation exposure if named scholar affected"}}, | |
| } | |
| for j in ["MY","AE","PK","SA","GB","US","BH"]: | |
| JURISDICTION_EXPOSURE.setdefault(0,{})[j] = { | |
| "exposure":"none","penalty":"No regulatory exposure — content is compliant"} | |
| # HONESTY NOTE: Jurisdiction penalty citations (including the new BH/Bahrain | |
| # entries added 2026-08-07) are constructed from general regulatory knowledge, | |
| # not verified against current statute text or confirmed with local counsel. | |
| # Treat as indicative/directional, not as certified legal citations, until | |
| # reviewed by someone qualified in each jurisdiction's actual regulatory | |
| # framework — same caveat that already applies to the AAOIFI standard | |
| # citations elsewhere in this file. | |
| # ── Self-claim detection ─────────────────────────────────────── | |
| SELF_CLAIM_SIGNALS = [ | |
| "shariah compliant","shariah-compliant","halal certified", | |
| "shariah approved","board approved","board certified", | |
| "islamic approved","fully compliant","شرعی","حلال", | |
| "متوافق مع الشريعة", | |
| ] | |
| def has_self_claim(text: str) -> bool: | |
| t = text.lower() | |
| return any(s in t for s in SELF_CLAIM_SIGNALS) | |
| # ── Language detection ───────────────────────────────────────── | |
| URDU_RE = re.compile(r"[\u0679\u0688\u0691\u06BA\u06BE\u06C1\u06C3\u06D2\u06D3]") | |
| FARSI_RE = re.compile(r"[\u067E\u0686\u06AF\u06A9\u06CC\u06F0-\u06F9]") | |
| TAJIK_KW = ["қуръон","фоиз","ҳалол","ҳаром","шариат","тиҷорат","закот"] | |
| def detect_lang(text: str) -> str: | |
| has_ar = bool(re.search(r"[\u0600-\u06FF]", text)) | |
| has_la = bool(re.search(r"[a-zA-Z]", text)) | |
| has_cy = bool(re.search(r"[\u0400-\u04FF]", text)) | |
| has_ur = bool(URDU_RE.search(text)) | |
| has_fa = bool(FARSI_RE.search(text)) | |
| has_tjk = has_cy and any(s in text.lower() for s in TAJIK_KW) | |
| if has_tjk: return "TJK" | |
| if has_ur: return "UR" | |
| if has_fa: return "FA" | |
| if has_ar: return "AR" | |
| return "EN" | |
| # ── Overclaiming guards ──────────────────────────────────────── | |
| def modulate_severity(base_score: float, confidence: float): | |
| score = base_score * confidence | |
| if score >= 0.75: return "severe", round(score, 4) | |
| elif score >= 0.45: return "moderate", round(score, 4) | |
| elif score >= 0.20: return "light", round(score, 4) | |
| else: return "indicative", round(score, 4) | |
| DISCLAIMER = { | |
| "HIGH" : None, | |
| "MEDIUM": "Moderate confidence — review by qualified Shariah scholar before action.", | |
| "LOW" : "Low confidence — indicative only. Mandatory human review required.", | |
| } | |
| # ── Global state ─────────────────────────────────────────────── | |
| STATE = {} | |
| # ── Startup ──────────────────────────────────────────────────── | |
| async def load_models(): | |
| MODEL_DIR.mkdir(exist_ok=True) | |
| (MODEL_DIR / "xlmr").mkdir(exist_ok=True) | |
| print(f"Loading models | repo={HF_REPO} | device={DEVICE}") | |
| files_to_download = [ | |
| ("ml_model.pkl", MODEL_DIR / "ml_model.pkl"), | |
| ("v5/fiqh_rulings.json", MODEL_DIR / "fiqh_rulings.json"), | |
| ("v5/xlmr/config.json", MODEL_DIR / "xlmr" / "config.json"), | |
| ("v5/xlmr/model.safetensors", MODEL_DIR / "xlmr" / "model.safetensors"), | |
| ("v5/xlmr/tokenizer.json", MODEL_DIR / "xlmr" / "tokenizer.json"), | |
| ("v5/xlmr/tokenizer_config.json", MODEL_DIR / "xlmr" / "tokenizer_config.json"), | |
| ("v5/xlmr/class_names.json", MODEL_DIR / "xlmr" / "class_names.json"), | |
| ("v5/fiqh_index_xlmr.faiss", MODEL_DIR / "fiqh_index_xlmr.faiss"), | |
| ("v5/metadata_v53.json", MODEL_DIR / "metadata_v53.json"), | |
| ] | |
| for repo_path, dst in files_to_download: | |
| dst.parent.mkdir(parents=True, exist_ok=True) | |
| if not dst.exists(): | |
| print(f" Downloading {repo_path}...") | |
| try: | |
| downloaded = hf_hub_download( | |
| repo_id=HF_REPO, filename=repo_path, | |
| repo_type="model", token=HF_TOKEN, | |
| ) | |
| shutil.copy(downloaded, dst) | |
| size = dst.stat().st_size / (1024*1024) | |
| print(f" ✅ {dst.name} ({size:.1f} MB)") | |
| except Exception as e: | |
| print(f" ❌ FAILED {repo_path}: {e}") | |
| raise | |
| else: | |
| print(f" ✓ cached {dst.name}") | |
| with open(MODEL_DIR/"ml_model.pkl","rb") as f: | |
| STATE["ml"] = pickle.load(f) | |
| print(f" ✅ ML model F1={STATE['ml']['macro_f1']:.4f}") | |
| with open(MODEL_DIR/"fiqh_rulings.json",encoding="utf-8") as f: | |
| STATE["rulings"] = json.load(f) | |
| print(f" ✅ Rulings {len(STATE['rulings'])}") | |
| xlmr_dir = str(MODEL_DIR/"xlmr") | |
| STATE["tokenizer"] = AutoTokenizer.from_pretrained(xlmr_dir) | |
| STATE["xlmr"] = AutoModelForSequenceClassification.from_pretrained( | |
| xlmr_dir, num_labels=8, ignore_mismatched_sizes=True).to(DEVICE) | |
| STATE["xlmr"].eval() | |
| print(f" ✅ XLM-R on {DEVICE}") | |
| if hasattr(STATE["xlmr"], "roberta"): | |
| STATE["base_enc"] = STATE["xlmr"].roberta | |
| elif hasattr(STATE["xlmr"], "xlm_roberta"): | |
| STATE["base_enc"] = STATE["xlmr"].xlm_roberta | |
| else: | |
| STATE["base_enc"] = AutoModel.from_pretrained( | |
| "xlm-roberta-base").to(DEVICE) | |
| print(" ⚠️ Using AutoModel fallback for FAISS encoding") | |
| STATE["base_enc"].eval() | |
| STATE["faiss"] = faiss.read_index( | |
| str(MODEL_DIR/"fiqh_index_xlmr.faiss")) | |
| print(f" ✅ FAISS {STATE['faiss'].ntotal} vectors dim=768") | |
| try: | |
| with open(MODEL_DIR/"metadata_v53.json") as f: | |
| STATE["meta"] = json.load(f) | |
| except Exception: | |
| STATE["meta"] = {"version":"5.5"} | |
| print(f"\n✅ MACI v5.5 ready | XLM-R F1=0.9539 | Structural pattern layer: {len(scan_structural_patterns('test')) if False else 'loaded'}") | |
| # ── Pydantic schemas ─────────────────────────────────────────── | |
| class ClassifyRequest(BaseModel): | |
| # v5.4 FIX: raised from 2000 to 8000 | |
| # Internal truncation to 1500 chars happens inside classify() | |
| text : str = Field(..., min_length=3, max_length=8000, | |
| description="Text to classify. " | |
| "Long texts are truncated to 1500 chars internally.") | |
| context : Optional[str] = Field("general", | |
| description="Deployment context: general, fintech, " | |
| "islamic_bank, social_media, regulatory") | |
| user_role : Optional[str] = "unknown" | |
| deployment_context : Optional[str] = "general" | |
| class MechanismItem(BaseModel): | |
| """Single mechanism for multi-mechanism audit.""" | |
| mechanism_id : str = Field(..., description="e.g. 'Mechanism_1' or 'SWAP'") | |
| description : str = Field(..., min_length=10, max_length=8000, | |
| description="Factual description of the mechanism — " | |
| "no compliance assertions, no headers") | |
| context : Optional[str] = "islamic_finance" | |
| class AuditRequest(BaseModel): | |
| """Multi-mechanism audit — for institutional submissions like Faisal's.""" | |
| submission_id : Optional[str] = Field(None, | |
| description="Your reference ID for this audit") | |
| mechanisms : List[MechanismItem] = Field(..., | |
| min_items=1, max_items=20, | |
| description="List of mechanisms to audit") | |
| jurisdiction : Optional[str] = Field("MY", | |
| description="Primary jurisdiction: MY AE PK SA GB US BH") | |
| auditor_note : Optional[str] = Field(None, max_length=500) | |
| # NOTE: ClassifyResponse is kept for reference / potential future use in | |
| # /docs, but it is intentionally NOT attached as response_model on the | |
| # /api/v1/classify route below. Attaching it there was the v5.4 bug: | |
| # FastAPI filters the returned dict down to exactly the fields declared | |
| # in response_model, silently dropping maqasid_enrichment, jurisdiction_overlay, | |
| # overclaiming_controls, handoff_integrity, runtime_handoff, and _compact | |
| # from every real response — even though classify() builds all of them. | |
| class MACIEvaluation(BaseModel): | |
| result : str | |
| violation_class : str | |
| confidence_score : float | |
| confidence_band : str | |
| severity : str | |
| model_used : str | |
| class BoundaryBehavior(BaseModel): | |
| recommendation : str | |
| safe_next_step : str | |
| review_required: bool | |
| review_reason : Optional[str] = None | |
| class ClassifyResponse(BaseModel): | |
| schema_version : str | |
| packet_id : str | |
| created_at_utc : str | |
| language : str | |
| maci_evaluation : MACIEvaluation | |
| authority_required : str | |
| evidence_pointer : str | |
| proposed_movement : str | |
| protected_effect_risk : str | |
| refusal_condition : Optional[str] = None | |
| boundary_behavior : BoundaryBehavior | |
| payload_hash : str | |
| # ── Core classify function ───────────────────────────────────── | |
| def classify(text: str, context: str = "general", | |
| jurisdiction: str = "MY") -> dict: | |
| # ── v5.4 FIX 1: Smart truncation ────────────────────────── | |
| # XLM-R tokenizes to max 128 tokens (~400-600 chars) | |
| # Truncate to 1500 chars to preserve most relevant content | |
| # while allowing long institutional descriptions | |
| original_length = len(text) | |
| # NEW (Phase 4): run the structural pattern layer on the FULL, | |
| # UNTRUNCATED text before truncation happens below. Structural | |
| # patterns (buy-back clauses, price-variation clauses, etc.) can | |
| # sit anywhere in a long institutional document, well past the | |
| # 1500-char point where the ML layer's view of the text is cut off. | |
| structural_flags = scan_structural_patterns(text) | |
| structural_flags_list = structural_flags_to_dict(structural_flags) | |
| if len(text) > 1500: | |
| text = text[:1500] | |
| was_truncated = original_length > 1500 | |
| ml = STATE["ml"] | |
| xlmr = STATE["xlmr"] | |
| tok = STATE["tokenizer"] | |
| idx = STATE["faiss"] | |
| rulings = STATE["rulings"] | |
| base_e = STATE["base_enc"] | |
| lang = detect_lang(text) | |
| partial = lang in ("FA","TJK") | |
| # XLM-R | |
| enc = tok(text, max_length=128, padding="max_length", | |
| truncation=True, return_tensors="pt").to(DEVICE) | |
| with torch.no_grad(): | |
| xlmr_probs = torch.softmax( | |
| xlmr(**enc).logits, dim=-1).cpu().numpy()[0] | |
| xlmr_conf = float(xlmr_probs.max()) | |
| # ML baseline | |
| X = hstack([ml["vw"].transform([text]), | |
| ml["vc"].transform([text]), | |
| ml["vs"].transform([text])]) | |
| ml_probs = ml["clf"].predict_proba(X)[0] | |
| # Ensemble | |
| if xlmr_conf >= 0.50: | |
| probs, model_used = xlmr_probs, "XLM-R" | |
| else: | |
| probs = 0.60*ml_probs + 0.40*xlmr_probs | |
| model_used = "ML+XLM-R blend" | |
| # FAISS evidence | |
| with torch.no_grad(): | |
| q_out = base_e(**enc) | |
| q_mask = enc["attention_mask"].unsqueeze(-1).float() | |
| q_emb = (q_out.last_hidden_state * q_mask).sum(1) / q_mask.sum(1) | |
| q_emb = torch.nn.functional.normalize(q_emb, dim=-1) | |
| q_emb = q_emb.cpu().numpy().astype(np.float32) | |
| sem_thresh = 0.38 if partial else 0.42 | |
| scores, idxs = idx.search(q_emb, 3) | |
| evidence = "no_fiqh_match" | |
| for sc, ri in zip(scores[0], idxs[0]): | |
| if ri != -1 and float(sc) >= sem_thresh: | |
| evidence = rulings[ri].get("text","")[:120] | |
| break | |
| # Core decision | |
| pred = int(np.argmax(probs)) | |
| conf = float(probs[pred]) | |
| label = LABEL_NAMES[pred] | |
| tier = ("HIGH" if conf>0.75 else "MEDIUM" if conf>0.45 else "LOW") | |
| verdict = ("VIOLATION" if pred!=0 and conf>0.45 else | |
| "AUTHENTIC" if pred==0 and conf>0.45 else | |
| "LOW_CONFIDENCE") | |
| terry = ("FLAGGED" if verdict=="VIOLATION" else | |
| "PASS" if verdict=="AUTHENTIC" else "UNCERTAIN") | |
| severity = SEV.get((pred,tier),"MEDIUM") | |
| # Self-claim downgrade | |
| if (severity=="CRITICAL" and verdict=="VIOLATION" | |
| and has_self_claim(text)): | |
| severity = "MEDIUM" | |
| verdict = "REVIEW_REQUIRED" | |
| terry = "UNCERTAIN" | |
| mv = MOVE[pred] | |
| if verdict in ("AUTHENTIC","PASS"): mv = "deliver_output_to_user" | |
| if severity=="CRITICAL" and verdict=="VIOLATION": mv = "quarantine_output" | |
| needs_review = tier=="LOW" or partial or verdict=="REVIEW_REQUIRED" | |
| review_reason = ( | |
| "self_claimed_compliance_with_violation_signals" | |
| if verdict=="REVIEW_REQUIRED" else | |
| "low_confidence" if tier=="LOW" else | |
| "farsi_tajik_partial" if partial else None) | |
| recommendation = ( | |
| "ALLOW" if verdict in ("AUTHENTIC","PASS") and not needs_review else | |
| "QUARANTINE" if severity=="CRITICAL" and verdict=="VIOLATION" else | |
| "ESCALATE" if verdict=="VIOLATION" else | |
| "NARROW" if verdict=="REVIEW_REQUIRED" else | |
| "REVIEW") | |
| # NEW (Phase 4): structural flags override the ML-only recommendation | |
| # whenever the ML layer alone would have let the text through cleanly. | |
| # CRITICAL structural flags force quarantine. HIGH structural flags | |
| # force escalation for human review, even if ML said AUTHENTIC/PASS. | |
| # A structural flag is never silently absorbed into an ALLOW verdict — | |
| # it can only make the outcome MORE cautious, never less. | |
| structural_critical = any(f["severity"] == "CRITICAL" for f in structural_flags_list) | |
| structural_high = any(f["severity"] == "HIGH" for f in structural_flags_list) | |
| structural_override_reason = None | |
| if structural_critical: | |
| recommendation = "QUARANTINE" | |
| needs_review = True | |
| structural_override_reason = "structural_pattern_critical" | |
| elif structural_high and recommendation == "ALLOW": | |
| recommendation = "ESCALATE" | |
| needs_review = True | |
| structural_override_reason = "structural_pattern_high" | |
| elif structural_flags_list and recommendation == "ALLOW": | |
| # Any structural flag at all (e.g. MEDIUM) still forces at least | |
| # a review step rather than a clean pass-through. | |
| recommendation = "REVIEW" | |
| needs_review = True | |
| structural_override_reason = "structural_pattern_present" | |
| if structural_override_reason: | |
| review_reason = structural_override_reason | |
| # ══════════════════════════════════════════════════════════ | |
| # NEW — PHASE 5: RECONCILIATION LAYER | |
| # | |
| # The Phase 4 block above only ever touched `recommendation` — | |
| # the top-line `violation_class` / `severity` / `result` fields | |
| # that the frontend actually displays still came straight from | |
| # the ML layer untouched. That's what caused the demo to show | |
| # "Authentic / NONE" on a real Arabic Bay'-al-Inah buy-back and | |
| # a real Urdu risk/ownership-decoupling clause even though the | |
| # structural layer had already matched them correctly — the | |
| # match existed in structural_pattern_flags, but nothing | |
| # upstream of it was told to care. | |
| # | |
| # This block does two things, and only these two things: | |
| # | |
| # (a) STRUCTURAL OVERRIDE — if a structural pattern matched, | |
| # it now overrides pred/label/conf/tier/verdict/terry/ | |
| # severity themselves, not just recommendation. Because | |
| # this runs BEFORE the Phase 2/3 enrichment sections below | |
| # (which key off `pred`), the pillar, AAOIFI standard, and | |
| # jurisdiction exposure/penalty all become consistent with | |
| # the corrected verdict automatically — no separate mapping | |
| # needed for those sections. | |
| # | |
| # (b) HARD-NEGATIVE DOWNGRADE — only when NO structural pattern | |
| # matched, and the ML layer flagged a violation, checks a | |
| # narrow set of known-compliant patterns (currently one: | |
| # CHN-01, the AAOIFI-permitted actual-loss indemnity formula | |
| # that was mis-flagged as Gharar/HIGH during pre-meeting | |
| # testing). A match downgrades the verdict to | |
| # REVIEW_REQUIRED — never silently back to AUTHENTIC. A | |
| # human still confirms; the system just stops presenting a | |
| # compliant clause as a confirmed violation. | |
| # | |
| # Neither path touches the base model's weights. Both are | |
| # deterministic, narrow, and fully logged in `reconciliation` | |
| # below so the audit trail always shows which layer produced the | |
| # final call. | |
| # ══════════════════════════════════════════════════════════ | |
| reconciliation = { | |
| "applied" : False, | |
| "source" : None, | |
| "reason" : None, | |
| "original_ml_violation_class": label, | |
| "original_ml_severity" : severity, | |
| "original_ml_confidence" : round(conf, 4), | |
| } | |
| if structural_flags_list: | |
| top = max( | |
| structural_flags_list, | |
| key=lambda f: STRUCTURAL_SEVERITY_RANK.get(f["severity"], 0), | |
| ) | |
| mapped_pred = STRUCTURAL_PATTERN_TO_PRED.get(top["pattern_id"]) | |
| if mapped_pred is not None: | |
| pred = mapped_pred | |
| label = f"{LABEL_NAMES[mapped_pred]} — {top['name']}" | |
| conf = max(conf, 0.95) # deterministic match — high confidence by construction | |
| tier = "HIGH" | |
| verdict = "VIOLATION" | |
| terry = "FLAGGED" | |
| severity = top["severity"] | |
| reconciliation.update({ | |
| "applied": True, | |
| "source" : "structural_override", | |
| "reason" : ( | |
| f"Deterministic structural pattern {top['pattern_id']} " | |
| f"({top['name']}) matched. This overrides the base " | |
| f"classifier's verdict for this text regardless of " | |
| f"language — structural patterns are validated directly " | |
| f"against SRB findings and are trusted over a " | |
| f"probabilistic language-signal miss." | |
| ), | |
| "structural_matches": [f["pattern_id"] for f in structural_flags_list], | |
| }) | |
| elif verdict == "VIOLATION": | |
| hard_negative = scan_hard_negatives(text, label, severity) | |
| if hard_negative is not None: | |
| severity = "LOW" | |
| verdict = "REVIEW_REQUIRED" | |
| terry = "UNCERTAIN" | |
| needs_review = True | |
| recommendation = "REVIEW" | |
| review_reason = f"hard_negative_pattern_match:{hard_negative.pattern_id}" | |
| reconciliation.update({ | |
| "applied": True, | |
| "source" : "hard_negative_downgrade", | |
| "reason" : ( | |
| f"Base classifier flagged '{reconciliation['original_ml_violation_class']}' " | |
| f"({reconciliation['original_ml_severity']}), but this text matches known " | |
| f"AAOIFI-compliant pattern {hard_negative.pattern_id} " | |
| f"({hard_negative.name}). Downgraded to REVIEW_REQUIRED — " | |
| f"a human confirms, the system does not self-clear to Authentic." | |
| ), | |
| "hard_negative_match": hard_negative.pattern_id, | |
| "hard_negative_aaoifi_standard": hard_negative.aaoifi_standard, | |
| }) | |
| # Recompute movement + recommendation now that pred/verdict/severity | |
| # may have changed above, so downstream fields stay consistent. | |
| mv = MOVE[pred] | |
| if verdict in ("AUTHENTIC","PASS"): mv = "deliver_output_to_user" | |
| if severity=="CRITICAL" and verdict=="VIOLATION": mv = "quarantine_output" | |
| if reconciliation["source"] == "structural_override": | |
| recommendation = "QUARANTINE" if severity == "CRITICAL" else "ESCALATE" | |
| needs_review = True | |
| # Phase 2 enrichment | |
| conseq = CONSEQUENCE_TABLE[pred] | |
| altmap = ALTERNATIVE_MAP[pred] | |
| sev_label, sev_score = modulate_severity(conseq["severity_score_base"], conf) | |
| econ_signal = (conseq["economic_signal_high"] if tier=="HIGH" | |
| else conseq["economic_signal_low"]) | |
| disclaimer = DISCLAIMER.get(tier) | |
| if partial and not disclaimer: | |
| disclaimer = "Farsi/Tajik input — partial coverage. Scholar review recommended." | |
| # NEW (Phase 5): when a structural override fired, use that pattern's | |
| # own AAOIFI citation and explanation instead of the generic per-class | |
| # ALTERNATIVE_MAP entry — it's more specific to the actual clause than | |
| # the class-level default. | |
| compliant_alternative = altmap["primary"] | |
| structural_fix = altmap["structural_fix"] | |
| aaoifi_standard = altmap["aaoifi_standard"] | |
| # NEW (Phase 6): for a generic (non-structural, non-hard-negative) | |
| # Gharar flag, replace the old hardcoded Takaful/Salam citation with | |
| # one chosen by detected contract type — see | |
| # GHARAR_CITATION_BY_CONTRACT_TYPE above for why this changed. | |
| if pred == 2 and reconciliation["source"] not in ("structural_override", "hard_negative_downgrade"): | |
| contract_type = detect_contract_type(text) | |
| citation = GHARAR_CITATION_BY_CONTRACT_TYPE.get(contract_type, GHARAR_CITATION_BY_CONTRACT_TYPE["murabaha"]) | |
| aaoifi_standard = citation["aaoifi_standard"] | |
| structural_fix = citation["structural_fix"] | |
| # NEW — a generic Riba flag on card-related text should cite SS2's | |
| # actual card-specific rule (credit cards can't carry an interest- | |
| # bearing revolving facility), not the generic Murabaha/Musharakah | |
| # citation, which says nothing about cards at all. | |
| elif pred == 1 and reconciliation["source"] not in ("structural_override", "hard_negative_downgrade"): | |
| if detect_contract_type(text) == "cards": | |
| citation = GHARAR_CITATION_BY_CONTRACT_TYPE["cards"] | |
| aaoifi_standard = citation["aaoifi_standard"] | |
| structural_fix = citation["structural_fix"] | |
| if reconciliation["source"] == "structural_override": | |
| aaoifi_standard = top["aaoifi_standard"] | |
| structural_fix = top["explanation"] | |
| elif reconciliation["source"] == "hard_negative_downgrade": | |
| aaoifi_standard = hard_negative.aaoifi_standard | |
| compliant_alternative = "No change required — matches a recognized AAOIFI-compliant pattern" | |
| structural_fix = hard_negative.explanation | |
| # Phase 3 jurisdiction | |
| juris_data = JURISDICTION_EXPOSURE.get(pred, {}) | |
| jurisdiction_recognized = jurisdiction in juris_data | |
| jcode = jurisdiction if jurisdiction_recognized else "MY" | |
| juris_info = juris_data.get(jcode, {"exposure":"unknown","penalty":"unknown"}) | |
| # Integrity | |
| pid = str(uuid.uuid4()) | |
| ts = datetime.now(timezone.utc).isoformat() | |
| phash = hashlib.sha256( | |
| json.dumps({"text":text[:200],"verdict":verdict, | |
| "label":label,"conf":round(conf,6)}, | |
| sort_keys=True).encode()).hexdigest() | |
| rhash = hashlib.sha256((pid+phash).encode()).hexdigest() | |
| packet = { | |
| "schema_version" : "MACI-0.4", | |
| "packet_id" : pid, | |
| "created_at_utc" : ts, | |
| "environment" : os.getenv("ENVIRONMENT","production"), | |
| "language" : lang, | |
| # Core | |
| "maci_evaluation" : { | |
| "result" : terry, | |
| "violation_class" : label, | |
| "confidence_score" : round(conf,4), | |
| "confidence_band" : tier, | |
| "severity" : severity, | |
| "model_used" : model_used, | |
| "all_probs" : {LABEL_NAMES[i]:round(float(probs[i]),4) | |
| for i in range(8)}, | |
| }, | |
| # Terry fields v0.1 | |
| "authority_required" : AUTH[pred], | |
| "evidence_pointer" : evidence, | |
| "proposed_movement" : mv, | |
| "protected_effect_risk" : RISK[pred], | |
| "refusal_condition" : REFU[pred], | |
| # Phase 2: Maqasid | |
| "maqasid_enrichment" : { | |
| "pillar" : conseq["maqasid_pillar"], | |
| "meaning" : conseq["maqasid_meaning"], | |
| "harm_scope" : conseq["harm_scope"], | |
| "severity_weight" : sev_label, | |
| "severity_score" : sev_score, | |
| "economic_signal" : econ_signal, | |
| "action_required" : (conseq["action_required"] | |
| if tier=="HIGH" else "human_review_first"), | |
| "compliant_alternative": compliant_alternative, | |
| "structural_fix" : structural_fix, | |
| "aaoifi_standard" : aaoifi_standard, | |
| }, | |
| # Phase 3: Jurisdiction | |
| "jurisdiction_overlay" : { | |
| "requested_jurisdiction" : jurisdiction, | |
| "jurisdiction_recognized" : jurisdiction_recognized, | |
| "jurisdiction" : jcode, | |
| "fallback_note" : (None if jurisdiction_recognized else | |
| f"'{jurisdiction}' is not a recognized jurisdiction code — " | |
| f"showing '{jcode}' data instead. Recognized codes: " | |
| f"MY, AE, PK, SA, GB, US, BH."), | |
| "exposure" : juris_info.get("exposure","unknown"), | |
| "regulatory_penalty" : juris_info.get("penalty","unknown"), | |
| "all_jurisdictions" : { | |
| j: {"exposure": juris_data.get(j,{}).get("exposure","N/A"), | |
| "penalty" : juris_data.get(j,{}).get("penalty","N/A")} | |
| for j in ["MY","AE","PK","SA","GB","US","BH"] | |
| } if pred != 0 else "compliant — no exposure", | |
| }, | |
| # Phase 4: Structural pattern layer | |
| "structural_pattern_flags": { | |
| "layer_type": "rule-based structural detection (Phase 4)", | |
| "note": ( | |
| "Separate from the ML language-signal layer above (maci_evaluation). " | |
| "Detects known structural red-flag patterns (e.g. Bay' al-Inah " | |
| "buy-back structures, disguised Riba via late-payment framed as " | |
| "income, unilateral price variation, risk/ownership decoupling, " | |
| "total liability waivers) that language-pattern classification does " | |
| "not catch by design. Runs on the full input text, prior to the " | |
| "1500-char truncation applied to the ML layer below. Coverage is " | |
| "limited to the patterns currently defined — absence of a flag " | |
| "here is NOT confirmation of structural compliance, only that none " | |
| "of the currently-modeled red-flag patterns matched. As of Phase 5, " | |
| "a match here also overrides maci_evaluation directly — see " | |
| "`reconciliation` below." | |
| ), | |
| "flags": structural_flags_list, | |
| "flag_count": len(structural_flags_list), | |
| }, | |
| # NEW — Phase 5: Reconciliation layer | |
| "reconciliation": reconciliation, | |
| # Overclaiming controls | |
| "overclaiming_controls" : { | |
| "severity_is_confidence_modulated": True, | |
| "base_severity_score" : conseq["severity_score_base"], | |
| "applied_severity_score": sev_score, | |
| "modulation_formula" : "base_score × confidence", | |
| "epistemic_disclaimer": disclaimer, | |
| "input_truncated" : was_truncated, | |
| "original_char_count" : original_length, | |
| "structural_layer_overrode_ml_verdict": bool(structural_override_reason), | |
| "structural_override_reason": structural_override_reason, | |
| "reconciliation_applied": reconciliation["applied"], | |
| "reconciliation_source": reconciliation["source"], | |
| }, | |
| "boundary_behavior" : { | |
| "recommendation" : recommendation, | |
| "safe_next_step" : mv, | |
| "review_required" : needs_review, | |
| "review_reason" : review_reason, | |
| }, | |
| "handoff_integrity" : { | |
| "maci_receipt_hash" : f"sha256:{rhash}", | |
| "payload_hash" : f"sha256:{phash}", | |
| "signature_status" : "unsigned", | |
| "audit_record_available" : True, | |
| "replay_context_available" : True, | |
| }, | |
| "runtime_handoff" : { | |
| "handoff_target" : | |
| "Elyria_Consequence_Boundary_Runtime_Surface", | |
| "handoff_type" : "signal_packet", | |
| "protected_kernel_material_requested" : False, | |
| "expected_response_fields" : [ | |
| "boundary_decision","public_reason_class", | |
| "allowed_next_movement","receipt_hash", | |
| "replay_token","protected_material_disclosed", | |
| ], | |
| }, | |
| "_compact" : { | |
| "timestamp" : ts, | |
| "language" : lang, | |
| "result" : terry, | |
| "violation" : label, | |
| "confidence" : tier, | |
| "severity" : severity, | |
| "action" : recommendation, | |
| "jurisdiction": jcode, | |
| "exposure" : juris_info.get("exposure","unknown"), | |
| "truncated" : was_truncated, | |
| "structural_flags": len(structural_flags_list), | |
| "reconciliation_applied": reconciliation["applied"], | |
| "audit_hash" : f"sha256:{phash[:16]}...", | |
| }, | |
| "payload_hash" : f"sha256:{phash}", | |
| } | |
| return packet | |
| # ── Routes ───────────────────────────────────────────────────── | |
| def root(): | |
| meta = STATE.get("meta",{}) | |
| return { | |
| "name" : "MACI — Maqasid AI Compliance Intelligence", | |
| "version" : "5.5", | |
| "status" : "online", | |
| "xlmr_f1" : meta.get("xlmr_test_f1", 0.9539), | |
| "ml_f1" : meta.get("ml_f1", 0.9026), | |
| "schema" : "MACI-0.4", | |
| "classes" : list(LABEL_NAMES.values()), | |
| "languages" : ["EN","AR","FA","UR","TJK","Arabizi"], | |
| "structural_pattern_layer": "Phase 4 — 5 patterns active " | |
| "(Bay' al-Inah, disguised Riba late-fee, " | |
| "unilateral price variation, risk/ownership " | |
| "decoupling, total liability waiver)", | |
| "reconciliation_layer": "Phase 5 — structural patterns override " | |
| "maci_evaluation directly; one hard-negative " | |
| "pattern (CHN-01, actual-loss indemnity) " | |
| "downgrades known false positives to REVIEW_REQUIRED", | |
| "endpoints" : { | |
| "single" : "POST /api/v1/classify", | |
| "batch" : "POST /api/v1/classify/batch", | |
| "audit" : "POST /api/v1/classify/audit", | |
| "simple" : "POST /api/v1/classify/simple", | |
| "health" : "GET /health", | |
| "docs" : "GET /docs", | |
| }, | |
| "docs" : "/docs", | |
| } | |
| def health(): | |
| loaded = "xlmr" in STATE and "ml" in STATE | |
| return { | |
| "status" : "ok" if loaded else "loading", | |
| "models_loaded": loaded, | |
| "device" : DEVICE, | |
| "version" : "5.5", | |
| "xlmr_f1" : 0.9539, | |
| "ml_f1" : 0.9026, | |
| "rulings" : len(STATE.get("rulings",[])), | |
| "structural_pattern_layer": "active", | |
| "reconciliation_layer": "active", | |
| } | |
| def classify_endpoint(req: ClassifyRequest): | |
| if "xlmr" not in STATE: | |
| raise HTTPException(503,"Models still loading — retry in 30s") | |
| try: | |
| return classify(req.text, req.context, | |
| req.deployment_context or "MY") | |
| except Exception as e: | |
| raise HTTPException(500, str(e)) | |
| def classify_simple(text: str = Form(..., | |
| min_length=3, | |
| max_length=8000)): | |
| """Accept plain form text — no JSON wrapper needed.""" | |
| if "xlmr" not in STATE: | |
| raise HTTPException(503,"Models still loading") | |
| try: | |
| return classify(text) | |
| except Exception as e: | |
| raise HTTPException(500, str(e)) | |
| def classify_batch(texts: List[str]): | |
| if len(texts) > 50: | |
| raise HTTPException(400,"Max 50 texts per batch") | |
| if "xlmr" not in STATE: | |
| raise HTTPException(503,"Models still loading") | |
| return [classify(t) for t in texts] | |
| def classify_audit(req: AuditRequest): | |
| """ | |
| Multi-mechanism audit endpoint. | |
| Each mechanism is classified independently. | |
| Returns individual packets + aggregate summary. | |
| """ | |
| if "xlmr" not in STATE: | |
| raise HTTPException(503,"Models still loading") | |
| ts = datetime.now(timezone.utc).isoformat() | |
| audit_id = req.submission_id or str(uuid.uuid4()) | |
| jurisdiction = req.jurisdiction or "MY" | |
| results = [] | |
| flags = [] | |
| passes = [] | |
| for mech in req.mechanisms: | |
| try: | |
| packet = classify(mech.description, | |
| mech.context or "islamic_finance", | |
| jurisdiction) | |
| packet["mechanism_id"] = mech.mechanism_id | |
| results.append(packet) | |
| result = packet["maci_evaluation"]["result"] | |
| structural_count = packet.get("structural_pattern_flags", {}).get("flag_count", 0) | |
| if result == "FLAGGED" or structural_count > 0: | |
| flags.append({ | |
| "mechanism_id" : mech.mechanism_id, | |
| "violation" : packet["maci_evaluation"]["violation_class"], | |
| "severity" : packet["maci_evaluation"]["severity"], | |
| "confidence" : packet["maci_evaluation"]["confidence_score"], | |
| "recommendation" : packet["boundary_behavior"]["recommendation"], | |
| "alternative" : packet.get("maqasid_enrichment",{}) | |
| .get("compliant_alternative"), | |
| "structural_fix" : packet.get("maqasid_enrichment",{}) | |
| .get("structural_fix",""), | |
| "structural_pattern_flags": packet.get("structural_pattern_flags", {}) | |
| .get("flags", []), | |
| }) | |
| else: | |
| passes.append(mech.mechanism_id) | |
| except Exception as e: | |
| results.append({ | |
| "mechanism_id" : mech.mechanism_id, | |
| "error" : str(e), | |
| "maci_evaluation": {"result":"ERROR"}, | |
| }) | |
| # Aggregate summary | |
| total = len(req.mechanisms) | |
| flag_count = len(flags) | |
| pass_count = len(passes) | |
| overall = ("ALL_PASS" if flag_count == 0 else | |
| "CRITICAL_FLAGS" if any( | |
| f["severity"] == "CRITICAL" for f in flags) else | |
| "FLAGS_PRESENT") | |
| overall_action= ("APPROVE" if overall == "ALL_PASS" else | |
| "QUARANTINE" if overall == "CRITICAL_FLAGS" else | |
| "REVIEW_REQUIRED") | |
| # Jurisdiction summary for flagged mechanisms | |
| juris_exposure = {} | |
| for f in flags: | |
| vid = next((k for k,v in LABEL_NAMES.items() | |
| if v == f["violation"]), None) | |
| if vid is not None: | |
| je = JURISDICTION_EXPOSURE.get(vid,{}).get(jurisdiction,{}) | |
| juris_exposure[f["mechanism_id"]] = { | |
| "exposure": je.get("exposure","unknown"), | |
| "penalty" : je.get("penalty","unknown"), | |
| } | |
| audit_hash = hashlib.sha256( | |
| json.dumps({"audit_id":audit_id,"ts":ts, | |
| "total":total,"flags":flag_count}, | |
| sort_keys=True).encode()).hexdigest() | |
| return { | |
| "schema_version" : "MACI-0.4-audit", | |
| "audit_id" : audit_id, | |
| "created_at_utc" : ts, | |
| "jurisdiction" : jurisdiction, | |
| "auditor_note" : req.auditor_note, | |
| # ── Aggregate summary ────────────────────────────────── | |
| "audit_summary" : { | |
| "overall_result" : overall, | |
| "overall_action" : overall_action, | |
| "total_mechanisms" : total, | |
| "flagged" : flag_count, | |
| "passed" : pass_count, | |
| "pass_rate" : f"{pass_count/total:.0%}" if total else "0%", | |
| "flags_detail" : flags, | |
| "passed_mechanisms": passes, | |
| "jurisdiction_exposure": juris_exposure, | |
| }, | |
| # ── Per-mechanism packets ────────────────────────────── | |
| "mechanism_results": results, | |
| "audit_hash" : f"sha256:{audit_hash}", | |
| } |