Spaces:
Sleeping
Sleeping
| import spacy | |
| import nltk | |
| from nltk.sentiment.vader import SentimentIntensityAnalyzer | |
| from spacy.cli import download | |
| class DeepPragmaEngine: | |
| """ | |
| Auxiliary lexical/syntactic analyzer. | |
| NOTE: these heuristics are NOT the primary classifier for sarcasm/hate — | |
| that responsibility belongs to GroqClient.classify(), since irony relies | |
| on pragmatic context that word-level features can't reliably capture. | |
| This class only surfaces supporting signals (sentiment, a naive | |
| syntactic "attack" pattern) to log or display alongside the LLM's | |
| decision, not to gate it. | |
| """ | |
| HATE_TERMS = ["stupid", "useless", "disease", "parasite"] | |
| IRONY_MARKERS = ["always", "oh", "sure"] | |
| def __init__(self): | |
| try: | |
| nltk.data.find("sentiment/vader_lexicon.zip") | |
| except LookupError: | |
| nltk.download("vader_lexicon") | |
| model_name = "en_core_web_sm" | |
| try: | |
| self.nlp = spacy.load(model_name) | |
| except OSError: | |
| print(f"Model {model_name} not found. Downloading...") | |
| download(model_name) | |
| self.nlp = spacy.load(model_name) | |
| self.sia = SentimentIntensityAnalyzer() | |
| def get_signals(self, text: str) -> dict: | |
| """Compute auxiliary lexical/syntactic signals (informational only).""" | |
| doc = self.nlp(text) | |
| sentiment = self.sia.polarity_scores(text) | |
| attack_pattern = False | |
| for token in doc: | |
| if token.dep_ == "nsubj" and token.head.lemma_ == "be": | |
| for child in token.head.children: | |
| if child.dep_ in ("acomp", "attr") and child.text.lower() in self.HATE_TERMS: | |
| attack_pattern = True | |
| irony_hint = (sentiment["pos"] > 0.3) and any( | |
| marker in text.lower() for marker in self.IRONY_MARKERS | |
| ) | |
| return { | |
| "sentiment": sentiment, | |
| "attack_pattern": attack_pattern, | |
| "irony_hint": irony_hint, | |
| } |