RJ3vans's picture
Update app.py
2dbcc55 verified
Raw History Blame Contribute Delete
107 kB
# app.py - Encapsulated version of the Colab notebook for deployment as a web app (e.g., on Hugging Face Spaces)
# This refactors the original 1501 lines into a modular structure: global model loading, a main processing function,
# and a Gradio interface. Based on the notebook's structure, which includes text preprocessing, sign tagging with
# Hugging Face models, span tagging, and recursive simplification. Full original functions are assumed to be
# integrated here (placeholders for complex ones like SSCCVsimplify). Adjust as needed if you have the exact defs.
'''
cd d:\TextSimplificationDemo
# Delete old venv (if needed)
Remove-Item -Recurse -Force .\venv
# Create venv with Python 3.11
py -3.11 -m venv venv
# Activate it
.\venv\Scripts\Activate.ps1
# Upgrade pip
python -m pip install --upgrade pip
# Install packages
pip install -r requirements.txt
# Install spacy-alignments safely
pip install spacy-alignments --only-binary :all:
# Download spaCy model
python -m spacy download en_core_web_sm
# Run the app
python app.py
# Interface at http://127.0.0.1:7860/
EXAMPLE SENTENCE:
John is one test, Mary, who lives in London, loves exams, and Bill eats bread.
John went to Wolverhampton, Bill set out for Coventry, and Mary returned from Birmingham.
The Environment Agency said record winter rainfall had helped to replenish reservoirs and groundwater, but reservoir storage has now fallen to 7.4% below the average for this time of year.
The Environment Agency said record winter rainfall had helped to replenish reservoirs and groundwater, but reservoir storage has now fallen to 7.4 below the average for this time of year.
Reservoir storage of 8.2 has been recorded and Mary went home.
The demo loops infinitely when more than 1 sentence is included on the line.
John, who lives in Wolverhampton, which is a town in the West Midlands, went home.
'''
import spaces
import re
import sys
import torch
from transformers import AutoModelForTokenClassification, AutoTokenizer
import spacy
import xml.etree.ElementTree as ET
import spacy_alignments as tokenizations
import xml.dom.minidom
import pandas as pd
import gradio as gr
import os
from collections import deque # add this at the top of app.py if not already there
from collections import defaultdict
import html
PRONOUNS = {
"he", "she", "it", "they", "him", "her", "them",
"his", "hers", "its", "their", "theirs",
}
seen = set()
import io
hf_token = os.environ.get("HF_TOKEN") or os.environ.get("HUGGING_FACE_HUB_TOKEN")
device = "cuda" if torch.cuda.is_available() else "cpu"
print("Device:", device)
# Fix for Windows Unicode printing issues
if sys.platform == "win32":
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8', errors='replace')
# Global configurations
test_on_unrestricted_text = True # Set to False for test sentences
# Load spaCy model
nlp = spacy.load("en_core_web_sm") # Assuming this is used; load once
# Load Hugging Face models and tokenizers (done globally to avoid reloading)
device = "cuda" if torch.cuda.is_available() else "cpu"
print("Device:", device)
sign_tagging_model_name = "RJ3vans/SignTagger"
SignTaggingTokenizer = AutoTokenizer.from_pretrained(sign_tagging_model_name, token=hf_token)
SignTaggingModel = AutoModelForTokenClassification.from_pretrained(
sign_tagging_model_name, token=hf_token
).to(device)
SignTaggingModel.eval()
ssccv_span_tagging_model_name = "RJ3vans/SSCCVspanTagger"
SSCCVspanTaggingTokenizer = AutoTokenizer.from_pretrained(ssccv_span_tagging_model_name, token=hf_token)
SSCCVspanTaggingModel = AutoModelForTokenClassification.from_pretrained(
ssccv_span_tagging_model_name, token=hf_token
).to(device)
SSCCVspanTaggingModel.eval()
ssmn_span_tagging_model_name = "RJ3vans/SSMNspanTagger"
SSMNspanTaggingTokenizer = AutoTokenizer.from_pretrained(ssmn_span_tagging_model_name, token=hf_token)
SSMNspanTaggingModel = AutoModelForTokenClassification.from_pretrained(
ssmn_span_tagging_model_name, token=hf_token
).to(device)
SSMNspanTaggingModel.eval()
ccv_span_tagging_model_name = "RJ3vans/CCVspanTagger"
CCVspanTaggingTokenizer = AutoTokenizer.from_pretrained(ccv_span_tagging_model_name, token=hf_token)
CCVspanTaggingModel = AutoModelForTokenClassification.from_pretrained(
ccv_span_tagging_model_name, token=hf_token
).to(device)
CCVspanTaggingModel.eval()
cmv1_span_tagging_model_name = "RJ3vans/CMV1spanTagger"
CMV1spanTaggingTokenizer = AutoTokenizer.from_pretrained(cmv1_span_tagging_model_name, token=hf_token)
CMV1spanTaggingModel = AutoModelForTokenClassification.from_pretrained(
cmv1_span_tagging_model_name, token=hf_token
).to(device)
CMV1spanTaggingModel.eval()
cmn1_span_tagging_model_name = "RJ3vans/CMN1spanTagger"
CMN1spanTaggingTokenizer = AutoTokenizer.from_pretrained(cmn1_span_tagging_model_name, token=hf_token)
CMN1spanTaggingModel = AutoModelForTokenClassification.from_pretrained(
cmn1_span_tagging_model_name, token=hf_token
).to(device)
CMN1spanTaggingModel.eval()
cln_span_tagging_model_name = "RJ3vans/CLNspanTagger"
CLNspanTaggingTokenizer = AutoTokenizer.from_pretrained(cln_span_tagging_model_name, token=hf_token)
CLNspanTaggingModel = AutoModelForTokenClassification.from_pretrained(
cln_span_tagging_model_name, token=hf_token
).to(device)
CLNspanTaggingModel.eval()
# Define label lists (extracted from notebook summary; expand with full 70+ if needed)
sign_label_list = [
"M:N_CCV",
"M:N_CIN",
"M:N_CLA",
"M:N_CLAdv",
"M:N_CLN",
"M:N_CLP",
"M:N_CLQ",
"M:N_CLV",
"M:N_CMA1",
"M:N_CMAdv",
"M:N_CMN1",
"M:N_CMN2",
"M:N_CMN3",
"M:N_CMN4",
"M:N_CMP",
"M:N_CMP2",
"M:N_CMV1",
"M:N_CMV2",
"M:N_CMV3",
"M:N_COMBINATORY",
"M:N_CPA",
"M:N_ESAdvP",
"M:N_ESCCV",
"M:N_ESCM",
"M:N_ESMA",
"M:N_ESMAdvP",
"M:N_ESMI",
"M:N_ESMN",
"M:N_ESMP",
"M:N_ESMV",
"M:N_HELP",
"M:N_SPECIAL",
"M:N_SSCCV",
"M:N_SSCM",
"M:N_SSMA",
"M:N_SSMAdvP",
"M:N_SSMI",
"M:N_SSMN",
"M:N_SSMP",
"M:N_SSMV",
"M:N_STQ",
"M:N_V",
"M:N_nan",
"M:Y_CCV",
"M:Y_CIN",
"M:Y_CLA",
"M:Y_CLAdv",
"M:Y_CLN",
"M:Y_CLP",
"M:Y_CLQ",
"M:Y_CLV",
"M:Y_CMA1",
"M:Y_CMAdv",
"M:Y_CMN1",
"M:Y_CMN2",
"M:Y_CMN4",
"M:Y_CMP",
"M:Y_CMP2",
"M:Y_CMV1",
"M:Y_CMV2",
"M:Y_CMV3",
"M:Y_COMBINATORY",
"M:Y_CPA",
"M:Y_ESAdvP",
"M:Y_ESCCV",
"M:Y_ESCM",
"M:Y_ESMA",
"M:Y_ESMAdvP",
"M:Y_ESMI",
"M:Y_ESMN",
"M:Y_ESMP",
"M:Y_ESMV",
"M:Y_HELP",
"M:Y_SPECIAL",
"M:Y_SSCCV",
"M:Y_SSCM",
"M:Y_SSMA",
"M:Y_SSMAdvP",
"M:Y_SSMI",
"M:Y_SSMN",
"M:Y_SSMP",
"M:Y_SSMV",
"M:Y_STQ",
]
SSCCVspan_label_list = [
"AFTER_ADJECTIVAL",
"AFTER_ADVERBIAL",
"AFTER_CLEFT_CLAUSE",
"AFTER_COGNITIVE_COMMUNICATIVE_VP",
"AFTER_COMPLEX_NP",
"AFTER_COMPLEX_PHRASE",
"AFTER_FREE_RELATIVE_CLAUSE",
"AFTER_INTENSIFYING_CLAUSE",
"AFTER_REPORTING_CLAUSE",
"AFTER_RESTRICTIVE_CLAUSE_GENERIC_HEAD",
"AFTER_WH_PHRASE",
"BEFORE_ADJECTIVAL",
"BEFORE_ADVERBIAL",
"BEFORE_CLEFT_CLAUSE",
"BEFORE_COGNITIVE_COMMUNICATIVE_VP",
"BEFORE_COMPLEX_NP",
"BEFORE_COMPLEX_PHRASE",
"BEFORE_FREE_RELATIVE_CLAUSE",
"BEFORE_INTENSIFYING_CLAUSE",
"BEFORE_REPORTING_CLAUSE",
"BEFORE_RESTRICTIVE_CLAUSE_GENERIC_HEAD",
"BEFORE_WH_PHRASE",
"ERROR",
"IN_ADJECTIVAL",
"IN_ADJECTIVAL_AFTERSIGN",
"IN_ADJECTIVAL_BEFORESIGN",
"IN_ADVERBIAL",
"IN_ADVERBIAL_AFTERSIGN",
"IN_ADVERBIAL_BEFORESIGN",
"IN_CLEFT_CLAUSE",
"IN_CLEFT_CLAUSE_AFTERSIGN",
"IN_CLEFT_CLAUSE_BEFORESIGN",
"IN_COGNITIVE_COMMUNICATIVE_VP",
"IN_COGNITIVE_COMMUNICATIVE_VP_AFTERSIGN",
"IN_COGNITIVE_COMMUNICATIVE_VP_BEFORESIGN",
"IN_COMPLEX_NP",
"IN_COMPLEX_NP_AFTERSIGN",
"IN_COMPLEX_NP_BEFORESIGN",
"IN_COMPLEX_PHRASE",
"IN_COMPLEX_PHRASE_AFTERSIGN",
"IN_COMPLEX_PHRASE_BEFORESIGN",
"IN_FREE_RELATIVE_CLAUSE",
"IN_FREE_RELATIVE_CLAUSE_AFTERSIGN",
"IN_FREE_RELATIVE_CLAUSE_BEFORESIGN",
"IN_INTENSIFYING_CLAUSE",
"IN_INTENSIFYING_CLAUSE_AFTERSIGN",
"IN_INTENSIFYING_CLAUSE_BEFORESIGN",
"IN_REPORTING_CLAUSE_AFTERSIGN",
"IN_REPORTING_CLAUSE_BEFORESIGN",
"IN_RESTRICTIVE_CLAUSE_GENERIC_HEAD",
"IN_RESTRICTIVE_CLAUSE_GENERIC_HEAD_AFTERSIGN",
"IN_RESTRICTIVE_CLAUSE_GENERIC_HEAD_BEFORESIGN",
"IN_WH_PHRASE",
"IN_WH_PHRASE_AFTERSIGN",
"IN_WH_PHRASE_BEFORESIGN",
"UNKNOWN",
]
SSMNspan_label_list = [
"AFTER_COMPLEX_NP",
"BEFORE_COMPLEX_NP",
"ERROR",
"IN_COMPLEX_NP",
"IN_COMPLEX_NP_AFTERSIGN",
"IN_COMPLEX_NP_BEFORESIGN",
"UNKNOWN",
]
CCVspan_label_list = [
"AFTER_COMPOUND",
"BEFORE_COMPOUND",
"IN_COMPOUND",
"IN_COMPOUND_AFTERSIGN",
"IN_COMPOUND_BEFORESIGN",
"NOT_CLAUSE_COORDINATOR",
"UNKNOWN",
]
CMV1span_label_list = [
"AFTER_COMPOUND",
"BEFORE_COMPOUND",
"IN_COMPOUND",
"IN_COMPOUND_AFTERSIGN",
"IN_COMPOUND_BEFORESIGN",
"NOT_VP_COORDINATOR",
"UNKNOWN",
]
CMN1span_label_list = [
"AFTER_COMPOUND",
"BEFORE_COMPOUND",
"IN_COMPOUND",
"IN_COMPOUND_AFTERSIGN",
"IN_COMPOUND_BEFORESIGN",
"NOT_NP_COORDINATOR",
"UNKNOWN",
]
CLNspan_label_list = [
"AFTER_COMPOUND",
"BEFORE_COMPOUND",
"IN_COMPOUND",
"IN_COMPOUND_AFTERSIGN",
"IN_COMPOUND_BEFORESIGN",
"NOT_NOUN_COORDINATOR",
"UNKNOWN",
]
required_sign_tags = [
"M:Y_SSCCV",
"M:Y_SSMN",
"M:Y_CCV",
"M:Y_CMV1",
"M:Y_CMP",
"M:Y_CMN1",
"M:Y_CLN",
"M:Y_CLV",
"M:Y_CLP",
]
###############################################################################
# Helper functions (refactored from notebook)
###############################################################################
###############################################################################
def is_cognitive_communicative_vp(preds):
return any("COGNITIVE_COMMUNICATIVE_VP" in str(p) for p in (preds or []))
###############################################################################
###############################################################################
# Third-person (and similar) pronouns that usually need an antecedent.
# Omit I/you/me — deictic, not anaphoric in the same way.
_ANAPHORIC_PRONOUNS = {
"he", "she", "it", "they",
"him", "her", "them",
# "we", "us", # optional; remove if you prefer
}
def sentence_has_anaphoric_pronoun(sent: str) -> bool:
tokens = re.findall(r"[A-Za-z']+", (sent or "").lower())
return any(t in _ANAPHORIC_PRONOUNS for t in tokens)
def sentence_entity_key(sent: str) -> str:
"""Primary entity ≈ subject head, lowercased lemma/text."""
doc = nlp(sent)
subj = next((t for t in doc if t.dep_ in ("nsubj", "nsubjpass")), None)
if subj is None:
subj = next((t for t in doc if t.pos_ in ("PROPN", "NOUN", "PRON")), None)
if subj is None:
return sent.lower()[:40]
# Use lemma for "man"/"men"; text for names
if subj.pos_ == "PROPN":
return subj.text.lower()
return subj.lemma_.lower()
def is_pronoun_sentence(sent: str) -> bool:
doc = nlp(sent)
for t in doc:
if t.dep_ in ("nsubj", "nsubjpass") and t.lower_ in PRONOUNS:
return True
return bool(re.search(r"\b(he|she|it|they|him|her|them)\b", sent, re.I))
def is_identifying(sent: str) -> bool:
"""Property / location / copula — good 'given' material."""
return bool(
re.search(
r"\b(is|are|was|were|lives?|lived|lives in|is a|is an)\b",
sent,
re.I,
)
)
def order_sentences_for_discourse(sents):
"""
1. Pronoun sentences last (antecedent first).
2. Group remaining sentences by subject entity.
3. Within a group: identifying sentences before event sentences.
4. Group order = first appearance in the current list
(≈ order derived from the source sentence).
"""
cleaned = [s.strip() for s in sents if s and s.strip()]
if len(cleaned) <= 1:
return cleaned
pronoun_sents = []
lexical_sents = []
for s in cleaned:
(pronoun_sents if is_pronoun_sentence(s) else lexical_sents).append(s)
groups = defaultdict(list)
order = []
for s in lexical_sents:
key = sentence_entity_key(s)
if key not in groups:
order.append(key)
groups[key].append(s)
ordered = []
for key in order:
chunk = groups[key]
ident = [s for s in chunk if is_identifying(s)]
rest = [s for s in chunk if s not in ident]
ordered.extend(ident + rest)
ordered.extend(pronoun_sents)
return ordered
###############################################################################
###############################################################################
def split_into_sentences(text):
"""
Split plain text into sentences using spaCy.
Falls back to a conservative regex if spaCy returns a single
span that still contains '. ' + capital letter.
"""
text = (text or "").strip()
if not text:
return []
blocks = [b.strip() for b in text.split("\n") if b.strip()]
sents = []
for block in blocks:
doc = nlp(block)
piece = [s.text.strip() for s in doc.sents if s.text.strip()]
# Fallback: spaCy sometimes keeps two sentences as one span
expanded = []
for p in piece:
parts = re.split(r'(?<=[.!?])\s+(?=[A-Z])', p)
expanded.extend(x.strip() for x in parts if x.strip())
sents.extend(expanded)
# de-dupe while preserving order (can happen with odd whitespace)
seen = set()
unique = []
for s in sents:
key = re.sub(r"\s+", " ", s.lower())
if key not in seen:
seen.add(key)
unique.append(s)
return unique
###############################################################################
###############################################################################
def strip_xml_tags(text):
"""Remove all <M:Y_...> and </M:Y_...> tags."""
text = re.sub(r'<M:Y_[^>]+>', '', text)
text = re.sub(r'</M:Y_[^>]+>', '', text)
text = re.sub(r'\s+', ' ', text).strip()
return text
###############################################################################
###############################################################################
def clean_for_spaCy(sent):
print("CLEANING sent BEFORE spaCy processes it", sent)
# Single [...] region; no (.|\s)* — avoids ReDoS
# Supports: [,], [and], [who], [, who], [, and], [,_and] (legacy)
m = re.search(
r"^(.*?)"
r"\["
r"("
r"[,;:(]|"
r"and|but|or|that|what|when|where|which|while|who"
r")"
r"("
r"_(?:and|but|or|that|what|when|where|which|while|who)|"
r"\s+(?:and|but|or|that|what|when|where|which|while|who)"
r")?"
r"\]"
r"(.*)$",
sent,
flags=re.IGNORECASE | re.DOTALL,
)
if m:
prefix = m.group(1)
head = m.group(2)
extra = m.group(3) or ""
suffix = m.group(4)
if extra.startswith("_"):
# legacy bigram: ,_and → ,_and inside brackets
inside = head + extra
elif extra:
# [, who] / [, and]
inside = f"{head} {extra.strip()}"
else:
inside = head
clean_sent = f"{prefix}[{inside}]{suffix}"
else:
clean_sent = sent
clean_sent = re.sub(r"\s+", " ", clean_sent).strip()
print("clean_sent is\n" + clean_sent)
return clean_sent
###############################################################################
###############################################################################
# Tidying up predicted tags involving [ and ].
def sign_cleaning(
aligned_tokens, aligned_pos_tags, aligned_lemmas, aligned_predictions
):
clean_tokens = []
clean_pos_tags = []
clean_lemmas = []
clean_predictions = []
x = 0
while x < len(aligned_tokens):
# and
if aligned_tokens[x] == "[" and aligned_tokens[x + 1] == ":]":
clean_predictions.append(aligned_predictions[x])
clean_tokens.append(":")
clean_pos_tags.append(":")
clean_lemmas.append(":")
x += 1
elif aligned_tokens[x] == "[" and aligned_tokens[x + 2] == "]":
clean_predictions.append(aligned_predictions[x])
clean_tokens.append(aligned_tokens[x + 1])
clean_pos_tags.append(aligned_pos_tags[x + 1])
clean_lemmas.append(aligned_lemmas[x + 1])
x += 1
# comma-and
elif aligned_tokens[x] == "[" and aligned_tokens[x + 4] == "]":
clean_predictions.append(aligned_predictions[x])
clean_predictions.append(aligned_predictions[x])
clean_predictions.append(aligned_predictions[x])
clean_pos_tags.append(aligned_pos_tags[x + 1])
clean_pos_tags.append(aligned_pos_tags[x + 2])
clean_pos_tags.append(aligned_pos_tags[x + 3])
clean_lemmas.append(aligned_lemmas[x + 1])
clean_lemmas.append(aligned_lemmas[x + 2])
clean_lemmas.append(aligned_lemmas[x + 3])
clean_tokens.append(aligned_tokens[x + 1])
clean_tokens.append(aligned_tokens[x + 2])
clean_tokens.append(aligned_tokens[x + 3])
x += 3
elif aligned_tokens[x] == "]" and aligned_tokens[x - 2] == "[":
True
elif aligned_tokens[x] == "]" and aligned_tokens[x - 4] == "[":
True
else:
clean_tokens.append(aligned_tokens[x])
clean_pos_tags.append(aligned_pos_tags[x])
clean_lemmas.append(aligned_lemmas[x])
clean_predictions.append(aligned_predictions[x])
x += 1
new_clean_predictions = improve_predicted_tag_sequence(clean_predictions)
clean_predictions = new_clean_predictions
return (clean_tokens, clean_pos_tags, clean_lemmas, clean_predictions)
###############################################################################
###############################################################################
def improve_predicted_tag_sequence(predictions):
new_predictions = []
# print('IMPROVING PREDICTED TAG SEQUENCE')
# print(predictions)
core_tags = []
x = 0
sign_position = False
while x < len(predictions):
px = predictions[x]
core_tag = re.sub(
"(IN_|BEFORE_|AFTER_|_AFTERSIGN|_BEFORESIGN)",
"",
px,
)
core_tags.append(core_tag)
x += 1
most_frequent_core_tag = max(set(core_tags), key=core_tags.count)
sign_tag = re.compile(r"IN_" + most_frequent_core_tag + "$")
x = 0
while x < len(predictions):
px = predictions[x]
if re.match(
sign_tag,
px,
):
sign_position = x
break
x += 1
improved_predicted_tag_sequence = []
x = 0
while x < len(predictions):
px = predictions[x]
if x < sign_position:
if px == "UNKNOWN" or px == "NOT_CLAUSE_COORDINATOR":
px = "IN_" + core_tag + "_BEFORESIGN"
improved_predicted_tag_sequence.append(px)
x += 1
predictions = improved_predicted_tag_sequence
# print('improved_predicted_tag_sequence:', improved_predicted_tag_sequence)
x = 0
previous_prediction = ""
while x < len(predictions):
if x > 0:
if (
predictions[x] == "UNKNOWN"
or predictions[x] == "NOT_CLAUSE_COORDINATOR"
):
# new_predictions.append(previous_prediction)
try:
new_predictions.append(new_predictions[-1])
except:
new_predictions.append(predictions[x])
else:
new_predictions.append(predictions[x])
else:
new_predictions.append(predictions[x])
x += 1
# print('new_predictions:', new_predictions)
return new_predictions
###############################################################################
###############################################################################
def indefinite_article_for(word: str) -> str:
"""Choose a/an from the following orthographic word."""
if not word:
return "a"
w = re.sub(r"[^A-Za-z]", "", word).lower()
if not w:
return "a"
# simple orthographic rule (good enough for lecturer, hour, university, …)
if w[0] in "aeiou":
return "an"
return "a"
###############################################################################
###############################################################################
def make_indefinite_np(np_text: str) -> str:
"""
the famous lecturer → a famous lecturer
an old friend → an old friend (unchanged)
famous lecturer → a famous lecturer (if bare, add article)
"""
np_text = re.sub(r"\s+", " ", (np_text or "")).strip()
if not np_text:
return np_text
m = re.match(r"^(the|a|an)\s+(.*)$", np_text, flags=re.I)
if m:
rest = m.group(2).strip()
first = rest.split()[0] if rest else ""
return f"{indefinite_article_for(first)} {rest}".strip()
first = np_text.split()[0]
return f"{indefinite_article_for(first)} {np_text}".strip()
###############################################################################
###############################################################################
YELLOW_CCV = {"IN_COGNITIVE_COMMUNICATIVE_VP"}
GREEN_CCV = {
"IN_COGNITIVE_COMMUNICATIVE_VP_BEFORESIGN",
"IN_COGNITIVE_COMMUNICATIVE_VP_AFTERSIGN",
}
HIGHLIGHT_CCV = YELLOW_CCV | GREEN_CCV
CCV_COLOURS = {
"IN_COGNITIVE_COMMUNICATIVE_VP": "#FFEB3B",
"IN_COGNITIVE_COMMUNICATIVE_VP_BEFORESIGN": "#81C784",
"IN_COGNITIVE_COMMUNICATIVE_VP_AFTERSIGN": "#81C784",
}
###############################################################################
###############################################################################
def _has_ssccv_sign(tagged_list):
return any("M:Y_SSCCV" in str(item) for item in tagged_list)
###############################################################################
###############################################################################
def _glue_tokens(toks):
"""Join tokens without spaces around ' or punctuation."""
buf = []
for tok in toks:
if not tok:
continue
if not buf:
buf.append(tok)
continue
if tok in ".,!?;:" or tok == "'" or tok.startswith("'"):
buf[-1] = buf[-1] + tok
elif buf[-1].endswith("'"):
buf[-1] = buf[-1] + tok
else:
buf.append(tok)
return " ".join(buf)
###############################################################################
###############################################################################
def highlight_tokens_html(tokens, labels):
parts = []
i = 0
n = len(tokens)
while i < n:
lab = labels[i] if i < len(labels) else ""
if lab in HIGHLIGHT_CCV:
j = i + 1
while j < n and (labels[j] if j < len(labels) else "") == lab:
j += 1
chunk = _glue_tokens(tokens[i:j])
esc = html.escape(chunk)
if lab == "IN_COGNITIVE_COMMUNICATIVE_VP":
parts.append(f"<strong><u>{esc}</u></strong>")
else:
parts.append(f"<u>{esc}</u>")
i = j
else:
j = i + 1
while j < n:
lab_j = labels[j] if j < len(labels) else ""
if lab_j in HIGHLIGHT_CCV:
break
j += 1
chunk = _glue_tokens(tokens[i:j])
parts.append(html.escape(chunk))
i = j
return " ".join(parts)
###############################################################################
###############################################################################
def highlight_sentence_if_ssccv(sentence):
"""
SignTagger -> bracket one SSCCV sign -> SSCCVspanTagger
-> HTML with cognitive/communicative VP underlined.
"""
sentence = (sentence or "").strip()
if not sentence:
return ""
tagged_list = tag_all_signs_in_sent(sentence)
if not _has_ssccv_sign(tagged_list):
return html.escape(sentence)
bracketed = build_single_sign_bracketed(tagged_list, mode="ssccv")
if not bracketed:
return html.escape(sentence)
tokens, pos, lemmas, preds = tag_SSCCV_span(bracketed)
return highlight_tokens_html(tokens, preds)
###############################################################################
###############################################################################
def highlight_tokens_html(tokens, labels):
parts = []
i = 0
n = len(tokens)
while i < n:
lab = labels[i] if i < len(labels) else ""
if lab in HIGHLIGHT_CCV:
j = i + 1
while j < n and (labels[j] if j < len(labels) else "") == lab:
j += 1
chunk = re.sub(r"\s+([.,!?;:])", r"\1", " ".join(tokens[i:j]))
esc = html.escape(chunk)
if lab == "IN_COGNITIVE_COMMUNICATIVE_VP":
parts.append(f"<strong><u>{esc}</u></strong>")
else:
parts.append(f"<u>{esc}</u>")
i = j
else:
tok = tokens[i]
if tok in ".,!?;:" and parts:
parts[-1] = parts[-1] + html.escape(tok)
else:
parts.append(html.escape(tok))
i += 1
return " ".join(parts)
###############################################################################
###############################################################################
def highlight_final_output(final_text):
if not final_text or str(final_text).startswith("ERROR"):
return f'<div class="simp-out">{html.escape(final_text or "")}</div>'
para_html = []
for para in re.split(r"\n\s*\n", final_text):
para = para.strip()
if not para:
continue
sents = split_into_sentences(para) if para else [para]
bits = [highlight_sentence_if_ssccv(s) for s in sents]
para_html.append("<p>" + " ".join(bits) + "</p>")
return '<div class="simp-out">' + "".join(para_html) + "</div>"
###############################################################################
###############################################################################
def tag_SSCCV_span(sentence):
if re.search(r"\[\:\]\s*$", sentence):
sentence = sentence + " blah blah blah."
# --- BERT side ---
encoded = SSCCVspanTaggingTokenizer(
sentence,
return_tensors="pt",
truncation=True,
max_length=512,
)
# Same device as the model (Windows CPU or HF GPU/ZeroGPU)
model_device = next(SSCCVspanTaggingModel.parameters()).device
encoded = {k: v.to(model_device) for k, v in encoded.items()}
with torch.no_grad():
outputs = SSCCVspanTaggingModel(**encoded)[0]
predictions = torch.argmax(outputs, dim=2)
tokens = SSCCVspanTaggingTokenizer.tokenize(
SSCCVspanTaggingTokenizer.decode(
SSCCVspanTaggingTokenizer.encode(sentence)
)
)
SSCCV_span_tagged_predictions = [
SSCCVspan_label_list[p] for p in predictions[0].tolist()
]
# --- spaCy side (CPU) ---
aligned_tokens = []
aligned_predictions = []
aligned_pos_tags = []
aligned_lemmas = []
sc_sentence = (
clean_for_spaCy(sentence)
if "[" in sentence and "]" in sentence
else sentence
)
doc = nlp(sc_sentence)
spacy_pos_tags = [t.tag_ for t in doc]
spacy_tokens = [t.text for t in doc]
spacy_lemmas = [t.lemma_ for t in doc]
print("Checking proper nouns after building doc")
for t in doc:
if "NNP" in t.tag_:
print(t.tag_, t.text)
a2b, b2a = tokenizations.get_alignments(spacy_tokens, tokens)
for i in range(len(spacy_tokens)):
for j in a2b[i]:
if j < len(SSCCV_span_tagged_predictions):
aligned_tokens.append(spacy_tokens[i])
aligned_pos_tags.append(spacy_pos_tags[i])
aligned_lemmas.append(spacy_lemmas[i])
aligned_predictions.append(SSCCV_span_tagged_predictions[j])
break
clean_tokens, clean_pos_tags, clean_lemmas, clean_predictions = sign_cleaning(
aligned_tokens, aligned_pos_tags, aligned_lemmas, aligned_predictions
)
simplifiable = all("COMPLEX_NP" in p for p in clean_predictions)
if not simplifiable:
True # placeholder for future logic
return (
clean_tokens,
clean_pos_tags,
clean_lemmas,
clean_predictions,
)
###############################################################################
###############################################################################
def tag_SSMN_span(sentence):
"""
sentence must contain the left appositive boundary in brackets, e.g.:
John[,] the famous teacher, went home.
"""
if re.search(r"\[\:\]\s*$", sentence):
sentence = sentence + " blah blah blah."
# --- BERT side ---
# Prefer the full call so you get input_ids + attention_mask
encoded = SSMNspanTaggingTokenizer(
sentence,
return_tensors="pt",
truncation=True,
max_length=512,
)
# Put every tensor on the *same* device as the model
model_device = next(SSMNspanTaggingModel.parameters()).device
encoded = {k: v.to(model_device) for k, v in encoded.items()}
with torch.no_grad():
outputs = SSMNspanTaggingModel(**encoded)[0]
predictions = torch.argmax(outputs, dim=2)
# Wordpiece tokens for alignment (same string as above)
tokens = SSMNspanTaggingTokenizer.tokenize(
SSMNspanTaggingTokenizer.decode(
SSMNspanTaggingTokenizer.encode(sentence)
)
)
ssmn_span_tagged_predictions = [
SSMNspan_label_list[p] for p in predictions[0].tolist()
]
# --- spaCy side (CPU is fine) ---
aligned_tokens, aligned_predictions = [], []
aligned_pos_tags, aligned_lemmas = [], []
sc_sentence = (
clean_for_spaCy(sentence)
if "[" in sentence and "]" in sentence
else sentence
)
doc = nlp(sc_sentence)
spacy_pos_tags = [t.tag_ for t in doc]
spacy_tokens = [t.text for t in doc]
spacy_lemmas = [t.lemma_ for t in doc]
a2b, b2a = tokenizations.get_alignments(spacy_tokens, tokens)
for i in range(len(spacy_tokens)):
for j in a2b[i]:
if j < len(ssmn_span_tagged_predictions):
aligned_tokens.append(spacy_tokens[i])
aligned_pos_tags.append(spacy_pos_tags[i])
aligned_lemmas.append(spacy_lemmas[i])
aligned_predictions.append(ssmn_span_tagged_predictions[j])
break
return sign_cleaning(
aligned_tokens, aligned_pos_tags, aligned_lemmas, aligned_predictions
)
###############################################################################
###############################################################################
def tag_CCV_span(sentence):
if re.search(r"\[\:\]\s*$", sentence):
sentence = sentence + " blah blah blah."
# --- BERT side ---
encoded = CCVspanTaggingTokenizer(
sentence,
return_tensors="pt",
truncation=True,
max_length=512,
)
# Same device as the model (Windows CPU or HF GPU/ZeroGPU)
model_device = next(CCVspanTaggingModel.parameters()).device
encoded = {k: v.to(model_device) for k, v in encoded.items()}
with torch.no_grad():
outputs = CCVspanTaggingModel(**encoded)[0]
predictions = torch.argmax(outputs, dim=2)
tokens = CCVspanTaggingTokenizer.tokenize(
CCVspanTaggingTokenizer.decode(
CCVspanTaggingTokenizer.encode(sentence)
)
)
CCV_span_tagged_predictions = [
CCVspan_label_list[p] for p in predictions[0].tolist()
]
# --- spaCy side (CPU) ---
aligned_tokens = []
aligned_predictions = []
aligned_pos_tags = []
aligned_lemmas = []
sc_sentence = (
clean_for_spaCy(sentence)
if "[" in sentence and "]" in sentence
else sentence
)
doc = nlp(sc_sentence)
spacy_pos_tags = [t.tag_ for t in doc]
spacy_tokens = [t.text for t in doc]
spacy_lemmas = [t.lemma_ for t in doc]
a2b, b2a = tokenizations.get_alignments(spacy_tokens, tokens)
for i in range(len(spacy_tokens)):
for j in a2b[i]:
if j < len(CCV_span_tagged_predictions):
aligned_tokens.append(spacy_tokens[i])
aligned_pos_tags.append(spacy_pos_tags[i])
aligned_lemmas.append(spacy_lemmas[i])
aligned_predictions.append(CCV_span_tagged_predictions[j])
break
clean_tokens, clean_pos_tags, clean_lemmas, clean_predictions = sign_cleaning(
aligned_tokens, aligned_pos_tags, aligned_lemmas, aligned_predictions
)
simplifiable = all("COMPOUND" in p for p in clean_predictions)
if not simplifiable:
True # placeholder for future logic
return (
clean_tokens,
clean_pos_tags,
clean_lemmas,
clean_predictions,
)
###############################################################################
###############################################################################
def tag_CMV1_span(sentence):
if re.search(r"\[\:\]\s*$", sentence):
sentence = sentence + " blah blah blah."
# --- BERT side ---
encoded = CMV1spanTaggingTokenizer(
sentence,
return_tensors="pt",
truncation=True,
max_length=512,
)
# Same device as the model (Windows CPU or HF GPU/ZeroGPU)
model_device = next(CMV1spanTaggingModel.parameters()).device
encoded = {k: v.to(model_device) for k, v in encoded.items()}
with torch.no_grad():
outputs = CMV1spanTaggingModel(**encoded)[0]
predictions = torch.argmax(outputs, dim=2)
tokens = CMV1spanTaggingTokenizer.tokenize(
CMV1spanTaggingTokenizer.decode(
CMV1spanTaggingTokenizer.encode(sentence)
)
)
CMV1_span_tagged_predictions = [
CMV1span_label_list[p] for p in predictions[0].tolist()
]
# --- spaCy side (CPU) ---
aligned_tokens = []
aligned_predictions = []
aligned_pos_tags = []
aligned_lemmas = []
sc_sentence = (
clean_for_spaCy(sentence)
if "[" in sentence and "]" in sentence
else sentence
)
doc = nlp(sc_sentence)
spacy_pos_tags = [t.tag_ for t in doc]
spacy_tokens = [t.text for t in doc]
spacy_lemmas = [t.lemma_ for t in doc]
a2b, b2a = tokenizations.get_alignments(spacy_tokens, tokens)
for i in range(len(spacy_tokens)):
for j in a2b[i]:
if j < len(CMV1_span_tagged_predictions):
aligned_tokens.append(spacy_tokens[i])
aligned_pos_tags.append(spacy_pos_tags[i])
aligned_lemmas.append(spacy_lemmas[i])
aligned_predictions.append(CMV1_span_tagged_predictions[j])
break
clean_tokens, clean_pos_tags, clean_lemmas, clean_predictions = sign_cleaning(
aligned_tokens, aligned_pos_tags, aligned_lemmas, aligned_predictions
)
simplifiable = all("COMPOUND" in p for p in clean_predictions)
if not simplifiable:
True # placeholder for future logic
return (
clean_tokens,
clean_pos_tags,
clean_lemmas,
clean_predictions,
)
###############################################################################
###############################################################################
def tag_CMN1_span(sentence):
if re.search(r"\[\:\]\s*$", sentence):
sentence = sentence + " blah blah blah."
# --- BERT side ---
encoded = CMN1spanTaggingTokenizer(
sentence,
return_tensors="pt",
truncation=True,
max_length=512,
)
# Same device as the model (Windows CPU or HF GPU/ZeroGPU)
model_device = next(CMN1spanTaggingModel.parameters()).device
encoded = {k: v.to(model_device) for k, v in encoded.items()}
with torch.no_grad():
outputs = CMN1spanTaggingModel(**encoded)[0]
predictions = torch.argmax(outputs, dim=2)
tokens = CMN1spanTaggingTokenizer.tokenize(
CMN1spanTaggingTokenizer.decode(
CMN1spanTaggingTokenizer.encode(sentence)
)
)
CMN1_span_tagged_predictions = [
CMN1span_label_list[p] for p in predictions[0].tolist()
]
# --- spaCy side (CPU) ---
aligned_tokens = []
aligned_predictions = []
aligned_pos_tags = []
aligned_lemmas = []
sc_sentence = (
clean_for_spaCy(sentence)
if "[" in sentence and "]" in sentence
else sentence
)
doc = nlp(sc_sentence)
spacy_pos_tags = [t.tag_ for t in doc]
spacy_tokens = [t.text for t in doc]
spacy_lemmas = [t.lemma_ for t in doc]
a2b, b2a = tokenizations.get_alignments(spacy_tokens, tokens)
for i in range(len(spacy_tokens)):
for j in a2b[i]:
if j < len(CMN1_span_tagged_predictions):
aligned_tokens.append(spacy_tokens[i])
aligned_pos_tags.append(spacy_pos_tags[i])
aligned_lemmas.append(spacy_lemmas[i])
aligned_predictions.append(CMN1_span_tagged_predictions[j])
break
clean_tokens, clean_pos_tags, clean_lemmas, clean_predictions = sign_cleaning(
aligned_tokens, aligned_pos_tags, aligned_lemmas, aligned_predictions
)
simplifiable = all("COMPOUND" in p for p in clean_predictions)
if not simplifiable:
True # placeholder for future logic
return (
clean_tokens,
clean_pos_tags,
clean_lemmas,
clean_predictions,
)
###############################################################################
###############################################################################
def tag_CLN_span(sentence):
if re.search(r"\[\:\]\s*$", sentence):
sentence = sentence + " blah blah blah."
# --- BERT side ---
encoded = CLNspanTaggingTokenizer(
sentence,
return_tensors="pt",
truncation=True,
max_length=512,
)
# Same device as the model (works on Windows CPU and HF GPU/ZeroGPU)
model_device = next(CLNspanTaggingModel.parameters()).device
encoded = {k: v.to(model_device) for k, v in encoded.items()}
with torch.no_grad():
outputs = CLNspanTaggingModel(**encoded)[0]
predictions = torch.argmax(outputs, dim=2)
tokens = CLNspanTaggingTokenizer.tokenize(
CLNspanTaggingTokenizer.decode(
CLNspanTaggingTokenizer.encode(sentence)
)
)
CLN_span_tagged_predictions = [
CLNspan_label_list[p] for p in predictions[0].tolist()
]
# --- spaCy side (CPU) ---
aligned_tokens = []
aligned_predictions = []
aligned_pos_tags = []
aligned_lemmas = []
sc_sentence = (
clean_for_spaCy(sentence)
if "[" in sentence and "]" in sentence
else sentence
)
doc = nlp(sc_sentence)
spacy_pos_tags = [t.tag_ for t in doc]
spacy_tokens = [t.text for t in doc]
spacy_lemmas = [t.lemma_ for t in doc]
a2b, b2a = tokenizations.get_alignments(spacy_tokens, tokens)
for i in range(len(spacy_tokens)):
for j in a2b[i]:
if j < len(CLN_span_tagged_predictions):
aligned_tokens.append(spacy_tokens[i])
aligned_pos_tags.append(spacy_pos_tags[i])
aligned_lemmas.append(spacy_lemmas[i])
aligned_predictions.append(CLN_span_tagged_predictions[j])
break
clean_tokens, clean_pos_tags, clean_lemmas, clean_predictions = sign_cleaning(
aligned_tokens, aligned_pos_tags, aligned_lemmas, aligned_predictions
)
simplifiable = all("COMPOUND" in p for p in clean_predictions)
if not simplifiable:
True # placeholder for future logic
return (
clean_tokens,
clean_pos_tags,
clean_lemmas,
clean_predictions,
)
###############################################################################
###############################################################################
def merge_wordpieces(sign_tagged_tokens):
"""
Merge BERT WordPiece tokens back into whole words.
Example: [('ma', 'M:N_CCV'), ('##ry', 'M:N_CCV')] → [('mary', 'M:N_CCV')]
"""
merged = []
for tok, lab in sign_tagged_tokens:
if tok in ('[CLS]', '[SEP]'):
continue
if tok.startswith('##') and merged:
prev_tok, prev_lab = merged[-1]
merged[-1] = (prev_tok + tok[2:], prev_lab)
else:
# also strip a lone '#' that sometimes appears after bad cleaning
if tok in ('#', '##'):
continue
merged.append((tok, lab))
return merged
###############################################################################
###############################################################################
def tag_all_signs_in_sent(sent):
"""Returns a LIST of strings with <M:Y_...> tags inserted."""
sent = re.sub("do not", "don't", sent)
# Remove any existing square brackets around signs
sent = re.sub(
r"\[((,|;|:|and|but|or|that|what|when|where|which|while|who)"
r"(\_(and|but|or|that|what|when|where|which|while|who))?)\]",
r"\1\2",
sent,
flags=re.IGNORECASE,
)
# --- BERT side ---
encoded = SignTaggingTokenizer(
sent,
return_tensors="pt",
truncation=True,
max_length=512,
)
# Same device as the model (Windows CPU or HF GPU/ZeroGPU)
model_device = next(SignTaggingModel.parameters()).device
encoded = {k: v.to(model_device) for k, v in encoded.items()}
with torch.no_grad():
outputs = SignTaggingModel(**encoded)[0]
predictions = torch.argmax(outputs, dim=2)
tokens = SignTaggingTokenizer.tokenize(
SignTaggingTokenizer.decode(SignTaggingTokenizer.encode(sent))
)
sign_tagged_tokens = [
(token, sign_label_list[prediction])
for token, prediction in zip(tokens, predictions[0].tolist())
]
# Merge sub-word pieces before any further processing
sign_tagged_tokens = merge_wordpieces(sign_tagged_tokens)
print("sign_tagged_tokens:", sign_tagged_tokens)
relevant_sign_offsets = []
relevant_tags = []
t = 0
while t < len(sign_tagged_tokens):
lab = sign_tagged_tokens[t][1]
if re.match("M:Y", lab):
relevant_sign_offsets.append(t)
relevant_tags.append(lab)
t += 1
tagged_sent_list = [tok for tok, _ in sign_tagged_tokens]
x = 0
while x < len(relevant_sign_offsets):
o = relevant_sign_offsets[x]
rtag = relevant_tags[x]
tagged_sent_list[o] = (
"<" + rtag + ">" + tagged_sent_list[o] + "</" + rtag + ">"
)
x += 1
# Build the final list (excluding [CLS] and [SEP])
tagged_sent = []
x = 0
while x < len(tagged_sent_list):
if sign_tagged_tokens[x][0] not in ["[CLS]", "[SEP]"]:
tagged_sent.append(tagged_sent_list[x])
x += 1
print(f"Returning tagged_sent: result = {tagged_sent}")
return tagged_sent # LIST, not a string
###############################################################################
###############################################################################
def split_coordinated_relative_body(body):
"""
Split 'Steven kicked and who Mary liked'
-> ['Steven kicked', 'Mary liked']
Also handles 'and which', bare 'and' between relative-like conjuncts.
"""
body = body.strip()
# Prefer explicit repeated relative marker
parts = re.split(
r"\s+and\s+(?:who|which|whom|that)\s+",
body,
flags=re.IGNORECASE,
)
if len(parts) >= 2:
return [p.strip() for p in parts if p.strip()]
return [body]
###############################################################################
###############################################################################
def _region_text(elem):
if elem is None:
return ""
parts = []
for w in elem.findall("W"):
if w.text and str(w.text).strip():
parts.append(str(w.text).strip())
return " ".join(parts).strip()
###############################################################################
###############################################################################
def _split_coord_relative_body(body):
"""
'Steven kicked and who Mary liked'
-> ['Steven kicked', 'Mary liked']
"""
body = re.sub(r"\s+", " ", (body or "")).strip()
if not body:
return []
parts = re.split(
r"\s+and\s+(?:who|which|whom|that)\s+",
body,
flags=re.IGNORECASE,
)
parts = [p.strip() for p in parts if p.strip()]
return parts if parts else [body]
###############################################################################
###############################################################################
def _tidy_plain(s):
s = re.sub(r"\s+", " ", (s or "")).strip()
s = re.sub(r"\s+([.,!?;:])", r"\1", s)
s = re.sub(r",+\s*\.", ".", s)
s = re.sub(r",+\s*$", "", s)
if s and not s.endswith((".", "!", "?")):
s += "."
if s:
s = s[0].upper() + s[1:]
return s
###############################################################################
# Placeholder for simplification functions (core of the 1501 lines; implement from original)
###############################################################################
def SSCCVsimplify(xml_sent):
simplification = []
xml_sent2 = xml_sent
xml_string = ET.tostring(xml_sent).decode("utf-8")
xml_for_printing = xml.dom.minidom.parseString(xml_string)
pretty_xml_as_string = xml_for_printing.toprettyxml()
# print(pretty_xml_as_string)
simp_sent1 = ET.Element("S")
simp_sent2 = ET.Element("S")
if xml_sent.attrib["TYPE"] == "COMPLEX_NP":
complex_NP_type = get_COMPLEX_NP_type(xml_sent)
print("SSCCV complex_NP_type:", complex_NP_type)
# print('SSCCV SIMPLIFYING:', complex_NP_type + '\nINPUT SENTENCE:\n'+ pretty_xml_as_string, file=sys.stderr)
#######################################################################
#######################################################################
#######################################################################
if complex_NP_type == "SUBJECT_RELATIVISED":
# Subject relative, e.g.:
# John, who lives in London, went home.
# John, who lives in London and who works in Paris, went home.
#
# Matrix: [BEFORE] + head + [AFTER]
# Relative: head + each conjunct (predication)
#
# Coordinated body "lives in London and who works in Paris"
# -> ["lives in London", "works in Paris"]
# -> "John lives in London." / "John works in Paris."
before_el = xml_sent.find("BEFORE_COMPLEX_CONSTITUENT")
head_el = xml_sent.find("IN_COMPLEX_CONSTITUENT_BEFORESIGN")
body_el = xml_sent.find("IN_COMPLEX_CONSTITUENT_AFTERSIGN")
after_el = xml_sent.find("AFTER_COMPLEX_CONSTITUENT")
head = re.sub(r"[,\s]+$", "", _region_text(head_el)).strip()
body = re.sub(r"^[,\s]+", "", _region_text(body_el)).strip()
after = re.sub(r"^[,\s]+", "", _region_text(after_el)).strip()
before = _region_text(before_el)
matrix = _tidy_plain(" ".join(x for x in [before, head, after] if x))
conjuncts = _split_coord_relative_body(body)
out = []
if matrix:
out.append(matrix)
for conj in conjuncts:
if conj and head:
out.append(_tidy_plain(f"{head} {conj}"))
return out
#######################################################################
#######################################################################
#######################################################################
# '''
# [John saw] [the bucket] [into] [which] [the coin had fallen][].
# [John saw] [the bucket][].
# [The coin had fallen] [into] [the bucket]
#
# '''
elif complex_NP_type == "PREPOSITION_OBJECT_RELATIVISED":
copula = ""
xml_prep = ET.Element("W")
for vor_sent_el in xml_sent:
if vor_sent_el.tag == "IN_COMPLEX_CONSTITUENT_BEFORESIGN":
for vor_sent_subel in vor_sent_el:
if vor_sent_subel.attrib["POS"] == "IN":
xml_prep = vor_sent_subel
vor_sent_matrix_subject = ET.Element("NA")
vor_sent_el_bcc = ET.Element("NA")
vor_sent_prep = ET.Element("NA")
vor_sent_acc = ET.Element("NA")
vor_sent2_prep = ET.Element("NA")
vor_sent2_matrix_subject = ET.Element("NA")
vor_sent2_clause = ET.Element("NA")
for vor_sent_el in xml_sent:
if vor_sent_el.tag == "IN_COMPLEX_CONSTITUENT":
try:
if re.match("(\,|\:|\;)$", vor_sent_el[-2].text):
punctuation_boundary = True
except:
True
# DELETING STRAY COMMA IMMEDIATELY PRECEDING THE SIGN DETECTED BY THE BERT MODEL
elif vor_sent_el.tag == "IN_COMPLEX_CONSTITUENT_BEFORESIGN":
if re.match("(\,|\:|\;)$", vor_sent_el[-1].text):
del vor_sent_el[-1]
# DELETING STRAY COMMA IMMEDIATELY FOLLOWING THE SIGN DETECTED BY THE BERT MODEL
elif vor_sent_el.tag == "AFTER_COMPLEX_CONSTITUENT":
if re.match("(\,|\:|\;)$", vor_sent_el[0].text):
del vor_sent_el[0]
for vor_sent_el in xml_sent:
if vor_sent_el.tag == "BEFORE_COMPLEX_CONSTITUENT":
vor_sent_el_bcc = vor_sent_el
elif vor_sent_el.tag == "IN_COMPLEX_CONSTITUENT_BEFORESIGN":
vor_sent_prep = xml_prep
vor_sent2_prep = xml_prep
vor_sent_matrix_subject = vor_sent_el
vor_sent_matrix_subject.remove(vor_sent_matrix_subject[-1])
elif vor_sent_el.tag == "AFTER_COMPLEX_CONSTITUENT":
if re.match("(\,|\:|\;)$", vor_sent_el[0].text):
del vor_sent_el[0]
vor_sent_acc = vor_sent_el
simp_sent1.append(vor_sent_el_bcc)
simp_sent1.append(vor_sent_matrix_subject)
simp_sent1.append(vor_sent_acc)
for vor_sent2_el in xml_sent2:
if vor_sent2_el.tag == "IN_COMPLEX_CONSTITUENT_BEFORESIGN":
try:
if vor_sent2_el[-1].attrib["POS"] == "IN":
vor_sent2_prep = vor_sent2_el[-1]
vor_sent2_matrix_subject = vor_sent2_el
except:
True
elif vor_sent2_el.tag == "IN_COMPLEX_CONSTITUENT_AFTERSIGN":
vor_sent2_clause = vor_sent2_el
simp_sent2.append(vor_sent2_clause)
simp_sent2.append(vor_sent2_prep)
simp_sent2.append(vor_sent2_matrix_subject)
#######################################################################
#######################################################################
#######################################################################
# '''
# [John saw] [the bucket] [which] [Peter despised][].
# [John saw] [the bucket][].
# [Peter despised] [the bucket].
#
# '''
elif complex_NP_type == "VERB_OBJECT_RELATIVISED":
# Object relative, e.g.:
# John, who Steven kicked, went home.
# John, who Steven kicked and who Mary liked, went home.
#
# Matrix: head + AFTER
# Relative(s): each conjunct + head as object
before_el = xml_sent.find("BEFORE_COMPLEX_CONSTITUENT")
head_el = xml_sent.find("IN_COMPLEX_CONSTITUENT_BEFORESIGN")
body_el = xml_sent.find("IN_COMPLEX_CONSTITUENT_AFTERSIGN")
after_el = xml_sent.find("AFTER_COMPLEX_CONSTITUENT")
head = re.sub(r"[,\s]+$", "", _region_text(head_el)).strip()
body = re.sub(r"^[,\s]+", "", _region_text(body_el)).strip()
after = re.sub(r"^[,\s]+", "", _region_text(after_el)).strip()
before = _region_text(before_el)
matrix = _tidy_plain(" ".join(x for x in [before, head, after] if x))
conjuncts = _split_coord_relative_body(body)
out = []
if matrix:
out.append(matrix)
for conj in conjuncts:
if conj and head:
out.append(_tidy_plain(f"{conj} {head}"))
return out # leave function here; do not use simp_sent1/simp_sent2
if len(simp_sent1) > 0 and len(simp_sent2) > 0:
s1_xml_string = ET.tostring(simp_sent1).decode("utf-8")
s1_xml_for_printing = xml.dom.minidom.parseString(s1_xml_string)
s1_pretty_xml_as_string = s1_xml_for_printing.toprettyxml()
s2_xml_string = ET.tostring(simp_sent2).decode("utf-8")
s2_xml_for_printing = xml.dom.minidom.parseString(s2_xml_string)
s2_pretty_xml_as_string = s2_xml_for_printing.toprettyxml()
"""
Be aware that simp_sent1 and simp_sent2 have a different structure
from xml_sent. This affects the xml_to_text function/process.
"""
try:
simplification.append(simplified_sent_xml_to_text(simp_sent1))
simplification.append(simplified_sent_xml_to_text(simp_sent2))
except Exception as e:
print("SSCCVsimplify xml_to_text error:", e)
print("simp_sent1:", ET.tostring(simp_sent1).decode("utf-8"))
print("simp_sent2:", ET.tostring(simp_sent2).decode("utf-8"))
# print('EXITING SSCCVsimplify:', simplification, file=sys.stderr)
return simplification
###############################################################################
###############################################################################
def looks_like_proper_name_np(np_text: str) -> bool:
toks = np_text.split()
return bool(toks) and all(t[:1].isupper() for t in toks if t.isalpha())
###############################################################################
###############################################################################
def SSMNsimplify(xml_sent):
"""
Appositive / non-finite nominal modifier (SSMN):
John Smith, the famous lecturer, went home.
→ John Smith went home.
→ John Smith is a famous lecturer.
"""
simplification = []
try:
if xml_sent.attrib.get("TYPE") not in ("COMPLEX_NP", "APPOSITIVE", "SSMN"):
# still try if elements exist
pass
def elem_text(elem):
if elem is None:
return ""
parts = []
for w in elem.findall("W"):
if not w.text:
continue
t = str(w.text).strip()
t = re.sub(r"[\[\]]+", "", t)
if t:
parts.append(t)
return " ".join(parts).strip()
def strip_edge_commas(s: str) -> str:
s = re.sub(r"\s+", " ", (s or "")).strip()
s = re.sub(r"^[,;:\s]+", "", s)
s = re.sub(r"[,;:\s]+$", "", s)
return s.strip()
def tidy_sent(s: str) -> str:
s = re.sub(r"\s+", " ", (s or "")).strip()
s = re.sub(r"\s+([.,!?;:])", r"\1", s)
s = re.sub(r"([.,!?;:])\s+", r"\1 ", s)
s = re.sub(r",+\s*\.", ".", s)
if s and not s.endswith((".", "!", "?")):
s += "."
if s:
s = s[0].upper() + s[1:]
return s
before = elem_text(xml_sent.find(".//BEFORE_COMPLEX_CONSTITUENT"))
head = elem_text(xml_sent.find(".//IN_COMPLEX_CONSTITUENT_BEFORESIGN"))
# trigger comma lives in IN_COMPLEX_CONSTITUENT — discarded
appositive = elem_text(xml_sent.find(".//IN_COMPLEX_CONSTITUENT_AFTERSIGN"))
after = elem_text(xml_sent.find(".//AFTER_COMPLEX_CONSTITUENT"))
head = strip_edge_commas(head)
appositive = strip_edge_commas(appositive)
before = strip_edge_commas(before)
after = strip_edge_commas(after)
if not head:
return []
# 1) Matrix clause without appositive
matrix = " ".join(x for x in [before, head, after] if x).strip()
matrix = tidy_sent(matrix)
# 2) Equative: Head is a/an …
if looks_like_proper_name_np(appositive):
pred_np = appositive
else:
pred_np = make_indefinite_np(appositive)
equative = tidy_sent(f"{head} is {pred_np}")
if matrix:
simplification.append(matrix)
if equative and equative != matrix:
simplification.append(equative)
print(f"SSMNsimplify produced {len(simplification)} sentences: {simplification}")
return simplification
except Exception as e:
print(f"SSMNsimplify error: {e}")
return []
###############################################################################
###############################################################################
def extract_coordinator_from_xml(xml_sent):
"""Surface of IN_COMPOUND_CONSTITUENT (and / or / but / ; / , and …)."""
elem = xml_sent.find(".//IN_COMPOUND_CONSTITUENT")
if elem is None:
return None
parts = []
for w in elem.findall("W"):
if w.text and str(w.text).strip():
parts.append(str(w.text).strip())
if not parts:
return None
raw = " ".join(parts).strip().lower()
raw = re.sub(r"\s+", " ", raw)
# ", and" / "; and" → and
for word in ("and", "or", "but"):
if re.search(rf"\b{word}\b", raw):
return word
if raw in {";", ","} or raw.endswith(";") or raw.endswith(","):
return "and" # list punctuation → treat as "and" for residuals
return None
###############################################################################
###############################################################################
def rewrite_residual_list_comma(sent, coordinator="and"):
"""
Last list-separator comma → same conjunction as the split sign,
so the next pass sees CMN1/CMV1/CCV more reliably (not SSMN).
Skips comma before a bare number (e.g. Smith, 43).
"""
if not sent or not coordinator:
return sent
coord = coordinator.strip().lower()
if coord not in {"and", "or", "but"}:
return sent
s = sent.strip()
m = re.search(r",\s+(?!\d\b)([^,]+?)(\s*[.!?])?\s*$", s)
if not m:
return s
head = s[: m.start()].rstrip()
tail = m.group(1).strip()
punct = (m.group(2) or "").strip() or "."
out = f"{head} {coord} {tail}{punct}"
out = re.sub(r"\s+", " ", out).strip()
return out
###############################################################################
###############################################################################
def repair_mangled_infinitive_be(sent: str) -> str:
"""Undo earlier bugs: to bes / to beses → to be."""
if not sent:
return sent
sent = re.sub(r"\bto\s+bes(?:es)?\b", "to be", sent, flags=re.I)
sent = re.sub(r"\bto\s+bees\b", "to be", sent, flags=re.I)
return sent
###############################################################################
###############################################################################
def _singular_present_form(lemma: str) -> str:
"""Map verb lemma → 3rd person singular present."""
lemma = (lemma or "").lower()
irregular = {
"be": "is",
"have": "has",
"do": "does",
"go": "goes",
"say": "says",
}
if lemma in irregular:
return irregular[lemma]
if lemma.endswith(("s", "sh", "ch", "x", "z", "o")):
return lemma + "es"
if len(lemma) > 1 and lemma.endswith("y") and lemma[-2] not in "aeiou":
return lemma[:-1] + "ies"
return lemma + "s"
###############################################################################
def subject_is_singular(subj_token) -> bool:
"""True if the nominal subject should take singular agreement."""
# Coordinated subjects → plural
if any(c.dep_ == "cc" for c in subj_token.subtree):
return False
# Explicit plural morph / tags
num = subj_token.morph.get("Number")
if num == ["Plur"]:
return False
if num == ["Sing"]:
return True
if subj_token.tag_ in ("NNS", "NNPS", "PRP") and subj_token.lower_ in {
"we", "they", "you", "i"
}:
# I/you are special; treat as not 3sg
return False
if subj_token.tag_ in ("NNS", "NNPS"):
return False
if subj_token.lower_ in {"i", "you", "we", "they"}:
return False
# he/she/it / singular NN/NNP / this/that
if subj_token.lower_ in {"he", "she", "it", "this", "that"}:
return True
if subj_token.tag_ in ("NN", "NNP", "DT", "PRP$"):
return True
# fallback: whole subject span text
text = " ".join(t.text for t in subj_token.subtree).lower()
if " and " in text:
return False
return True
###############################################################################
MODALS = {
"can", "could", "may", "might", "must",
"shall", "should", "will", "would",
}
###############################################################################
MODAL_MID = {
"not", "n't", "even", "never", "also", "really",
"just", "only", "still", "already", "always",
"actually", "probably", "possibly", "perhaps",
"certainly", "definitely", "simply", "almost",
}
###############################################################################
# do-support / modal on the raw string to the left of the verb
_LEFT_BARE_INF = re.compile(
r"(?:"
r"\b(?:can|could|may|might|must|shall|should|will|would)\b|"
r"\b(?:do|does|did)\b|"
r"\b(?:don't|doesn't|didn't|don’t|doesn’t|didn’t)\b|"
r"\b(?:do|does|did)\s+not\b|"
r"\b(?:don|doesn|didn)\s*['’]\s*t\b"
r")"
r"(?:\s+\b(?:not|even|never|also|really|just|only|still|"
r"already|always|actually|probably|possibly|perhaps|"
r"certainly|definitely|simply|almost)\b)*"
r"\s+$",
flags=re.I,
)
###############################################################################
###############################################################################
def fix_subject_verb_agreement(sent: str) -> str:
"""
After a split, repair a finite verb / aux that no longer matches
a singular subject.
Never inflect a bare infinitive after a modal or do-support
(may have, didn't want, didn ' t want).
"""
if not sent or not sent.strip():
return sent
if re.search(r"\bdo(?:es|ing)?\s+so\b", sent, flags=re.I):
return sent
sent = sent.strip()
doc = nlp(sent)
if not doc:
return sent
subj = next((t for t in doc if t.dep_ in ("nsubj", "nsubjpass")), None)
if subj is None:
subj = next((t for t in doc if t.tag_ in ("NN", "NNP", "PRP")), None)
if subj is None:
return sent
if any(t.dep_ == "cc" for t in subj.subtree):
return sent
if subj.lower_ in {"i", "you", "we", "they"}:
return sent
if subj.tag_ in ("NNS", "NNPS"):
return sent
if subj.tag_ not in ("NN", "NNP", "PRP"):
return sent
def is_infinitive(tok) -> bool:
if tok.i > 0 and doc[tok.i - 1].lower_ == "to":
return True
if tok.head is not None and tok.head.lower_ == "to":
return True
if tok.dep_ in {"xcomp", "acl", "oprd"} and tok.tag_ in {"VB", "VBG"}:
return True
return False
def is_do_aux(t) -> bool:
low = re.sub(r"[^a-z]", "", t.lower_)
if t.lemma_ == "do":
return True
return low in {
"do", "does", "did",
"dont", "doesnt", "didnt",
"don", "doesn", "didn",
}
def is_modal(t) -> bool:
return t.tag_ == "MD" or t.lemma_.lower() in MODALS or t.lower_ in MODALS
def in_bare_infinitive_complement(tok) -> bool:
left = sent[: tok.idx]
if _LEFT_BARE_INF.search(left):
return True
i = tok.i - 1
while i >= 0:
t = doc[i]
if is_modal(t) or is_do_aux(t):
return True
frag = re.sub(r"[^a-z]", "", t.lower_)
if (
t.lower_ in MODAL_MID
or frag in {"", "t", "nt", "n"}
or t.dep_ in {"neg", "advmod", "punct"}
or t.pos_ in {"ADV", "PART", "PUNCT", "SPACE"}
or t.text in {"'", "’", "‘"}
):
i -= 1
continue
return False
return False
def skip_target(t) -> bool:
if t.tag_ == "MD":
return True
if is_infinitive(t):
return True
if in_bare_infinitive_complement(t):
return True
if t.lemma_ == "be" and t.lower_ in {"be", "bes", "beses"}:
return True
return False
def to_3sg(word: str) -> str:
low = word.lower()
irregular = {
"are": "is",
"am": "is",
"were": "was",
"have": "has",
"do": "does",
"go": "goes",
"live": "lives",
}
if low in irregular:
new = irregular[low]
elif low.endswith(("s", "sh", "ch", "x", "z", "o")):
new = low + "es"
elif low.endswith("y") and len(low) > 1 and low[-2] not in "aeiou":
new = low[:-1] + "ies"
else:
new = low + "s"
if word[:1].isupper():
new = new[0].upper() + new[1:]
return new
target = None
for t in doc:
if skip_target(t):
continue
if t.dep_ in ("aux", "auxpass") and t.lower_ in {
"are", "were", "have", "do", "am"
}:
target = t
break
if target is None:
for t in doc:
if skip_target(t):
continue
if t.tag_ in ("VBP", "VB") and t.lower_ not in {"to", "be"}:
target = t
break
out = sent
if target is not None and target.lower_ not in {"is", "was", "has", "does", "be"}:
new = to_3sg(target.text)
out = sent[: target.idx] + new + sent[target.idx + len(target.text) :]
# Safety net: if a 3sg form still sits after do-support / a modal,
# put it back to the lemma (wants→want, has→have, goes→go).
doc2 = nlp(out)
for t in reversed(list(doc2)):
if t.lemma_ == "do" or t.tag_ == "MD":
continue
left = out[: t.idx]
if not _LEFT_BARE_INF.search(left):
continue
if t.tag_ != "VBZ" and t.lower_ not in {
"wants", "goes", "has", "does", "is", "lives",
}:
continue
base = t.lemma_ if t.lemma_ and t.lemma_ != "-PRON-" else t.text
if t.lower_ == "has":
base = "have"
elif t.lower_ == "is":
base = "be"
elif t.lower_ == "does":
base = "do"
elif t.lower_ == "goes":
base = "go"
elif t.lower_.endswith("s") and t.lemma_:
base = t.lemma_
if base.lower() == t.lower_:
continue
if t.text[:1].isupper():
base = base[0].upper() + base[1:]
out = out[: t.idx] + base + out[t.idx + len(t.text) :]
return out
###############################################################################
###############################################################################
def CompoundSimplify(xml_sent):
"""
Reconstruct two sentences from the compound XML structure.
sent1 = BEFORE + BEFORESIGN + AFTER
sent2 = BEFORE + AFTERSIGN + AFTER
The coordinator (IN_COMPOUND_CONSTITUENT) is discarded.
Recovery: if AFTERSIGN is empty but IN_COMPOUND contains tokens
after the coordinator, treat those tokens as AFTERSIGN.
"""
try:
def get_text(elem):
if elem is None:
return ""
parts = []
for w in elem.findall("W"):
if w.text:
t = str(w.text).strip()
t = re.sub(r"[\[\]]+", "", t)
if t:
parts.append(t)
return " ".join(parts).strip()
def get_tokens(elem):
if elem is None:
return []
parts = []
for w in elem.findall("W"):
if w.text:
t = str(w.text).strip()
t = re.sub(r"[\[\]]+", "", t)
if t:
parts.append(t)
return parts
before_elem = xml_sent.find(".//BEFORE_COMPOUND_CONSTITUENT")
beforesign_elem = xml_sent.find(".//IN_COMPOUND_CONSTITUENT_BEFORESIGN")
sign_elem = xml_sent.find(".//IN_COMPOUND_CONSTITUENT")
aftersign_elem = xml_sent.find(".//IN_COMPOUND_CONSTITUENT_AFTERSIGN")
after_elem = xml_sent.find(".//AFTER_COMPOUND_CONSTITUENT")
before_text = get_text(before_elem)
beforesign_text = get_text(beforesign_elem)
aftersign_text = get_text(aftersign_elem)
after_text = get_text(after_elem)
# ----- recovery: second clause folded into IN_COMPOUND -----
if not aftersign_text and sign_elem is not None:
sign_toks = get_tokens(sign_elem)
coord = {",", ";", ":", "and", "but", "or", "nor"}
i = 0
while i < len(sign_toks) and sign_toks[i].lower() in coord:
i += 1
if i < len(sign_toks):
aftersign_text = " ".join(sign_toks[i:]).strip()
# -----------------------------------------------------------
sent1 = " ".join(
x for x in [before_text, beforesign_text, after_text] if x
).strip()
sent2 = " ".join(
x for x in [before_text, aftersign_text, after_text] if x
).strip()
def tidy(s):
if not s:
return ""
s = str(s)
s = re.sub(r"\s+", " ", s).strip()
s = re.sub(r"\s+([.,!?;:])", r"\1", s)
s = re.sub(r"([.,!?;:])\s+", r"\1 ", s)
# decimals: 8 . 2 → 8.2
s = re.sub(r"(\d)\s*\.\s*(\d)", r"\1.\2", s)
# list debris only at the end / before final period
s = re.sub(r"[,;:]+\s*\.", ".", s)
s = re.sub(r"[,;:]+\s*$", "", s)
s = s.strip()
if s and not s.endswith((".", "!", "?")):
s += "."
if s:
s = s[0].upper() + s[1:]
return s
sent1 = tidy(sent1)
sent2 = tidy(sent2)
result = []
if sent1:
result.append(sent1)
if sent2 and sent2 != sent1:
result.append(sent2)
print(f"CompoundSimplify produced {len(result)} sentences: {result}")
return result
except Exception as e:
print(f"CompoundSimplify error: {e}")
return []
###############################################################################
###############################################################################
def create_xml_sent(tokens, pos_tags, lemmas, predictions, sign_tag):
xml_sent = ET.Element("SENT")
print(
"tokens:",
tokens,
"\npos_tags:",
pos_tags,
"\nlemmas:",
lemmas,
"\npredictions:",
predictions,
"\nsign_tag:",
sign_tag,
)
before_complex_constituent = None
in_complex_constituent_beforesign = None
in_complex_constituent = None
in_complex_constituent_aftersign = None
after_complex_constituent = None
if sign_tag in "M:Y_SSCCV":
before_complex_constituent = ET.SubElement(
xml_sent, "BEFORE_COMPLEX_CONSTITUENT"
)
in_complex_constituent_beforesign = ET.SubElement(
xml_sent, "IN_COMPLEX_CONSTITUENT_BEFORESIGN"
)
in_complex_constituent = ET.SubElement(xml_sent, "IN_COMPLEX_CONSTITUENT")
in_complex_constituent_aftersign = ET.SubElement(
xml_sent, "IN_COMPLEX_CONSTITUENT_AFTERSIGN"
)
after_complex_constituent = ET.SubElement(xml_sent, "AFTER_COMPLEX_CONSTITUENT")
elif sign_tag in ["M:Y_SSMN"]: # SSMA later if you want ages
xml_sent.set("TYPE", "COMPLEX_NP") # or "APPOSITIVE"
before_complex_constituent = ET.SubElement(
xml_sent, "BEFORE_COMPLEX_CONSTITUENT"
)
in_complex_constituent_beforesign = ET.SubElement(
xml_sent, "IN_COMPLEX_CONSTITUENT_BEFORESIGN"
)
in_complex_constituent = ET.SubElement(
xml_sent, "IN_COMPLEX_CONSTITUENT"
)
in_complex_constituent_aftersign = ET.SubElement(
xml_sent, "IN_COMPLEX_CONSTITUENT_AFTERSIGN"
)
after_complex_constituent = ET.SubElement(
xml_sent, "AFTER_COMPLEX_CONSTITUENT"
)
elif sign_tag in ["M:Y_CCV", "M:Y_CMV1", "M:Y_CMN1", "M:Y_CLN"]:
xml_sent.set("TYPE", "COMPOUND")
before_complex_constituent = ET.SubElement(
xml_sent, "BEFORE_COMPOUND_CONSTITUENT"
)
in_complex_constituent_beforesign = ET.SubElement(
xml_sent, "IN_COMPOUND_CONSTITUENT_BEFORESIGN"
)
in_complex_constituent = ET.SubElement(xml_sent, "IN_COMPOUND_CONSTITUENT")
in_complex_constituent_aftersign = ET.SubElement(
xml_sent, "IN_COMPOUND_CONSTITUENT_AFTERSIGN"
)
after_complex_constituent = ET.SubElement(
xml_sent, "AFTER_COMPOUND_CONSTITUENT"
)
core_prediction = predictions[0]
core_prediction = re.sub("IN_", "", core_prediction)
core_prediction = re.sub("BEFORE_", "", core_prediction)
core_prediction = re.sub("AFTER_", "", core_prediction)
core_prediction = re.sub("_AFTERSIGN", "", core_prediction)
core_prediction = re.sub("_BEFORESIGN", "", core_prediction)
xml_sent.attrib["TYPE"] = core_prediction
x = 0
while x < len(predictions):
if "_BEFORESIGN" in predictions[x]:
word = ET.SubElement(in_complex_constituent_beforesign, "W")
word.text = tokens[x]
word.attrib["POS"] = pos_tags[x]
word.attrib["LEMMA"] = lemmas[x]
elif "_AFTERSIGN" in predictions[x]:
if re.match(
"(and|but|or|that|what|when|where|which|while|who)",
tokens[x],
re.IGNORECASE,
):
if x > 0 and tokens[x - 1] in [",", ":", ";"]:
word = ET.SubElement(in_complex_constituent, "W")
word.text = tokens[x]
word.attrib["POS"] = pos_tags[x]
word.attrib["LEMMA"] = lemmas[x]
else:
word = ET.SubElement(in_complex_constituent_aftersign, "W")
word.text = tokens[x]
word.attrib["POS"] = pos_tags[x]
word.attrib["LEMMA"] = lemmas[x]
else:
word = ET.SubElement(in_complex_constituent_aftersign, "W")
word.text = tokens[x]
word.attrib["POS"] = pos_tags[x]
word.attrib["LEMMA"] = lemmas[x]
elif (
"NOT_NOUN_COORDINATOR" in predictions[x]
or "NOT_NP_COORDINATOR" in predictions[x]
or "NOT_CLAUSE_COORDINATOR" in predictions[x]
or "NOT_VP_COORDINATOR" in predictions[x]
):
word = ET.SubElement(in_complex_constituent_beforesign, "W")
word.text = tokens[x]
word.attrib["POS"] = pos_tags[x]
word.attrib["LEMMA"] = lemmas[x]
elif "IN_" in predictions[x]:
word = ET.SubElement(in_complex_constituent, "W")
word.text = tokens[x]
word.attrib["POS"] = pos_tags[x]
word.attrib["LEMMA"] = lemmas[x]
elif "BEFORE_" in predictions[x]:
word = ET.SubElement(before_complex_constituent, "W")
word.text = tokens[x]
word.attrib["POS"] = pos_tags[x]
word.attrib["LEMMA"] = lemmas[x]
elif "AFTER_" in predictions[x]:
word = ET.SubElement(after_complex_constituent, "W")
word.text = tokens[x]
word.attrib["POS"] = pos_tags[x]
word.attrib["LEMMA"] = lemmas[x]
x += 1
# Print the XML (as you requested)
print("BUILT XML SENTENCE:\n" + ET.tostring(xml_sent).decode("utf-8"))
return xml_sent
#############################################################################
#############################################################################
def get_COMPLEX_NP_type(xml_sent):
complex_NP_type = "UNKNOWN"
final_icca_pos_tag = "UNKNOWN"
first_icca_pos_tag = "UNKNOWN"
final_iccb_pos_tag = "UNKNOWN"
punct = re.compile(r"^[\,\:\;]$")
VERB_TAGS = ["VB", "VBD", "VBG", "VBN", "VBZ", "VBP", "MD"]
NOMINAL_TAGS = ["NNP", "NNPS", "NN", "NNS", "DT", "CD", "PRP", "PRP$"]
for el in xml_sent:
if el.tag == "IN_COMPLEX_CONSTITUENT_AFTERSIGN":
try:
# first *content* token, not a leading comma
for w in el:
if w.text and not punct.match(w.text.strip()):
first_icca_pos_tag = w.attrib.get("POS", "UNKNOWN")
break
# last *content* token, not a trailing comma
for w in reversed(list(el)):
if w.text and not punct.match(w.text.strip()):
final_icca_pos_tag = w.attrib.get("POS", "UNKNOWN")
break
except Exception:
pass
elif el.tag == "IN_COMPLEX_CONSTITUENT_BEFORESIGN":
try:
# last *content* token, not a trailing comma
for w in reversed(list(el)):
if w.text and not punct.match(w.text.strip()):
final_iccb_pos_tag = w.attrib.get("POS", "UNKNOWN")
break
except Exception:
pass
if final_iccb_pos_tag in ["IN"]:
complex_NP_type = "PREPOSITION_OBJECT_RELATIVISED"
elif first_icca_pos_tag in NOMINAL_TAGS and final_icca_pos_tag in VERB_TAGS:
complex_NP_type = "VERB_OBJECT_RELATIVISED"
elif first_icca_pos_tag in VERB_TAGS:
complex_NP_type = "SUBJECT_RELATIVISED"
return complex_NP_type
#############################################################################
#############################################################################
def simplified_sent_xml_to_text(xml):
parts = []
for el in xml:
if el.tag == "W":
if el.text:
parts.append(el.text.strip())
else:
for sub_el in el:
if sub_el.text:
parts.append(sub_el.text.strip())
# Join the tokens with single spaces
text = " ".join(parts).strip()
# Clean up any leftover multiple spaces
text = re.sub(r"\s+", " ", text)
# Ensure the sentence ends with a full stop
if text and not text.endswith((".", "!", "?")):
text += "."
# Capitalise only the first character, leave everything else unchanged
if text:
text = text[0].upper() + text[1:]
return text
#############################################################################
###############################################################################
def clean_bpe_text(text):
"""Remove BPE fragments, labels, brackets, and normalize whitespace."""
text = re.sub(r"##", "", text) # remove subword markers
text = re.sub(r"\s*\[[^\]]+\]\s*\(M:Y_[^\)]+\)", "", text) # remove labels
text = re.sub(r"\s*\[[^\]]+\]", "", text) # remove brackets
text = re.sub(r"\s*\(M:Y_[^\)]+\)", "", text)
text = re.sub(r"\s*\(M:N_[^\)]+\)", "", text)
text = re.sub(r"M Y _[^\s]+", "", text)
text = re.sub(r"\s+", " ", text).strip()
return text
###############################################################################
###############################################################################
def fix_apostrophe_spacing(s):
if not s:
return s
s = s.replace("’", "'")
# didn't / don't / isn't / can't ...
s = re.sub(r"\bn\s+'\s+t\b", "n't", s, flags=re.I)
s = re.sub(r"(\w)n\s+'\s+t\b", r"\1n't", s, flags=re.I)
# she's / it's / he's / I'm / you're / we'll / I've ...
s = re.sub(
r"(\w)\s+'\s+(s|re|ll|ve|d|m)\b",
r"\1'\2",
s,
flags=re.I,
)
# any leftover spaces touching an apostrophe
s = re.sub(r"\s+'\s+", "'", s)
s = re.sub(r"(\w)'\s+(\w)", r"\1'\2", s)
s = re.sub(r"(\w)\s+'(\w)", r"\1'\2", s)
return s
###############################################################################
###############################################################################
def clean_output(text):
if not text:
return ""
# brackets and WordPiece fragments
text = re.sub(r'[\[\]]+', '', text)
text = re.sub(r'\s*##\s*', '', text)
text = re.sub(r'#', '', text)
text = re.sub(r'\s+', ' ', text).strip()
# normal spacing around punctuation
text = re.sub(r'\s+([.,!?;:])', r'\1', text)
text = re.sub(r'([.,!?;:])\s+', r'\1 ', text)
# --- fix the trailing ",." pattern ---
text = re.sub(r',+\s*\.', '.', text)
text = re.sub(r',+\s*$', '', text)
text = re.sub(r'\.+$', '.', text)
# --- spaces around apostrophes (didn't / she's / it's) ---
text = text.replace("’", "'")
text = re.sub(r"\bn\s+'\s+t\b", "n't", text, flags=re.I)
text = re.sub(r"(\w)n\s+'\s+t\b", r"\1n't", text, flags=re.I)
text = re.sub(
r"(\w)\s+'\s+(s|re|ll|ve|d|m)\b",
r"\1'\2",
text,
flags=re.I,
)
text = re.sub(r"\s+'\s+", "'", text)
text = re.sub(r"(\w)'\s+(\w)", r"\1'\2", text)
text = re.sub(r"(\w)\s+'(\w)", r"\1'\2", text)
# --- restore infinitival be that SVA rewrote as is/are/was/were ---
text = repair_mangled_infinitive_be(text)
text = text.strip()
if text:
text = text[0].upper() + text[1:]
if not text.endswith(('.', '!', '?')):
text += '.'
return text.strip()
###############################################################################
###############################################################################
def drop_bracket_tokens(tokens, pos, lemmas, preds):
"""Remove literal [ and ] that the tokenizer sometimes emits"""
kept_tokens, kept_pos, kept_lemmas, kept_preds = [], [], [], []
for t, p, l, pr in zip(tokens, pos, lemmas, preds):
if t in ('[', ']'):
continue
kept_tokens.append(t)
kept_pos.append(p)
kept_lemmas.append(l)
kept_preds.append(pr)
return kept_tokens, kept_pos, kept_lemmas, kept_preds
###############################################################################
###############################################################################
def build_single_sign_bracketed(tagged_list, mode="ccv", skip_indices=None):
"""
mode = "ccv" → bracket only the rightmost CCV sign
mode = "cmv1" → bracket only the rightmost CMV1 sign
mode = "cmn1" → bracket only the rightmost CMN1 sign
mode = "cln" → bracket only the rightmost CLN sign
mode = "ssccv" → bracket one relative SSCCV trigger:
- nested relatives (2+ opens before next ESCCV): innermost
- coordinated relatives (CCV between opens): leftmost
- otherwise: leftmost relative trigger
- fallback: leftmost SSCCV/ESCCV (e.g. bare "that")
mode = "ssmn" → bracket only the leftmost SSMN sign
skip_indices: token offsets already tried and rejected
(e.g. SSCCV signs whose span is a cognitive/communicative VP).
Returns None if nothing remains to bracket for this mode.
"""
skip_indices = set(skip_indices or [])
def surface_of(item):
if item.startswith("<M:Y_") and ">" in item:
start = item.find(">") + 1
end = item.rfind("<")
if end > start:
return item[start:end]
return item
ccv_indices = []
cmv1_indices = []
cmn1_indices = []
cln_indices = []
ssccv_indices = []
esccv_indices = []
ssmn_indices = []
for i, item in enumerate(tagged_list):
if i in skip_indices:
continue
s = str(item)
if "<M:Y_CCV>" in s or "M:Y_CCV" in s:
ccv_indices.append(i)
if "<M:Y_CMV1>" in s or "M:Y_CMV1" in s:
cmv1_indices.append(i)
if "<M:Y_CMN1>" in s or "M:Y_CMN1" in s:
cmn1_indices.append(i)
if "<M:Y_CLN>" in s or "M:Y_CLN" in s:
cln_indices.append(i)
if "M:Y_SSCCV" in s:
ssccv_indices.append(i)
if "M:Y_ESCCV" in s:
esccv_indices.append(i)
if i not in ssccv_indices:
ssccv_indices.append(i)
if "M:Y_SSMN" in s:
ssmn_indices.append(i)
target = set()
if mode == "ccv" and ccv_indices:
target.add(ccv_indices[-1])
elif mode == "cmv1" and cmv1_indices:
target.add(cmv1_indices[-1])
elif mode == "cmn1" and cmn1_indices:
target.add(cmn1_indices[-1])
elif mode == "cln" and cln_indices:
target.add(cln_indices[-1])
elif mode == "ssmn" and ssmn_indices:
target.add(ssmn_indices[0])
elif mode == "ssccv" and ssccv_indices:
relative_pronouns = {"who", "which", "whom", "whose"}
preferred = []
for i in ssccv_indices:
surface = surface_of(tagged_list[i]).strip().lower()
if surface in relative_pronouns:
preferred.append(i)
continue
if surface in {",", ";", ":"} and i + 1 < len(tagged_list):
nxt = surface_of(tagged_list[i + 1]).strip().lower()
if nxt in relative_pronouns or nxt == "that":
preferred.append(i)
continue
if preferred:
nested = False
coordinated = False
if esccv_indices:
first_es = min(esccv_indices)
opens_before_close = [i for i in preferred if i < first_es]
if len(opens_before_close) >= 2:
first_open = opens_before_close[0]
last_open = opens_before_close[-1]
ccv_between = any(
first_open < j < last_open for j in ccv_indices
)
if ccv_between:
coordinated = True
target.add(opens_before_close[0])
else:
nested = True
target.add(opens_before_close[-1])
if not nested and not coordinated:
target.add(preferred[0])
else:
pure_ssccv = [
i
for i, item in enumerate(tagged_list)
if i not in skip_indices and "M:Y_SSCCV" in str(item)
]
pool = pure_ssccv or ssccv_indices
if pool:
target.add(pool[0])
if not target:
return None
parts = []
for i, item in enumerate(tagged_list):
if item.startswith("<M:Y_") and item.endswith(">"):
start = item.find(">") + 1
end = item.rfind("<")
surface = item[start:end] if end > start else item
if i in target:
parts.append(f"[{surface}]")
else:
parts.append(surface)
else:
parts.append(item)
text = " ".join(parts)
text = re.sub(r"\s+\[([,;:])", r"[\1", text)
text = re.sub(r"(?<!\[)\[([A-Za-z])", r" [\1", text)
text = re.sub(r"\[\s+", "[", text)
text = re.sub(r"\s+\]", "]", text)
text = re.sub(r"\s+", " ", text).strip()
text = re.sub(
r"\[([,;:])\]\s+(who|which|that|whom|whose)\b",
r"[\1 \2]",
text,
flags=re.I,
)
return text
###############################################################################
###############################################################################
@spaces.GPU(duration=120)
def process_text(input_text):
print("=== ITERATIVE SIMPLIFICATION (priority + clean tokens) ===")
try:
MAX_ITERS = 40
# ==================== PREPROCESSING ====================
input_text = re.sub(r"supposingthat", "supposing that", input_text)
input_text = re.sub(r'"', "''", input_text)
input_text = re.sub(r"\.\s+([a-z])", ".\n\\g<1>", input_text)
input_text = re.sub(r"mainly f or", "mainly for", input_text)
input_text = re.sub(r"sothat", "so that", input_text)
input_text = re.sub(r"in alabama", "in Alabama", input_text)
input_text = re.sub(
r"LEISURE DESIGN IS IMPORTANT WHY is",
"LEISURE DESIGN IS IMPORTANT\n\nWhy is",
input_text,
)
input_text = re.sub(r"WHEN", "When", input_text)
input_text = re.sub(r"WHAT'S", "What's", input_text)
input_text = re.sub(r"(\s+)an'", " and", input_text)
input_text = re.sub(r"An'", "And", input_text)
# =======================================================
raw_paragraphs = re.split(r"\n\s*\n", input_text)
output_paragraphs = []
for para in raw_paragraphs:
para = para.strip()
if not para:
continue
para = re.sub(r"\s*\n\s*", " ", para)
para = re.sub(r"\s+", " ", para).strip()
sents = split_into_sentences(para)
if not sents:
continue
print(f"Paragraph → {len(sents)} sentence(s):")
for i, s in enumerate(sents, 1):
print(f" [{i}] {s}")
para_simplified = []
for original_sent in sents:
working_set = deque([original_sent])
seen = set()
sentence_final = []
iters = 0
while working_set:
iters += 1
if iters > MAX_ITERS:
print(
f"ABORT: exceeded {MAX_ITERS} iterations; "
f"draining {len(working_set)} item(s) to final"
)
while working_set:
sentence_final.append(
clean_output(working_set.popleft())
)
break
current = working_set.popleft().strip()
if not current:
continue
current = clean_output(current)
if not current:
continue
key = re.sub(r"\s+", " ", current.lower())
if key in seen:
print(f"SKIP duplicate: {current[:80]}...")
sentence_final.append(current)
continue
seen.add(key)
print(
f"\n--- iter {iters} | queue={len(working_set)} | "
f"{current[:100]}..."
)
tagged_list = tag_all_signs_in_sent(current)
has_ssccv = any(
any(x in t for x in ["M:Y_SSCCV", "M:Y_ESCCV"])
for t in tagged_list
)
has_ssmn = any(
any(x in t for x in ["M:Y_SSMN", "M:Y_ESMN"])
for t in tagged_list
)
has_ccv = any(
"<M:Y_CCV>" in t or "M:Y_CCV" in t for t in tagged_list
)
has_cmv1 = any(
"<M:Y_CMV1>" in t or "M:Y_CMV1" in t for t in tagged_list
)
has_cmn1 = any(
"<M:Y_CMN1>" in t or "M:Y_CMN1" in t for t in tagged_list
)
has_cln = any(
"<M:Y_CLN>" in t or "M:Y_CLN" in t for t in tagged_list
)
skipped_ssccv = set()
mode = xml_tag = None
simplifier = span_fn = None
bracketed_version = None
# SSCCV first, but skip cognitive/communicative VPs
if has_ssccv:
while True:
bracketed_version = build_single_sign_bracketed(
tagged_list,
mode="ssccv",
skip_indices=skipped_ssccv,
)
if not bracketed_version:
break
print(f"\nProcessing: {current}")
print(
f"Bracketed version (ssccv): {bracketed_version}"
)
cleaned = clean_for_spaCy(bracketed_version)
tokens, pos, lemmas, preds = tag_SSCCV_span(cleaned)
tokens, pos, lemmas, preds = drop_bracket_tokens(
tokens, pos, lemmas, preds
)
if is_cognitive_communicative_vp(preds):
ssccv_idxs = [
i
for i, t in enumerate(tagged_list)
if "M:Y_SSCCV" in str(t)
or "M:Y_ESCCV" in str(t)
]
remaining = [
i
for i in ssccv_idxs
if i not in skipped_ssccv
]
if remaining:
skipped_ssccv.add(remaining[0])
print(
" Skip SSCCV "
"(cognitive/communicative VP)"
)
continue
mode = "ssccv"
xml_tag = "M:Y_SSCCV"
simplifier = SSCCVsimplify
span_fn = tag_SSCCV_span
break
if mode is None and has_ssmn:
mode = "ssmn"
xml_tag = "M:Y_SSMN"
simplifier = SSMNsimplify
span_fn = tag_SSMN_span
elif mode is None and has_ccv:
mode = "ccv"
xml_tag = "M:Y_CCV"
simplifier = CompoundSimplify
span_fn = tag_CCV_span
elif mode is None and has_cmv1:
mode = "cmv1"
xml_tag = "M:Y_CMV1"
simplifier = CompoundSimplify
span_fn = tag_CMV1_span
elif mode is None and has_cmn1:
mode = "cmn1"
xml_tag = "M:Y_CMN1"
simplifier = CompoundSimplify
span_fn = tag_CMN1_span
elif mode is None and has_cln:
mode = "cln"
xml_tag = "M:Y_CLN"
simplifier = CompoundSimplify
span_fn = tag_CLN_span
elif mode is None:
sentence_final.append(current)
continue
if mode != "ssccv":
bracketed_version = build_single_sign_bracketed(
tagged_list, mode=mode
)
if not bracketed_version:
sentence_final.append(current)
continue
print(f"\nProcessing: {current}")
print(
f"Bracketed version ({mode}): {bracketed_version}"
)
cleaned = clean_for_spaCy(bracketed_version)
tokens, pos, lemmas, preds = span_fn(cleaned)
tokens, pos, lemmas, preds = drop_bracket_tokens(
tokens, pos, lemmas, preds
)
cleaned_tokens, clean_pos, clean_lemmas, cleaned_preds = (
sign_cleaning(tokens, pos, lemmas, preds)
)
improved_preds = improve_predicted_tag_sequence(
cleaned_preds
)
xml_sent = create_xml_sent(
cleaned_tokens,
clean_pos,
clean_lemmas,
improved_preds,
xml_tag,
)
simps = simplifier(xml_sent)
if simps:
simps = [fix_subject_verb_agreement(s) for s in simps]
if (
simps
and simplifier is CompoundSimplify
and xml_tag
in ("M:Y_CCV", "M:Y_CMV1", "M:Y_CMN1", "M:Y_CLN")
):
coord = extract_coordinator_from_xml(xml_sent)
if coord:
rewritten = []
for s in simps:
s2 = rewrite_residual_list_comma(s, coord)
if s2 != s:
print(
f" Residual list rewrite ({coord}): "
f"{s!r} → {s2!r}"
)
rewritten.append(s2)
simps = rewritten
if simps and xml_tag == "M:Y_CMN1":
simps = [fix_subject_verb_agreement(s) for s in simps]
if not simps:
print(
"Simplifier returned []; forcing final (no re-queue)"
)
current = fix_subject_verb_agreement(current)
sentence_final.append(current)
continue
queued_any = False
for s in simps:
plain_s = clean_output(s)
plain_s = fix_subject_verb_agreement(plain_s)
if not plain_s:
continue
plain_key = re.sub(r"\s+", " ", plain_s.lower())
if plain_key == key:
print(
f" No progress (same as input): "
f"{plain_s[:80]}..."
)
sentence_final.append(plain_s)
continue
if plain_key in seen:
print(
f" Skip already-seen result: "
f"{plain_s[:80]}..."
)
sentence_final.append(plain_s)
continue
working_set.append(plain_s)
queued_any = True
print(f" → Queued: {plain_s}")
if not queued_any:
sentence_final.append(current)
for s in order_sentences_for_discourse(sentence_final):
s = clean_output(s)
if s:
para_simplified.append(s)
if para_simplified:
para_simplified = order_sentences_for_discourse(para_simplified)
output_paragraphs.append(" ".join(para_simplified))
final_output = "\n\n".join(output_paragraphs)
print(f"\n=== FINAL OUTPUT ===\n{final_output}")
return highlight_final_output(final_output)
except Exception as e:
import traceback
print(traceback.format_exc())
return (
f'<div class="simp-out">'
f"{html.escape('ERROR: ' + str(e))}"
f"</div>"
)
###########################################################################
# Main processing function (encapsulates the pipeline)
# ================== GRADIO INTERFACE ===================
print(" All seven models loaded successfully!")
print("Creating Gradio interface...")
custom_css = """
.gradio-container, .gradio-container * {
font-size: 16px !important;
}
.input-text textarea {
font-size: 17px !important;
color: #ffd0d0 !important;
font-weight: 500 !important;
line-height: 1.45 !important;
}
.input-text label {
font-size: 15px !important;
font-weight: 600 !important;
color: #ffd0d0 !important;
}
/* Merged simplified output: green text, VP underlined */
.simp-out {
color: #7CFF7C !important;
font-size: 17px !important;
font-weight: 500 !important;
line-height: 1.55 !important;
padding: 12px 14px !important;
}
.simp-out p {
margin: 0 0 12px 0 !important;
color: #7CFF7C !important;
}
.simp-out strong u,
.simp-out u {
color: #7CFF7C !important;
font-weight: 700 !important;
text-decoration: underline !important;
text-underline-offset: 3px !important;
}
"""
with gr.Blocks(
title="Syntactic Sentence Simplifier",
css=custom_css,
) as demo:
gr.Markdown("# Syntactic Sentence Simplifier")
gr.Markdown(
"""
This demo simplifies sentences containing compound constituents:
* *clauses*
* *verb phrases*
* *noun phrases*
* *nouns*
and complex noun phrases modified by:
* *nominally bound non-restrictive finite relative clauses (subject relativised and verb/preposition object relativised)*
* *nominally bound non-restrictive non-finite nominal clauses*
Paste text, or click examples to **add** them into the box below
(you can edit the text afterwards).
"""
)
input_text = gr.Textbox(
lines=12,
label="Original text",
placeholder="Paste your text here, or add examples below...",
elem_classes=["input-text"],
)
btn = gr.Button("Process & Simplify", variant="primary")
output_html = gr.HTML(
value="",
label="Simplified text",
)
btn.click(
fn=process_text,
inputs=input_text,
outputs=output_html,
)
EXAMPLE_SENTENCES = [
"John left early and Mary stayed late.",
"John, who lives in London, went home.",
"John, who kicked the ball and walked away, went home.",
"The man and his dog, who live in Bristol, the English city, went home and Mary ran a marathon and smoked cigars and cigarettes.",
"The man and the woman, who live in Skegness, are going home.",
"John thought that he was going home and Mary believed that this was a bad time to do so."
]
def append_example(current_text, example):
example = (example or "").strip()
if not example:
return current_text or ""
current = (current_text or "").strip()
if not current:
return example
return current + "\n\n" + example
gr.Markdown("### Add an example to the box above (click to append; then edit freely)")
clear_btn = gr.Button("Clear input", size="sm")
clear_btn.click(fn=lambda: "", outputs=input_text)
for ex in EXAMPLE_SENTENCES:
label = ex if len(ex) <= 100 else ex[:97] + "..."
ex_btn = gr.Button(label, size="sm")
ex_btn.click(
fn=lambda current, s=ex: append_example(current, s),
inputs=input_text,
outputs=input_text,
)
print("Gradio interface ready. Starting web server...")
if __name__ == "__main__":
demo.launch(
server_name="0.0.0.0",
server_port=7860,
debug=True,
share=False,
)
'''
test = (
"Joshua told me that John, who lives in London, is going to Wolverhampton, Jack loves Norwich, and Mary, who is a bit nervous, adores Birmingham, just as Fred predicted."
)
print(process_text(test))
'''