BOTeome / scripts /utils.py
juan
latest
04e6023
Raw History Blame Contribute Delete
2.74 kB
from builtins import any as b_any
def extract_uniprot_locations(protein):
if 'comments' in protein:
all_locs = [locs['subcellularLocations'] for locs in protein['comments'] if (locs['commentType']=='SUBCELLULAR LOCATION' and 'subcellularLocations' in locs)][0]
locations = [locs['location']['value'] for locs in all_locs]
locations = ','.join(locations)
return locations
else:
return 'no location available from database'
def get_protein_by_accession(accession, proteins):
protein = [prot for prot in proteins if prot['primaryAccession']==accession][0]
return protein
def get_location_from_acession(accession, proteins):
try:
protein = get_protein_by_accession(accession, proteins)
locations = extract_uniprot_locations(protein)
return locations
except IndexError:
return 'Accession not found, maybe ir was merged/renamed ?'
def is_in_nucleus(locations):
try:
if b_any('nucleus' in loc.lower() for loc in locations):
return 'is'
else:
return 'is not'
except:
return 'not available'
def get_comments_from_accession(accession):
from scripts.uniprot import get_protein_comments
return get_protein_comments(accession)
_TF_KEYWORDS = {
'transcription factor': 3.0,
'coactivator': 2.5,
'corepressor': 2.5,
'dna-binding': 2.0,
'dna binding': 2.0,
'rna polymerase': 2.0,
'zinc finger': 1.5,
'homeobox': 1.5,
'helix-loop-helix': 1.5,
'leucine zipper': 1.5,
'transcription': 1.5,
'promoter': 1.5,
'gene expression': 1.0,
'chromatin remodeling': 1.5,
'chromatin': 1.0,
'regulatory': 1.0,
'regulator': 1.0,
'nuclear': 0.5,
'histone': 0.5,
}
# Normalise against the sum of the top-3 weights so a single strong hit ≈ 1.0
_TF_NORM = sum(sorted(_TF_KEYWORDS.values(), reverse=True)[:3])
_UNAVAILABLE = {'not available', 'no comments available', 'accession not found', 'api error'}
def tf_likelihood(comments):
if not isinstance(comments, str) or comments.lower() in _UNAVAILABLE:
return None
text = comments.lower()
score = sum(w for term, w in _TF_KEYWORDS.items() if term in text)
return round(min(score / _TF_NORM, 1.0), 3)
def search(values, searchFor):
for k in values:
try:
for v in values[k]:
if searchFor in v:
return k
else: return None
except TypeError:
continue