Spaces:
Sleeping
Sleeping
Download scripts/utils.py from jugacostase/BOTeome: direct link, hf CLI and curl.
- Browser
- Download file 2.74 kB
-
https://huggingface.co/spaces/jugacostase/BOTeome/resolve/main/scripts/utils.py
- Command line
-
hf download hf://spaces/jugacostase/BOTeome/scripts/utils.py
-
curl -L -o utils.py https://huggingface.co/spaces/jugacostase/BOTeome/resolve/main/scripts/utils.py
2.74 kB
| from builtins import any as b_any | |
| def extract_uniprot_locations(protein): | |
| if 'comments' in protein: | |
| all_locs = [locs['subcellularLocations'] for locs in protein['comments'] if (locs['commentType']=='SUBCELLULAR LOCATION' and 'subcellularLocations' in locs)][0] | |
| locations = [locs['location']['value'] for locs in all_locs] | |
| locations = ','.join(locations) | |
| return locations | |
| else: | |
| return 'no location available from database' | |
| def get_protein_by_accession(accession, proteins): | |
| protein = [prot for prot in proteins if prot['primaryAccession']==accession][0] | |
| return protein | |
| def get_location_from_acession(accession, proteins): | |
| try: | |
| protein = get_protein_by_accession(accession, proteins) | |
| locations = extract_uniprot_locations(protein) | |
| return locations | |
| except IndexError: | |
| return 'Accession not found, maybe ir was merged/renamed ?' | |
| def is_in_nucleus(locations): | |
| try: | |
| if b_any('nucleus' in loc.lower() for loc in locations): | |
| return 'is' | |
| else: | |
| return 'is not' | |
| except: | |
| return 'not available' | |
| def get_comments_from_accession(accession): | |
| from scripts.uniprot import get_protein_comments | |
| return get_protein_comments(accession) | |
| _TF_KEYWORDS = { | |
| 'transcription factor': 3.0, | |
| 'coactivator': 2.5, | |
| 'corepressor': 2.5, | |
| 'dna-binding': 2.0, | |
| 'dna binding': 2.0, | |
| 'rna polymerase': 2.0, | |
| 'zinc finger': 1.5, | |
| 'homeobox': 1.5, | |
| 'helix-loop-helix': 1.5, | |
| 'leucine zipper': 1.5, | |
| 'transcription': 1.5, | |
| 'promoter': 1.5, | |
| 'gene expression': 1.0, | |
| 'chromatin remodeling': 1.5, | |
| 'chromatin': 1.0, | |
| 'regulatory': 1.0, | |
| 'regulator': 1.0, | |
| 'nuclear': 0.5, | |
| 'histone': 0.5, | |
| } | |
| # Normalise against the sum of the top-3 weights so a single strong hit ≈ 1.0 | |
| _TF_NORM = sum(sorted(_TF_KEYWORDS.values(), reverse=True)[:3]) | |
| _UNAVAILABLE = {'not available', 'no comments available', 'accession not found', 'api error'} | |
| def tf_likelihood(comments): | |
| if not isinstance(comments, str) or comments.lower() in _UNAVAILABLE: | |
| return None | |
| text = comments.lower() | |
| score = sum(w for term, w in _TF_KEYWORDS.items() if term in text) | |
| return round(min(score / _TF_NORM, 1.0), 3) | |
| def search(values, searchFor): | |
| for k in values: | |
| try: | |
| for v in values[k]: | |
| if searchFor in v: | |
| return k | |
| else: return None | |
| except TypeError: | |
| continue |