File size: 2,739 Bytes
456f631
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
04e6023
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
456f631
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
from  builtins import any as b_any

def extract_uniprot_locations(protein):
    if 'comments' in protein:
        all_locs = [locs['subcellularLocations'] for locs in protein['comments'] if (locs['commentType']=='SUBCELLULAR LOCATION' and 'subcellularLocations' in locs)][0]
        locations = [locs['location']['value'] for locs in all_locs]
        locations = ','.join(locations)
        return locations
    else:
        return 'no location available from database'

def get_protein_by_accession(accession, proteins):
    protein = [prot for prot in proteins if prot['primaryAccession']==accession][0]
    return protein

def get_location_from_acession(accession, proteins):
    try:
        protein = get_protein_by_accession(accession, proteins)
        locations = extract_uniprot_locations(protein)
        return locations
    except IndexError:
        return 'Accession not found, maybe ir was merged/renamed ?'
    

    
def is_in_nucleus(locations):
    try:
        if b_any('nucleus' in loc.lower() for loc in locations):
            return 'is'
        else:
            return 'is not'
    except:
        return 'not available'
    
def get_comments_from_accession(accession):
    from scripts.uniprot import get_protein_comments
    return get_protein_comments(accession)


_TF_KEYWORDS = {
    'transcription factor':   3.0,
    'coactivator':            2.5,
    'corepressor':            2.5,
    'dna-binding':            2.0,
    'dna binding':            2.0,
    'rna polymerase':         2.0,
    'zinc finger':            1.5,
    'homeobox':               1.5,
    'helix-loop-helix':       1.5,
    'leucine zipper':         1.5,
    'transcription':          1.5,
    'promoter':               1.5,
    'gene expression':        1.0,
    'chromatin remodeling':   1.5,
    'chromatin':              1.0,
    'regulatory':             1.0,
    'regulator':              1.0,
    'nuclear':                0.5,
    'histone':                0.5,
}

# Normalise against the sum of the top-3 weights so a single strong hit ≈ 1.0
_TF_NORM = sum(sorted(_TF_KEYWORDS.values(), reverse=True)[:3])

_UNAVAILABLE = {'not available', 'no comments available', 'accession not found', 'api error'}

def tf_likelihood(comments):
    if not isinstance(comments, str) or comments.lower() in _UNAVAILABLE:
        return None
    text = comments.lower()
    score = sum(w for term, w in _TF_KEYWORDS.items() if term in text)
    return round(min(score / _TF_NORM, 1.0), 3)
    


def search(values, searchFor):
    for k in values:
        try:
            for v in values[k]:
                if searchFor in v:
                    return k
                else: return None
        except TypeError:
            continue