File size: 1,771 Bytes
63ec1fc
 
 
 
dde6930
63ec1fc
 
 
dde6930
 
 
 
 
 
63ec1fc
 
dde6930
04e6023
 
 
 
63ec1fc
6525afb
 
04e6023
dde6930
 
 
 
 
 
04e6023
6525afb
 
 
 
 
 
 
 
 
 
63ec1fc
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
import pandas as pd

from sqlalchemy import create_engine

from scripts.uniprot import get_proteins_info_batch
from scripts.utils import *


_NO_LOCATION = {'not available', 'api error', 'no location available'}

def _in_nucleus(locations):
    if not isinstance(locations, str) or locations.lower() in _NO_LOCATION:
        return 'not available'
    return 'is' if 'nucleus' in locations.lower() else 'is not'


def create_database(db_uri, csv_path, accession_col=None, expression_cols=None):
    if csv_path.endswith((".xlsx", ".xls")):
        df = pd.read_excel(csv_path)
    else:
        df = pd.read_csv(csv_path)

    if accession_col and accession_col in df.columns:
        df = df[~df[accession_col].isna()]
        accessions = df[accession_col].dropna().unique().tolist()
        info = get_proteins_info_batch(accessions)  # {acc: {comments, locations}}
        df['locations'] = df[accession_col].map(
            lambda a: info.get(a, {}).get('locations', 'not available'))
        df['nucleus'] = df['locations'].apply(_in_nucleus)
        df['uniprot_comments'] = df[accession_col].map(
            lambda a: info.get(a, {}).get('comments', 'not available'))
        df['tf_likelihood'] = df['uniprot_comments'].apply(tf_likelihood)

    if expression_cols and len(expression_cols) >= 2:
        threshold = 10
        df['region'] = 'inconclusive'
        for col in expression_cols:
            others = [c for c in expression_cols if c != col]
            mask = pd.Series(True, index=df.index)
            for other in others:
                mask = mask & (df[col] / df[other] > threshold)
            df.loc[mask, 'region'] = col

    engine = create_engine(db_uri, echo=False)
    df.to_sql(name='proteins', con=engine, if_exists='replace')