{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nLook on test data CAFA5 in particular on genes, descriptions etc \n\nSimilar for train:  https://www.kaggle.com/code/alexandervc/cafa5-train-eda-detailed\n\n","metadata":{}},{"cell_type":"markdown","source":"# Outcomes\n\n    in test we have almost exclusively swiss-prot\n    protein existence in test in shifted to something like \"less clear\"\n    sequence version is quite similar in train and test\n    mostly gene -> 1 protein within one ogranism in that dataset\n    about half proteins - unique gene in data, for the other half there are \"twins\" from the same gene (typically different ogranizm)\n    some genes seems to be present for very far organism - from human to bacteria - it might be error in notation or real biology \n    there are 6 protein with \"no description found\" , 4 of them are obsolete un uniprot, 2 are very new - may 2023\n    \n    Genes with top proteins:\n    rps3      22\n    atpa      22\n    fen1      22\n    rps4      21\n    mt-cyb    19\n    dph2      19\n    pth       19\n    rpl32     19\n    rps12     19\n    adi1      19\n\n    \n    sequence version\n    \n    Train:\n\n    1    98676\n    2    31377\n    3     9357\n    4     2020\n    5      345\n    6       67\n    7       14\n    0        6\n    8        2\n    9        1\n\n    Test:\n\n    1     106753\n    2      25025\n    3       7747\n    4       1973\n    5        332\n    8        246\n    6        107\n    9         31\n    7         24\n    10         5\n    0          3\n    \n    # 1. Experimental evidence at protein level\n    # 2. Experimental evidence at transcript level\n    # 3. Protein inferred from homology\n    # 4. Protein predicted\n    # 5. Protein uncertain\n\n    # For the train: \n    # 1    85719\n    # 2    25663\n    # 3    15401\n    # 4    15290\n    # 5      170\n    # 0        3\n\n    # For the test:\n    # 1    69116\n    # 2    39592\n    # 3    26255\n    # 4     5290\n    # 5     1606\n    # 0        6\n\n\n\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time()\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-17T16:41:34.472929Z","iopub.execute_input":"2023-05-17T16:41:34.473338Z","iopub.status.idle":"2023-05-17T16:41:35.061249Z","shell.execute_reply.started":"2023-05-17T16:41:34.473304Z","shell.execute_reply":"2023-05-17T16:41:35.060066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', 500 )","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:35.063086Z","iopub.execute_input":"2023-05-17T16:41:35.063421Z","iopub.status.idle":"2023-05-17T16:41:35.067993Z","shell.execute_reply.started":"2023-05-17T16:41:35.063393Z","shell.execute_reply":"2023-05-17T16:41:35.067039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Description data from uniprot ","metadata":{}},{"cell_type":"code","source":"%%time\nfrom Bio import SeqIO\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta'\nsequences = SeqIO.parse(fn, \"fasta\")\nlist_id_test = [seq.id for seq in sequences ]\nprint(len(list_id_test), list_id_test[:10])\nsequences = SeqIO.parse(fn, \"fasta\")\nlist_len_test = [len(seq.seq) for seq in sequences ]\nprint(len(list_len_test), list_len_test[:10])\n\nfn = '/kaggle/input/uniprot-description-only/test_source_output.csv'\ndf = pd.read_csv(fn, index_col = 0 ,) #  header = None \nprint(df.shape)\nprint('Check ids coincide:',  list(df['EntryID'] ) == list_id_test ); print()\ndf['index'] = range(len(df))\ndf['length'] = list_len_test\ndisplay(df)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:35.069069Z","iopub.execute_input":"2023-05-17T16:41:35.070036Z","iopub.status.idle":"2023-05-17T16:41:38.342112Z","shell.execute_reply.started":"2023-05-17T16:41:35.069996Z","shell.execute_reply":"2023-05-17T16:41:38.340892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df.head(50))\ndisplay(df.tail(50))","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:38.344473Z","iopub.execute_input":"2023-05-17T16:41:38.344832Z","iopub.status.idle":"2023-05-17T16:41:38.372921Z","shell.execute_reply.started":"2023-05-17T16:41:38.344802Z","shell.execute_reply":"2023-05-17T16:41:38.371732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(df['length'], bins = 1000)\ndisplay(df.describe() )\ndf.corr()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:38.374366Z","iopub.execute_input":"2023-05-17T16:41:38.374725Z","iopub.status.idle":"2023-05-17T16:41:40.313116Z","shell.execute_reply.started":"2023-05-17T16:41:38.374694Z","shell.execute_reply":"2023-05-17T16:41:40.312041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nl = [len(t) for t in df['Description']]\ndf['len description'] = l\nplt.figure(figsize = (10,4) )\nplt.hist(l,bins = 1000)\nplt.title('Length of description test',fontsize = 20)\nprint( pd.Series(l).describe() )\nfor t in range(20,62):\n    m = df['len description'] == t\n    print(t, m.sum() )\n    if m.sum()>0:\n        display(df[m] )\n        print()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:40.314579Z","iopub.execute_input":"2023-05-17T16:41:40.31499Z","iopub.status.idle":"2023-05-17T16:41:42.626244Z","shell.execute_reply.started":"2023-05-17T16:41:40.314961Z","shell.execute_reply":"2023-05-17T16:41:42.624041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = []; i = 0\nfor t in df['Description']:\n    if 'OX=' in t:\n        s1 = t.split('OS=')[0]\n        p = s1.find(' ')\n        s = s1[(p+1):]\n        l.append(s); i += 1\n    else:\n        l.append('')\nprint(i,len(l), l[:10])\ndf['Description Cleaned'] = l\nl =  [len(t) for t in l]\ndf['len description cleaned'] = l\ndisplay(df.head(3))\n\nplt.figure(figsize = (10,4) )\nplt.hist(l,bins = 1000)\nplt.title('Length of description cleaned -  test',fontsize = 20)\nprint( pd.Series(l).describe() )\ncc = 0\nfor t in range(0,10):# 20,62):\n    m = df['len description cleaned'] == t\n    print(t, m.sum() )\n    cc += 1\n    if cc >= 30: break\n    if m.sum()>0:\n        display(df[m] )\n        print()\n        \ncc = 0\nfor t in [100 ]:# 20,62):\n    m = df['len description cleaned'] >= t\n    print(t, m.sum() )\n    cc += 1\n    if cc >= 30: break\n    if m.sum()>0:\n        display(df[m] )\n        print()        ","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:42.627791Z","iopub.execute_input":"2023-05-17T16:41:42.628144Z","iopub.status.idle":"2023-05-17T16:41:45.411099Z","shell.execute_reply.started":"2023-05-17T16:41:42.628114Z","shell.execute_reply":"2023-05-17T16:41:45.409975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sequence length","metadata":{}},{"cell_type":"code","source":"m = df['length'] < 60\nprint( m.sum() )\ndisplay\n\nsr = df['length'].value_counts().sort_index()\n\nfig = plt.figure(figsize = (20,3) )\nplt.plot(sr.head(1500))\nplt.title('Test data - first 1500',fontsize = 20 )\nplt.xlabel('Sequence length',fontsize = 20)\nplt.ylabel('Count',fontsize = 20)\nplt.show()\n\nfig = plt.figure(figsize = (20,3) )\nplt.plot(sr.head(150))\nplt.title('Test data - first 150 ',fontsize = 20 )\nplt.xlabel('Sequence length',fontsize = 20)\nplt.ylabel('Count',fontsize = 20)\nplt.show()\n\nm = df['length'] < 60\nprint('count length < 60 : ', m.sum() )\n\nfor k in range(8):\n    m = df['length'] == k\n    if m.sum() == 0 : continue \n    print(k, m.sum() )\n#     print(dict(df_seq2[m]['description']))\n#     print(dict(df_seq[m]['organism']))    \n    display(df[m])\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:45.412829Z","iopub.execute_input":"2023-05-17T16:41:45.413244Z","iopub.status.idle":"2023-05-17T16:41:46.01344Z","shell.execute_reply.started":"2023-05-17T16:41:45.413204Z","shell.execute_reply":"2023-05-17T16:41:46.01225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:46.015058Z","iopub.execute_input":"2023-05-17T16:41:46.015481Z","iopub.status.idle":"2023-05-17T16:41:46.043006Z","shell.execute_reply.started":"2023-05-17T16:41:46.015441Z","shell.execute_reply":"2023-05-17T16:41:46.041617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# sequence version ","metadata":{}},{"cell_type":"code","source":"%%time\nl = [t for t in df['Description'] if 'SV=' not in t ]\nprint(len(l), l)\n\nl = []\nfor t in df['Description']:\n    if 'SV=' in t : \n        k = int(t.split('SV=')[1])\n    else:\n        k = 0\n    l.append(k)\nprint(len(l), np.max(l),  l[:10])\nprint( pd.Series(l).describe() )\nprint(pd.Series(l).value_counts() )\n\ndf['sequence version'] = l\ndf.corr().round(1)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:46.048347Z","iopub.execute_input":"2023-05-17T16:41:46.048717Z","iopub.status.idle":"2023-05-17T16:41:46.350813Z","shell.execute_reply.started":"2023-05-17T16:41:46.048687Z","shell.execute_reply":"2023-05-17T16:41:46.349623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"    sequence version\n    \n    Train:\n\n    1    98676\n    2    31377\n    3     9357\n    4     2020\n    5      345\n    6       67\n    7       14\n    0        6\n    8        2\n    9        1\n\n    Test:\n\n    1     106753\n    2      25025\n    3       7747\n    4       1973\n    5        332\n    8        246\n    6        107\n    9         31\n    7         24\n    10         5\n    0          3\n\n","metadata":{}},{"cell_type":"code","source":"\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# protein existence","metadata":{}},{"cell_type":"code","source":"%%time\n# protein existence\nl = [t for t in df['Description'] if 'PE=' not in t ]\nprint(len(l), l)\nl = []\nfor t in df['Description']:\n    if 'SV=' in t : \n        k = int(t.split('PE=')[1].split(' ')[0])\n    else:\n        k = 0\n    l.append(k)\nprint(len(l), np.max(l),  l[:10])\nprint( pd.Series(l).describe() )\nprint(pd.Series(l).value_counts() )\n\ndf['protein existence'] = l\n\ndf.corr().round(1)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:46.352965Z","iopub.execute_input":"2023-05-17T16:41:46.353477Z","iopub.status.idle":"2023-05-17T16:41:46.70527Z","shell.execute_reply.started":"2023-05-17T16:41:46.353439Z","shell.execute_reply":"2023-05-17T16:41:46.703982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Experimental evidence at protein level\n# 2. Experimental evidence at transcript level\n# 3. Protein inferred from homology\n# 4. Protein predicted\n# 5. Protein uncertain\n\n# For the train: \n# 1    85719\n# 2    25663\n# 3    15401\n# 4    15290\n# 5      170\n# 0        3\n\n# For the test:\n# 1    69116\n# 2    39592\n# 3    26255\n# 4     5290\n# 5     1606\n# 0        6\n\n\nfor col in ['protein existence']:\n    print(col)\n    display( df[col].value_counts() )\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:46.706607Z","iopub.execute_input":"2023-05-17T16:41:46.706945Z","iopub.status.idle":"2023-05-17T16:41:46.71888Z","shell.execute_reply.started":"2023-05-17T16:41:46.706915Z","shell.execute_reply":"2023-05-17T16:41:46.717997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# fragment","metadata":{}},{"cell_type":"code","source":"l = ['fragment' in t.lower() for t in df['Description'] ]\ndf['fragment'] = l\nprint( pd.Series(l).describe() )\nl = [t for t in df['Description']  if 'fragment' in t.lower() ]\nprint(len(l), l[:10])\ndf.corr().round(1)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:46.719911Z","iopub.execute_input":"2023-05-17T16:41:46.720221Z","iopub.status.idle":"2023-05-17T16:41:46.914852Z","shell.execute_reply.started":"2023-05-17T16:41:46.720194Z","shell.execute_reply":"2023-05-17T16:41:46.913615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# taxonomyID","metadata":{}},{"cell_type":"code","source":"%%time\n# taxonomyID\nl = [t for t in df['Description'] if 'OX=' not in t ]\nprint(len(l), l)\nl = []\nfor t in df['Description']:\n    if 'OX=' in t : \n        k = int(t.split('OX=')[1].split(' ')[0])\n    else:\n        k = 0\n    l.append(k)\nprint(len(l), np.max(l),  l[:10])\nprint( pd.Series(l).describe() )\nprint(pd.Series(l).value_counts() )\ndf['taxonomyID'] = l\ndf.corr().round(1)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:46.916125Z","iopub.execute_input":"2023-05-17T16:41:46.916429Z","iopub.status.idle":"2023-05-17T16:41:47.27586Z","shell.execute_reply.started":"2023-05-17T16:41:46.916403Z","shell.execute_reply":"2023-05-17T16:41:47.274476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# dbase ","metadata":{}},{"cell_type":"code","source":"l = []\nfor t in df['Description']:\n    if '|' in t:\n        l.append(t.split('|')[0])\n    else:\n        l.append('sp')\n        print(t)\n        \nprint(len(l),  l[:10]) # np.max(l), \nprint( pd.Series(l).describe() )\nprint(pd.Series(l).value_counts() )\ndf['dbase'] = l\ndf.corr().round(1)        ","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:47.277181Z","iopub.execute_input":"2023-05-17T16:41:47.277518Z","iopub.status.idle":"2023-05-17T16:41:47.487646Z","shell.execute_reply.started":"2023-05-17T16:41:47.277483Z","shell.execute_reply":"2023-05-17T16:41:47.48647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['dbase'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:47.489638Z","iopub.execute_input":"2023-05-17T16:41:47.490692Z","iopub.status.idle":"2023-05-17T16:41:47.511738Z","shell.execute_reply.started":"2023-05-17T16:41:47.490647Z","shell.execute_reply":"2023-05-17T16:41:47.510698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# In Train ","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\nsequences = SeqIO.parse(fn, \"fasta\")\nlist_ids_train = [seq.id for seq in sequences ]\nprint( len(list_ids_train), list_ids_train[:10] )\n\ndf['common train test'] = df['EntryID'].isin(list_ids_train ).astype(int) # .columns\ndf['common train test'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:47.513313Z","iopub.execute_input":"2023-05-17T16:41:47.513732Z","iopub.status.idle":"2023-05-17T16:41:48.991465Z","shell.execute_reply.started":"2023-05-17T16:41:47.513663Z","shell.execute_reply":"2023-05-17T16:41:48.990318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Organism","metadata":{"execution":{"iopub.status.busy":"2023-05-17T15:26:04.839615Z","iopub.execute_input":"2023-05-17T15:26:04.840689Z","iopub.status.idle":"2023-05-17T15:26:04.845812Z","shell.execute_reply.started":"2023-05-17T15:26:04.840639Z","shell.execute_reply":"2023-05-17T15:26:04.844623Z"}}},{"cell_type":"code","source":"#### ------- organism - extracted from \"OX\" (if no - from the second term) --------------------------------\nl = []\nfor t in df['Description']:\n    if 'OS=' in t:\n        s = t.split('OS=')[1].split('OX=')[0]\n        if s[-1] == ' ': s = s[:-1]\n    elif 'HUMAN' in t: # Manually checked that only for humans it can happen\n        print(t)\n        s = 'Homo sapiens'\n    else: \n        # that cannot happen - manually checked \n        print('Problem !!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!')\n        print(t)\n        s = ''# t.split('_')[1].split(' ')[0]\n    l.append(s)\ndf['organism'] = l\nprint(len(s), l[:30])","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:48.992883Z","iopub.execute_input":"2023-05-17T16:41:48.993452Z","iopub.status.idle":"2023-05-17T16:41:49.16596Z","shell.execute_reply.started":"2023-05-17T16:41:48.993416Z","shell.execute_reply":"2023-05-17T16:41:49.164895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = df['organism'].value_counts()#.to_frame()\n# d1.columns = ['Protein count']\ndisplay(d1.head(30) )    \nprint();\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:49.167701Z","iopub.execute_input":"2023-05-17T16:41:49.168368Z","iopub.status.idle":"2023-05-17T16:41:49.193467Z","shell.execute_reply.started":"2023-05-17T16:41:49.168332Z","shell.execute_reply":"2023-05-17T16:41:49.192047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_test = df['common train test'] == 0\ndf[ mask_test]['organism'].value_counts().head(60)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:49.195152Z","iopub.execute_input":"2023-05-17T16:41:49.195657Z","iopub.status.idle":"2023-05-17T16:41:49.272269Z","shell.execute_reply.started":"2023-05-17T16:41:49.195578Z","shell.execute_reply":"2023-05-17T16:41:49.271074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Genes","metadata":{}},{"cell_type":"code","source":"##### ---------------- gene ------------------------------------------------------------------------------------------------ \nl = []; ll = []; l_tmp = []\ncc = 0\nfor t in df['Description']:\n    if 'GN=' in t:\n        s = t.split('GN=')[-1].split(' ')[0]\n    else:\n        # Case like: sp|Q922M7|ASHWN_MOUSE Ashwin OS=Mus musculus OX=10090 PE=1 SV=1\n        # we take ASHWN\n        cc += 1\n#         print(t)\n        s = t.split('_')[0].split('|')[-1]\n        l_tmp.append(s)\n    l.append(s); ll.append(s.lower() )\nprint('GN not found:', cc )\nprint(len(l),len(set(ll)),  l[:30]) \nprint(len(l_tmp), l_tmp[:30])\ndf['gene name'] = l\ndf['gene name lower'] = ll\nplt.hist(df['gene name lower'].value_counts(), bins =50)\nplt.show()\nfor k in range(1,10):\n    m = df['gene name lower'].value_counts() == k\n    print(k, m.sum() )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:49.273835Z","iopub.execute_input":"2023-05-17T16:41:49.274455Z","iopub.status.idle":"2023-05-17T16:41:50.396919Z","shell.execute_reply.started":"2023-05-17T16:41:49.274422Z","shell.execute_reply":"2023-05-17T16:41:50.395759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_test = df['common train test'] == 0\nfor k in range(1,10):\n    m = df[mask_test]['gene name lower'].value_counts() == k\n    print(k, m.sum() )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:50.398813Z","iopub.execute_input":"2023-05-17T16:41:50.399603Z","iopub.status.idle":"2023-05-17T16:41:50.941381Z","shell.execute_reply.started":"2023-05-17T16:41:50.399554Z","shell.execute_reply":"2023-05-17T16:41:50.939839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### ------------------ Brief Look on frequent genes ----------------------------\nprint()\nv = df['gene name lower'].value_counts()\ndisplay( v.head(5) )\nm = df['gene name lower'] == 'ank2'\nprint(df[m]['organism'].value_counts()  )\n\nprint(); print(); \nprint('rps3')\nm = df['gene name lower'] == 'rps3'\nprint(df[m]['organism'].value_counts()  )\nprint(); print();print()\n\nprint(); print(); \nprint('atpa')\nm = df['gene name lower'] == 'atpa'\nprint(df[m]['organism'].value_counts()  )\nprint(); print();print()\n\nprint(); print(); \nprint('fen1')\nm = df['gene name lower'] == 'fen1'\nprint(df[m]['organism'].value_counts()  )\nprint(); print();print()\n\nfig = plt.figure(figsize = (20, 3))\nax = plt.plot(v.iloc[0:75], '*-')\nplt.xticks(rotation = 90, fontsize = 15)\nplt.show()\nv.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:50.942891Z","iopub.execute_input":"2023-05-17T16:41:50.943336Z","iopub.status.idle":"2023-05-17T16:41:51.639066Z","shell.execute_reply.started":"2023-05-17T16:41:50.943292Z","shell.execute_reply":"2023-05-17T16:41:51.637796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Genes with top number of proteins","metadata":{}},{"cell_type":"code","source":"df['gene name lower'].value_counts().head(60)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:00:02.722655Z","iopub.execute_input":"2023-05-17T17:00:02.723042Z","iopub.status.idle":"2023-05-17T17:00:02.802925Z","shell.execute_reply.started":"2023-05-17T17:00:02.723013Z","shell.execute_reply":"2023-05-17T17:00:02.801627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['gene name lower'].value_counts().head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:58:39.36882Z","iopub.execute_input":"2023-05-17T16:58:39.369187Z","iopub.status.idle":"2023-05-17T16:58:39.443378Z","shell.execute_reply.started":"2023-05-17T16:58:39.369159Z","shell.execute_reply":"2023-05-17T16:58:39.442611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_test = df['common train test'] == 0\ndf[ mask_test]['gene name lower'].value_counts().head(30)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:51.64038Z","iopub.execute_input":"2023-05-17T16:41:51.64081Z","iopub.status.idle":"2023-05-17T16:41:51.701684Z","shell.execute_reply.started":"2023-05-17T16:41:51.640772Z","shell.execute_reply":"2023-05-17T16:41:51.70064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor uv in df['gene name lower'].value_counts().index[:5]:\n    m = df['gene name lower'] == uv \n    print(uv, m.sum() )\n    display(df[m])#  ['organism'].value_counts()  )\n    print(); print();print()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:51.703516Z","iopub.execute_input":"2023-05-17T16:41:51.70431Z","iopub.status.idle":"2023-05-17T16:41:51.927639Z","shell.execute_reply.started":"2023-05-17T16:41:51.70427Z","shell.execute_reply":"2023-05-17T16:41:51.926606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d1 = df['organism'].value_counts().to_frame()\nd1.columns = ['Protein count']\n# display(d1.head(30) )    \nprint();\n\nprint()\nd = df.groupby('organism')[ 'gene name lower'].apply(lambda x: len(set(x))  ).sort_values(ascending = False)\nd = d1.join(d)\ndisplay( d.head(35) )#  value_counts().head(30) )    \nprint();\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:51.929028Z","iopub.execute_input":"2023-05-17T16:41:51.929431Z","iopub.status.idle":"2023-05-17T16:41:52.009029Z","shell.execute_reply.started":"2023-05-17T16:41:51.929394Z","shell.execute_reply":"2023-05-17T16:41:52.007753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:41:52.01142Z","iopub.execute_input":"2023-05-17T16:41:52.012082Z","iopub.status.idle":"2023-05-17T16:41:52.03635Z","shell.execute_reply.started":"2023-05-17T16:41:52.01204Z","shell.execute_reply":"2023-05-17T16:41:52.035253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save df","metadata":{}},{"cell_type":"code","source":"df.to_csv('df_test_eda.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:51:22.760368Z","iopub.execute_input":"2023-05-17T16:51:22.760794Z","iopub.status.idle":"2023-05-17T16:51:24.120016Z","shell.execute_reply.started":"2023-05-17T16:51:22.760758Z","shell.execute_reply":"2023-05-17T16:51:24.118877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Saving features ","metadata":{}},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:42:22.127243Z","iopub.execute_input":"2023-05-17T16:42:22.127683Z","iopub.status.idle":"2023-05-17T16:42:22.136648Z","shell.execute_reply.started":"2023-05-17T16:42:22.127649Z","shell.execute_reply":"2023-05-17T16:42:22.135194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = pd.DataFrame()\nd['sequence version'] = df['sequence version'].values\nd['protein existence'] = df['protein existence'].values\ndisplay(d.describe() )\nd.to_csv('test_features01_SV_PE.csv')\nnp.save( 'test_features01_SV_PE', d.values ) \nd","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:45:57.818888Z","iopub.execute_input":"2023-05-17T16:45:57.819295Z","iopub.status.idle":"2023-05-17T16:45:58.071278Z","shell.execute_reply.started":"2023-05-17T16:45:57.819264Z","shell.execute_reply":"2023-05-17T16:45:58.070081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:46:38.034308Z","iopub.execute_input":"2023-05-17T16:46:38.034719Z","iopub.status.idle":"2023-05-17T16:46:38.04141Z","shell.execute_reply.started":"2023-05-17T16:46:38.034685Z","shell.execute_reply":"2023-05-17T16:46:38.040636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = pd.DataFrame()\nl = ['gene name','gene name lower' ]\nfor col in l:\n    d[col] = df[col].values\ndisplay(d.describe() )\nd.to_csv('test_features02_genes.csv')\nnp.save( 'test_features02_genes', d.values ) \nd","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:47:57.487907Z","iopub.execute_input":"2023-05-17T16:47:57.488314Z","iopub.status.idle":"2023-05-17T16:47:58.087061Z","shell.execute_reply.started":"2023-05-17T16:47:57.488284Z","shell.execute_reply":"2023-05-17T16:47:58.085929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:49:14.137663Z","iopub.execute_input":"2023-05-17T16:49:14.138055Z","iopub.status.idle":"2023-05-17T16:49:14.146162Z","shell.execute_reply.started":"2023-05-17T16:49:14.138024Z","shell.execute_reply":"2023-05-17T16:49:14.14451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = pd.DataFrame()\nl = ['Description Cleaned','Description']\nfor col in l:\n    d[col] = df[col].values\ndisplay(d.describe() )\nd.to_csv('test_features03_description.csv')\nnp.save( 'test_features03_description', d.values ) ","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:49:57.713846Z","iopub.execute_input":"2023-05-17T16:49:57.714248Z","iopub.status.idle":"2023-05-17T16:49:58.858801Z","shell.execute_reply.started":"2023-05-17T16:49:57.714218Z","shell.execute_reply":"2023-05-17T16:49:58.85757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:50:19.883065Z","iopub.execute_input":"2023-05-17T16:50:19.88346Z","iopub.status.idle":"2023-05-17T16:50:19.892186Z","shell.execute_reply.started":"2023-05-17T16:50:19.883428Z","shell.execute_reply":"2023-05-17T16:50:19.890917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = pd.DataFrame()\nl = ['dbase','fragment']\nfor col in l:\n    d[col] = df[col].values\ndisplay(d.describe() )\nd.to_csv('test_features04_dbase_fragment.csv')\nnp.save( 'test_features04_dbase_fragment', d.values ) ","metadata":{"execution":{"iopub.status.busy":"2023-05-17T16:52:35.487117Z","iopub.execute_input":"2023-05-17T16:52:35.487516Z","iopub.status.idle":"2023-05-17T16:52:35.799804Z","shell.execute_reply.started":"2023-05-17T16:52:35.487488Z","shell.execute_reply":"2023-05-17T16:52:35.7984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}