{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Outcomes\n\n\nThese 2 ogranisms are in TEST (and train also), while seems to be OVER-presented terms  in TRAIN:\n\n    Leishmania donovani (strain BPK282A1)\t67.000000\t2\n    Fusarium oxysporum f. sp. lycopersici (strain 4287 / CBS 123668 / FGSC 9935 / NRRL 34936)\t64.000000\t1\n\nThere are 85 \"Fusarium\"-related samples in TEST, and only 4 in train ! That might be difficult case for the model.\nLeishmania - is vice versa - many in train, and only 4 in test - duplicates of train. \n    \nThese 3 ogranisms seems to be under-presented terms in average in TRAIN (the last two of them also exist in test) :\n    \n    Desulfovibrio vulgaris (strain ATCC 29579 / DSM 644 / NCIMB 8303 / VKM B-1760 / Hildenborough)\t3.656388\t227\n    Streptococcus pneumoniae serotype 4 (strain ATCC BAA-334 / TIGR4)\t3.923077\t195\n    Helicobacter pylori (strain ATCC 700392 / 26695)\t6.873874\t111\n    \n'bacter' in organism: train: 3450 test: 2898  \n\nMedian of target number for SwissProt is 32 while for TrEmbl is 16 - big difference. \n\ngenes symbols - abundant in test - are also abundant in train , but not quite vice versa\n\nWe have about 13k \"orphan\" proteins by \"max cleaned\" gene symbols - e.g. not common gene symbols with the other protein. \nAbout 19K for \"middle cleaned\" (only END digits removed) and about 50k if genes without cleaning - these quite similar for train and test.\nAbout 7k test genes not have a common \"max cleaned\" gene symbol with the train.\n\nAbout 75-83K \"orphans\" in description sense. \n\n\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time()\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-29T08:32:53.401014Z","iopub.execute_input":"2023-05-29T08:32:53.401462Z","iopub.status.idle":"2023-05-29T08:32:54.485544Z","shell.execute_reply.started":"2023-05-29T08:32:53.401427Z","shell.execute_reply":"2023-05-29T08:32:54.48427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom Bio import SeqIO","metadata":{"execution":{"iopub.status.busy":"2023-05-29T08:32:54.489796Z","iopub.execute_input":"2023-05-29T08:32:54.490152Z","iopub.status.idle":"2023-05-29T08:32:54.638938Z","shell.execute_reply.started":"2023-05-29T08:32:54.49012Z","shell.execute_reply":"2023-05-29T08:32:54.637554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train data","metadata":{}},{"cell_type":"code","source":"%%time\ndf = pd.read_csv('/kaggle/input/cafa5-data-selected/df_train_eda.csv', index_col = 0)\nprint(df.shape)\ndisplay(df.describe())\ndf","metadata":{"execution":{"iopub.status.busy":"2023-05-29T08:32:54.641432Z","iopub.execute_input":"2023-05-29T08:32:54.64241Z","iopub.status.idle":"2023-05-29T08:32:58.238359Z","shell.execute_reply.started":"2023-05-29T08:32:54.642366Z","shell.execute_reply":"2023-05-29T08:32:58.236974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SwissProt \"x2\" targets than TrEmbl ","metadata":{"execution":{"iopub.status.busy":"2023-05-26T19:48:23.990504Z","iopub.execute_input":"2023-05-26T19:48:23.991754Z","iopub.status.idle":"2023-05-26T19:48:23.999749Z","shell.execute_reply.started":"2023-05-26T19:48:23.991677Z","shell.execute_reply":"2023-05-26T19:48:23.998704Z"}}},{"cell_type":"code","source":"d2 = df.select_dtypes(include=np.number)\nfor t in ['sp', 'tr']:\n    m = df['dbase'] == t\n    print(t, m.sum() )\n    # pd.concat( [ df[m].mean(), df[m].median() ] , axis = 1 )\n    if t == 'sp':\n        _d = pd.concat( (d2[m].mean(), d2[m].median() ), axis = 1 ).round(1)\n        _d.columns = ['sp mean', 'sp median']\n    else:\n        _d = pd.concat( (_d, pd.concat( (d2[m].mean(), d2[m].median() ), axis = 1 ).round(1) ), axis = 1 )\n        _d.columns = ['sp mean', 'sp median', 'tr mean', 'tr median']\n        _d = _d[ [ 'sp mean', 'tr mean', 'sp median', 'tr median' ] ]\n    _v = df[m][ 'Terms Count' ]    \n    plt.hist(_v[_v<300], bins = 1000, label = t) \n    \ndisplay(_d)\nplt.legend()\nplt.show()    \n#print() \n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-29T08:32:58.240361Z","iopub.execute_input":"2023-05-29T08:32:58.240744Z","iopub.status.idle":"2023-05-29T08:33:02.817436Z","shell.execute_reply.started":"2023-05-29T08:32:58.240706Z","shell.execute_reply":"2023-05-29T08:33:02.816252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test data EDA","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/cafa5-data-selected/df_test_eda.csv'\ndt = pd.read_csv(fn, index_col = 0)\nprint(dt.shape)\ndisplay(dt.describe())\ndt","metadata":{"execution":{"iopub.status.busy":"2023-05-29T08:33:02.818658Z","iopub.execute_input":"2023-05-29T08:33:02.819006Z","iopub.status.idle":"2023-05-29T08:33:04.160912Z","shell.execute_reply.started":"2023-05-29T08:33:02.818977Z","shell.execute_reply":"2023-05-29T08:33:04.159767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Description analysis 1 ","metadata":{}},{"cell_type":"code","source":"d = {}\nfor k in ['Train', 'Test']:\n#     print() ; print(); print(k); print()\n    if k == 'Train':\n        l = [ str(t).lower() for t in  df['Description Cleaned'] ]\n    else:\n        l = [ str(t).lower() for t in  dt['Description Cleaned'] ]\n    d[k] = l\n    \ns = set(d['Train']) & set(d['Test'])\nlen(s), list(s)[:3]","metadata":{"execution":{"iopub.status.busy":"2023-05-29T09:44:54.685079Z","iopub.execute_input":"2023-05-29T09:44:54.685536Z","iopub.status.idle":"2023-05-29T09:44:54.88995Z","shell.execute_reply.started":"2023-05-29T09:44:54.685501Z","shell.execute_reply":"2023-05-29T09:44:54.888517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in ['Train', 'Test']:\n    print() ; print(); print(k); print()\n    if k == 'Train':\n        l = [ str(t).lower() for t in  df['Description Cleaned'] ]\n    else:\n        l = [ str(t).lower() for t in  dt['Description Cleaned'] ]\n        \n    print(len(set(l)), list(set(l))[:2])\n    sr = pd.Series(l).value_counts()\n    print('Count orphans:', (sr==1).sum() )\n    print(sr.describe(percentiles = [0.25, 0.5, 0.75,0.9,0.95,0.99]))\n    print(sr.head(50))\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-29T09:34:03.186034Z","iopub.execute_input":"2023-05-29T09:34:03.186905Z","iopub.status.idle":"2023-05-29T09:34:03.626832Z","shell.execute_reply.started":"2023-05-29T09:34:03.186856Z","shell.execute_reply":"2023-05-29T09:34:03.625441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Genes symbols analysis - 1","metadata":{}},{"cell_type":"code","source":"print('Train')\nprint('Count different genes symbols in train:')\n\nl = df['gene name lower']\nimport re\npattern = r\"\\d+$\"#  \\d+$\nl1 = [re.sub(pattern, \"\", str(t)) for t in l ]\npattern = r\"\\d+\"#  \\d+$\nl2 = [re.sub(pattern, \"\", str(t)) for t in l ]\nprint(len(set(l)), len(set(l1)), len(set(l2)),)\nprint( (pd.Series(l2).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ' , (pd.Series(l).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ' , (pd.Series(l1).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ',  (pd.Series(l2).value_counts() == 1).sum() )\n\n\nprint(); print('Test')\nprint('Count different genes symbols in test:')\nl = dt['gene name lower']\nimport re\npattern = r\"\\d+$\"#  \\d+$\nl1 = [re.sub(pattern, \"\", str(t)) for t in l ]\npattern = r\"\\d+\"#  \\d+$\nl2 = [re.sub(pattern, \"\", str(t)) for t in l ]\nprint(len(set(l)), len(set(l1)), len(set(l2)),)\nprint( 'Count \"orphans\" ' , (pd.Series(l).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ' , (pd.Series(l1).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ',  (pd.Series(l2).value_counts() == 1).sum() )\n\n\nprint()\nprint('in test, but not in train - maximal genes simimilarity - digits removed')\npattern = r\"\\d+\"#  \\d+$\nl = df['gene name lower']\nl1 = [re.sub(pattern, \"\", str(t)) for t in l ]\nl = dt['gene name lower']\nl2 = [re.sub(pattern, \"\", str(t)) for t in l ]\ns= set(l2)-set(l1)\nprint(len(s), list(s)[:30])\n\nprint()\nprint('in test, but not in train - middle genes simimilarity - LAST digits removed')\npattern = r\"\\d+$\"#  \\d+$\nl = df['gene name lower']\nl1 = [re.sub(pattern, \"\", str(t)) for t in l ]\nl = dt['gene name lower']\nl2 = [re.sub(pattern, \"\", str(t)) for t in l ]\ns= set(l2)-set(l1)\nprint(len(s), list(s)[:30])\n\nprint()\nprint('in test, but not in train - minimal genes simimilarity - nothing removed')\npattern = r\"\\d+$\"#  \\d+$\nl1 = df['gene name lower']\nl2 = dt['gene name lower']\ns= set(l2)-set(l1)\nprint(len(s), list(s)[:30])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gene symbols analysis \n\nabundant in test - are also abundant in train , but not quite vice versa","metadata":{}},{"cell_type":"code","source":"display( pd.concat( (df['gene name lower'].value_counts().reset_index(), dt['gene name lower'].value_counts().reset_index() ), axis = 1 ).head(15) )\n\n\n\ns = set(df['gene name lower']) & set( dt['gene name lower'])\nprint(len(s))\nm1 = df['gene name lower'].isin(s)\nprint(m1.sum())\nm2 = dt['gene name lower'].isin(s) \nprint(m2.sum())\n# t = ( df[m1]['gene name lower'].value_counts(),  dt[m2]['gene name lower'].value_counts() )\ndd = pd.concat( (df[m1]['gene name lower'].value_counts(),  dt[m2]['gene name lower'].value_counts() ),  axis  = 1 )\ndd.columns = ['train', 'test']\ndisplay( dd.sort_values(dd.columns[1],ascending = False).head(10) )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-29T08:44:28.30015Z","iopub.execute_input":"2023-05-29T08:44:28.301657Z","iopub.status.idle":"2023-05-29T08:44:28.868692Z","shell.execute_reply.started":"2023-05-29T08:44:28.3016Z","shell.execute_reply":"2023-05-29T08:44:28.86752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train')\nprint('Count different genes symbols in train:')\n\nl = df['gene name lower']\nimport re\npattern = r\"\\d+$\"#  \\d+$\nl1 = [re.sub(pattern, \"\", str(t)) for t in l ]\npattern = r\"\\d+\"#  \\d+$\nl2 = [re.sub(pattern, \"\", str(t)) for t in l ]\nprint(len(set(l)), len(set(l1)), len(set(l2)),)\nprint( (pd.Series(l2).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ' , (pd.Series(l).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ' , (pd.Series(l1).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ',  (pd.Series(l2).value_counts() == 1).sum() )\n\n\nprint(); print('Test')\nprint('Count different genes symbols in test:')\nl = dt['gene name lower']\nimport re\npattern = r\"\\d+$\"#  \\d+$\nl1 = [re.sub(pattern, \"\", str(t)) for t in l ]\npattern = r\"\\d+\"#  \\d+$\nl2 = [re.sub(pattern, \"\", str(t)) for t in l ]\nprint(len(set(l)), len(set(l1)), len(set(l2)),)\nprint( 'Count \"orphans\" ' , (pd.Series(l).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ' , (pd.Series(l1).value_counts() == 1).sum() )\nprint( 'Count \"orphans\" ',  (pd.Series(l2).value_counts() == 1).sum() )\n\n\nprint()\nprint('in test, but not in train - maximal genes simimilarity - digits removed')\npattern = r\"\\d+\"#  \\d+$\nl = df['gene name lower']\nl1 = [re.sub(pattern, \"\", str(t)) for t in l ]\nl = dt['gene name lower']\nl2 = [re.sub(pattern, \"\", str(t)) for t in l ]\ns= set(l2)-set(l1)\nprint(len(s), list(s)[:30])\n\nprint()\nprint('in test, but not in train - middle genes simimilarity - LAST digits removed')\npattern = r\"\\d+$\"#  \\d+$\nl = df['gene name lower']\nl1 = [re.sub(pattern, \"\", str(t)) for t in l ]\nl = dt['gene name lower']\nl2 = [re.sub(pattern, \"\", str(t)) for t in l ]\ns= set(l2)-set(l1)\nprint(len(s), list(s)[:30])\n\nprint()\nprint('in test, but not in train - minimal genes simimilarity - nothing removed')\npattern = r\"\\d+$\"#  \\d+$\nl1 = df['gene name lower']\nl2 = dt['gene name lower']\ns= set(l2)-set(l1)\nprint(len(s), list(s)[:30])","metadata":{"execution":{"iopub.status.busy":"2023-05-29T09:12:16.93625Z","iopub.execute_input":"2023-05-29T09:12:16.936662Z","iopub.status.idle":"2023-05-29T09:12:20.301963Z","shell.execute_reply.started":"2023-05-29T09:12:16.936632Z","shell.execute_reply":"2023-05-29T09:12:20.300854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\npattern = r\"\\d+\"#  \\d+$\nl1 = [re.sub(pattern, \"\", str(t)) for t in df['gene name lower'] ]\nl2 = [re.sub(pattern, \"\", str(t)) for t in dt['gene name lower'] ]\ns = set(l1) & set(l2)\nprint(len(s), list(s)[:5])\n\npattern = r\"\\d+$\"#  \\d+$\nl1 = [re.sub(pattern, \"\", str(t)) for t in df['gene name lower'] ]\nl2 = [re.sub(pattern, \"\", str(t)) for t in dt['gene name lower'] ]\ns = set(l1) & set(l2)\nprint(len(s), list(s)[:5])","metadata":{"execution":{"iopub.status.busy":"2023-05-29T08:59:34.497786Z","iopub.execute_input":"2023-05-29T08:59:34.498216Z","iopub.status.idle":"2023-05-29T08:59:35.894853Z","shell.execute_reply.started":"2023-05-29T08:59:34.498168Z","shell.execute_reply":"2023-05-29T08:59:35.893524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\npattern = r\"\\d+\"#  \\d+$\nl = [re.sub(pattern, \"\", str(t)) for t in df['gene name lower'] ]\ns = pd.Series(l).value_counts()\nprint( (s>1).sum() )\nprint(s.describe() )\ns.head(50)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-29T08:58:01.873699Z","iopub.execute_input":"2023-05-29T08:58:01.874233Z","iopub.status.idle":"2023-05-29T08:58:02.294087Z","shell.execute_reply.started":"2023-05-29T08:58:01.874167Z","shell.execute_reply":"2023-05-29T08:58:02.293253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\npattern = r\"\\d+$\"#  \\d+$\nl = [re.sub(pattern, \"\", str(t)) for t in df['gene name lower'] ]\ns = pd.Series(l).value_counts()\nprint( (s>1).sum() )\nprint(s.describe() )\ns.head(50)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-29T08:58:03.040358Z","iopub.execute_input":"2023-05-29T08:58:03.041533Z","iopub.status.idle":"2023-05-29T08:58:03.485949Z","shell.execute_reply.started":"2023-05-29T08:58:03.041488Z","shell.execute_reply":"2023-05-29T08:58:03.484473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt['gene name lower'].value_counts().head(30)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T08:34:18.494763Z","iopub.execute_input":"2023-05-29T08:34:18.495077Z","iopub.status.idle":"2023-05-29T08:34:18.573701Z","shell.execute_reply.started":"2023-05-29T08:34:18.495049Z","shell.execute_reply":"2023-05-29T08:34:18.572665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 'Description Cleaned' - top words ","metadata":{}},{"cell_type":"code","source":"%%time\nl = []\nfor t in  dt['Description Cleaned']:  l+=str(t).lower().split(' ')  \nprint(len(l))\nprint( pd.Series(l).value_counts().head(50) )\nprint( pd.Series(l).value_counts().iloc[50:100] )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-28T21:07:12.71626Z","iopub.execute_input":"2023-05-28T21:07:12.716642Z","iopub.status.idle":"2023-05-28T21:07:13.268389Z","shell.execute_reply.started":"2023-05-28T21:07:12.716614Z","shell.execute_reply":"2023-05-28T21:07:13.267296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v = dt['Description Cleaned'].value_counts()\nprint( (v>1).sum() )\n\nv2 = df['Description Cleaned'].value_counts()\ns = set(v.index[:500]) & set( v2.index[:500] )\nprint( len(s), list(s)[:20] )\n\nprint(v.describe() )\nv.head(50)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T21:00:05.933837Z","iopub.execute_input":"2023-05-28T21:00:05.934257Z","iopub.status.idle":"2023-05-28T21:00:06.112729Z","shell.execute_reply.started":"2023-05-28T21:00:05.934224Z","shell.execute_reply":"2023-05-28T21:00:06.111504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v = df['Description Cleaned'].value_counts()\nprint( (v>1).sum() )\nprint(v.describe() )\nprint( v.head(50) )\nprint( v.iloc[50:100] )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-28T20:57:16.919219Z","iopub.execute_input":"2023-05-28T20:57:16.920625Z","iopub.status.idle":"2023-05-28T20:57:17.021647Z","shell.execute_reply.started":"2023-05-28T20:57:16.920569Z","shell.execute_reply":"2023-05-28T20:57:17.020124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 'uncharacterized', 'putative', 'probable' - less terms","metadata":{}},{"cell_type":"code","source":"for kw in ['uncharacterized', 'putative', 'probable']:\n    m1 = np.array( [ kw in str(t).lower() for t in df['Description Cleaned'] ] )\n    m2 = np.array( [ kw in str(t).lower() for t in dt['Description Cleaned'] ])\n    print(kw, 'train:', m1.sum(), 'test:', m2.sum() )\n    d1 = df[m1][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\n    d2 = df[~m1][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\n    dd = pd.concat( (d1,d2), axis = 1 )\n    dd.columns = [ kw + ' in' , kw + ' not in']\n    display(dd)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T20:22:37.783301Z","iopub.execute_input":"2023-05-28T20:22:37.783862Z","iopub.status.idle":"2023-05-28T20:22:38.464134Z","shell.execute_reply.started":"2023-05-28T20:22:37.783818Z","shell.execute_reply":"2023-05-28T20:22:38.463002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ['Terms Count MFO']","metadata":{}},{"cell_type":"code","source":"m = df['dbase'] == 'sp'\nprint( df['Terms Count MFO'][ m].describe() )\nprint( df[m ] ['Terms Count MFO'].value_counts() )\n\nm = df['Terms Count MFO'] == 0\nprint(m.sum())\nm = m & ( df['dbase'] == 'sp' )\nprint(m.sum())\n","metadata":{"execution":{"iopub.status.busy":"2023-05-28T20:48:44.86587Z","iopub.execute_input":"2023-05-28T20:48:44.866294Z","iopub.status.idle":"2023-05-28T20:48:44.96627Z","shell.execute_reply.started":"2023-05-28T20:48:44.866262Z","shell.execute_reply":"2023-05-28T20:48:44.964879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Levenshtein distance data - train to test","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/cafa5-data-selected/_Levenshtein_stat_data/df_Levenshtein_TOP100_distances_train_to_test.csv'\ndlt = pd.read_csv(fn, index_col = 0)\ndlt\n","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:28.251257Z","iopub.execute_input":"2023-05-28T08:12:28.251543Z","iopub.status.idle":"2023-05-28T08:12:31.845248Z","shell.execute_reply.started":"2023-05-28T08:12:28.25152Z","shell.execute_reply":"2023-05-28T08:12:31.84372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dlt.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:31.846456Z","iopub.execute_input":"2023-05-28T08:12:31.846747Z","iopub.status.idle":"2023-05-28T08:12:32.20831Z","shell.execute_reply.started":"2023-05-28T08:12:31.846722Z","shell.execute_reply":"2023-05-28T08:12:32.206927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt2 = dt.set_index('EntryID')\nm = dt2['common train test'] == 0\ndisplay( dt2[m].head(2) )\n\ndlt.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:32.209785Z","iopub.execute_input":"2023-05-28T08:12:32.210183Z","iopub.status.idle":"2023-05-28T08:12:32.273862Z","shell.execute_reply.started":"2023-05-28T08:12:32.210152Z","shell.execute_reply":"2023-05-28T08:12:32.272879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test seqs which are far from train ","metadata":{}},{"cell_type":"code","source":"m = dt2.index.isin( dlt.index )\nprint(m.sum(), len(dlt) )\ndt2 = dt2[m]\n\nm = dt2['length'] > 100\nv = dlt[m].iloc[:,0] / dt2[m]['length']\nplt.hist(v,bins = 100)\nplt.show()\nprint( (v > 0.7).sum(),  (v > 0.7).sum()/len(v)  )\nprint( v.describe() ) ","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:32.275121Z","iopub.execute_input":"2023-05-28T08:12:32.276709Z","iopub.status.idle":"2023-05-28T08:12:33.025861Z","shell.execute_reply.started":"2023-05-28T08:12:32.276676Z","shell.execute_reply":"2023-05-28T08:12:33.024942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = (dt2['length'] > 200) & ( dt2['length'] < 600 )\nprint(m.sum())\n\nv = dlt[m].iloc[:,0] / dt2[m]['length']\nplt.hist(v,bins = 100)\nplt.show()\nprint( (v > 0.7).sum(),  (v > 0.7).sum()/len(v)  )\nprint( v.describe() ) \nfor t in [0.1, 0.5, 0.6, 0.7]:\n    print(t, (v > t).sum(),  (v > t).sum()/len(v)  )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:23:37.636462Z","iopub.execute_input":"2023-05-28T08:23:37.636822Z","iopub.status.idle":"2023-05-28T08:23:38.01759Z","shell.execute_reply.started":"2023-05-28T08:23:37.636794Z","shell.execute_reply":"2023-05-28T08:23:38.016279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nm2 = v > 0.75\nprint(m2.sum() )\nl = v[m2].index; c = 0\nprint(list(v[m2].index))\nfor p in l:\n    c +=1; \n    if c >= 100: break\n    print()\n    print(p, 'len:', dt2.loc[p]['length'] ,   dt2.loc[p]['organism'] , ' |  ',  dt2.loc[p]['gene name'] ,  ' |  ' , dt2.loc[p]['Description Cleaned'] )\n    p2 = dlt.loc[p].iat[100] \n    print('nearest train neigbour :', p2, 'dist:', dlt.loc[p].iat[0],  'len:', df.loc[p2]['length'] ,  df.loc[p2]['organism'] , ' |  ',  df.loc[p2]['gene name'] ,  ' |  ' , \n          df.loc[p2]['Description Cleaned'] ,)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:33.410929Z","iopub.execute_input":"2023-05-28T08:12:33.411376Z","iopub.status.idle":"2023-05-28T08:12:33.803295Z","shell.execute_reply.started":"2023-05-28T08:12:33.411339Z","shell.execute_reply":"2023-05-28T08:12:33.801657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load distances train to train - 1 ","metadata":{}},{"cell_type":"code","source":"%%time\ndl = pd.read_csv('/kaggle/input/cafa5-data-selected/_Levenshtein_stat_data/df_Levenshtein_TOP100_distances_train_to_train.csv', index_col = 0)\ndl","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:33.80504Z","iopub.execute_input":"2023-05-28T08:12:33.805382Z","iopub.status.idle":"2023-05-28T08:12:41.314716Z","shell.execute_reply.started":"2023-05-28T08:12:33.805353Z","shell.execute_reply":"2023-05-28T08:12:41.313293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dl.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:41.315791Z","iopub.execute_input":"2023-05-28T08:12:41.316085Z","iopub.status.idle":"2023-05-28T08:12:41.846692Z","shell.execute_reply.started":"2023-05-28T08:12:41.316061Z","shell.execute_reply":"2023-05-28T08:12:41.845409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Look on the Titin homolog - most far from the other in train ","metadata":{}},{"cell_type":"code","source":"m = dl.iloc[:,1]> 13800\nprint(m.sum() )\ndl[m] # Titin homolog # G4SLH0 18562 Caenorhabditis elegans","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:41.848763Z","iopub.execute_input":"2023-05-28T08:12:41.849155Z","iopub.status.idle":"2023-05-28T08:12:41.871171Z","shell.execute_reply.started":"2023-05-28T08:12:41.849124Z","shell.execute_reply":"2023-05-28T08:12:41.869531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt.set_index('EntryID').loc['G4SLH0']","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:41.874007Z","iopub.execute_input":"2023-05-28T08:12:41.874448Z","iopub.status.idle":"2023-05-28T08:12:41.941318Z","shell.execute_reply.started":"2023-05-28T08:12:41.874412Z","shell.execute_reply":"2023-05-28T08:12:41.939495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Look on genes which are quite far from the others (train)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T19:29:24.532Z","iopub.execute_input":"2023-05-27T19:29:24.532429Z","iopub.status.idle":"2023-05-27T19:29:24.536746Z","shell.execute_reply.started":"2023-05-27T19:29:24.532396Z","shell.execute_reply":"2023-05-27T19:29:24.535817Z"}}},{"cell_type":"code","source":"m = df['length'] > 100\nprint(m.sum())\nv = dl.iloc[:,1][m] / df['length'][m]\nplt.hist(v,bins = 100)\nplt.show()\nprint( v.describe() ) \nv1 = v.copy()\n\nm = ( df['length'] > 100 )& ( df['dbase'] == 'sp')\n\nprint(m.sum())\nv = dl.iloc[:,1][m] / df['length'][m]\nplt.hist(v,bins = 100)\nplt.title('only Swiss Prot')\nplt.show()\nprint( v.describe() ) \nv2 = v.copy()\n\nfor t in [0.1,0.2, 0.5, 0.6, 0.7]:\n    print(t, 'all: %.2f'%( (v1>t).sum()/len(v1 ) ),   'swiss prot: %.2f'%( (v2>t).sum()/len(v2) )  ) ","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:32:10.41445Z","iopub.execute_input":"2023-05-28T08:32:10.414782Z","iopub.status.idle":"2023-05-28T08:32:10.989857Z","shell.execute_reply.started":"2023-05-28T08:32:10.414756Z","shell.execute_reply":"2023-05-28T08:32:10.988315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = ( df['length'] > 200 ) & ( df['length'] < 600 ) \nm = m & (df['dbase'] == 'sp')\nprint(m.sum())\nv = dl.iloc[:,1][m] / df['length'][m]\nplt.hist(v,bins = 100)\nplt.show()\nprint( v.describe() ) \n\nd1 = df[m][v>0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\nd2 = df[m][v<0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\nd3 = df[m][v<0.1][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\ndd = pd.concat( (d1,d2,d3), axis = 1).round(1)\ndd.columns = [ 'far from others' , 'not so far', 'quite close' ]\ndisplay(dd  )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:49:58.752271Z","iopub.execute_input":"2023-05-28T08:49:58.752656Z","iopub.status.idle":"2023-05-28T08:49:59.11768Z","shell.execute_reply.started":"2023-05-28T08:49:58.752607Z","shell.execute_reply":"2023-05-28T08:49:59.116748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = 1\nprint(' neigbour n=', N)\nm = ( df['length'] > 200 ) & ( df['length'] < 600 ) \n# m = m & (df['dbase'] == 'sp')\nprint(m.sum())\nv = dl.iloc[:,N][m] / df['length'][m]\nplt.hist(v,bins = 100)\nplt.show()\nprint( v.describe() ) \n\nd1 = df[m][v>0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\nd2 = df[m][v<0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\nd3 = df[m][v<0.1][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\ndd = pd.concat( (d1,d2,d3), axis = 1).round(1)\ndd.columns = [ 'far from others' , 'not so far', 'quite close' ]\ndisplay(dd  )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:51:12.499783Z","iopub.execute_input":"2023-05-28T08:51:12.500161Z","iopub.status.idle":"2023-05-28T08:51:12.893734Z","shell.execute_reply.started":"2023-05-28T08:51:12.500133Z","shell.execute_reply":"2023-05-28T08:51:12.892797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for N in [ 1, 2,3,5,8,10, 20]:\n    #N = 10\n    print(' neigbour n=', N)\n    m = ( df['length'] > 200 ) & ( df['length'] < 600 ) \n    # m = m & (df['dbase'] == 'sp')\n    print(m.sum())\n    v = dl.iloc[:,N][m] / df['length'][m]\n    plt.figure( figsize = (10,3))\n    plt.hist(v,bins = 100)\n    plt.title( ' neigbour n= ' + str( N) )\n    plt.show()\n    print( v.describe() ) \n\n    d1 = df[m][v>0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\n    d2 = df[m][v<0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\n    d3 = df[m][v<0.1][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\n    dd = pd.concat( (d1,d2,d3), axis = 1).round(1)\n    dd.columns = [ 'far from others' , 'not so far', 'quite close' ]\n    display(dd  )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:54:08.801309Z","iopub.execute_input":"2023-05-28T08:54:08.80162Z","iopub.status.idle":"2023-05-28T08:54:12.208795Z","shell.execute_reply.started":"2023-05-28T08:54:08.801596Z","shell.execute_reply":"2023-05-28T08:54:12.207321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nm = ( df['length'] > 200 ) & ( df['length'] < 600 ) \nm = m & (df['dbase'] == 'sp')\nprint(m.sum())\nv = dl.iloc[:,10][m] / df['length'][m]\nplt.hist(v,bins = 100)\nplt.show()\nprint( v.describe() ) \n\nd1 = df[m][v>0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\nd2 = df[m][v<0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\nd3 = df[m][v<0.1][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean()\ndd = pd.concat( (d1,d2,d3), axis = 1).round(1)\ndd.columns = [ 'far from others' , 'not so far', 'quite close' ]\ndisplay(dd  )\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = ( df['length'] > 200 ) & ( df['length'] < 600 ) \nprint(m.sum())\nv = dl.iloc[:,1][m] / df['length'][m]\nprint( df[m][v>0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean() )\nprint( df[m][v<0.7][ [ 'Terms Count',       'Terms Count BPO', 'Terms Count CCO', 'Terms Count MFO',]].mean() )","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:40:02.039173Z","iopub.execute_input":"2023-05-28T08:40:02.039594Z","iopub.status.idle":"2023-05-28T08:40:02.155257Z","shell.execute_reply.started":"2023-05-28T08:40:02.039562Z","shell.execute_reply":"2023-05-28T08:40:02.153867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = ( df['length'] > 200 ) & ( df['length'] < 600 )\nprint(m.sum())\nv = dl.iloc[:,1][m] / df['length'][m]\nplt.hist(v,bins = 100)\nplt.show()\nprint( v.describe() ) \n\nm2 = v > 0.748\nprint(m2.sum() )\nl = v[m2].index; c = 0\nprint(list(v[m2].index))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:12:42.253493Z","iopub.execute_input":"2023-05-28T08:12:42.254227Z","iopub.status.idle":"2023-05-28T08:12:42.534663Z","shell.execute_reply.started":"2023-05-28T08:12:42.254195Z","shell.execute_reply":"2023-05-28T08:12:42.533654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = ( df['length'] > 200 ) & ( df['length'] < 600 )\nprint(m.sum())\nv = dl.iloc[:,1][m] / df['length'][m]\nplt.hist(v,bins = 100)\nplt.show()\nprint( v.describe() ) \n\nm2 = v > 0.74\nprint(m2.sum() )\nl = v[m2].index; c = 0\nprint(list(v[m2].index))\nfor p in l:\n    c +=1; \n    if c >= 300: break\n    print()\n    print(p,'len:', df.loc[p]['length'] ,   df.loc[p]['organism'] , ' |  ',  df.loc[p]['gene name'] ,  ' |  ' , df.loc[p]['Description Cleaned'] )\n    p2 = dl.loc[p].iat[101] \n    print('nearest neigbour:', p2, 'dist:', dl.loc[p].iat[1], 'len:', df.loc[p]['length'] , \n          df.loc[p2]['organism'] , ' |  ',  df.loc[p2]['gene name'] ,  ' |  ' , df.loc[p2]['Description Cleaned'] ,)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-28T08:14:17.603846Z","iopub.execute_input":"2023-05-28T08:14:17.604322Z","iopub.status.idle":"2023-05-28T08:14:18.23439Z","shell.execute_reply.started":"2023-05-28T08:14:17.604288Z","shell.execute_reply":"2023-05-28T08:14:18.233059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## another way - absolute, not related to length","metadata":{}},{"cell_type":"code","source":"print( df['length'].median() )\nm = (df['length'] > 300 ) & ( df['length'] < 500)\nprint(m.sum( ))\nprint( dl[m.values].iloc[:,1].max() )\n\nm2 = dl[m.values].iloc[:,1] > 370\nprint(m2.sum())\nl = m2[m2].index\nc = 0\nfor p in l:\n    c+=1\n    if c>=10: break\n    print()        \n    print(p, dl.loc[p].iat[1],dl.loc[p].iat[2], )\n#     print( dt.set_index('EntryID').loc[p] )\n    print( df.loc[p,:] )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:27:26.375046Z","iopub.execute_input":"2023-05-27T20:27:26.375956Z","iopub.status.idle":"2023-05-27T20:27:26.857159Z","shell.execute_reply.started":"2023-05-27T20:27:26.375862Z","shell.execute_reply":"2023-05-27T20:27:26.855612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## here similar - but not restrict length - and we get very lengthy proteins - titin, mucin, etc ...","metadata":{}},{"cell_type":"code","source":"m = dl.iloc[:,1]> 5800\nprint(m.sum() )\ndl[m] # Titin homolog # G4SLH0 18562 Caenorhabditis elegans\n\nl = dl[m].index\nfor p in l:\n    print(p)\n#     print( dt.set_index('EntryID').loc[p] )\n    print( df.loc[p,:] )\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:42.215352Z","iopub.execute_input":"2023-05-27T20:00:42.215763Z","iopub.status.idle":"2023-05-27T20:00:42.248749Z","shell.execute_reply.started":"2023-05-27T20:00:42.215728Z","shell.execute_reply":"2023-05-27T20:00:42.247248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Organisms in test , not in train ","metadata":{}},{"cell_type":"code","source":"# Bacteria\n# Salmonella choleraesuis (strain SC-B67)                                                               882\n\n# Snakes:\n# Austrelaps superbus                                                                                    42\n# Pseudechis australis                                                                                   32\n# Oxyuranus scutellatus scutellatus                                                                      28\n# Pseudechis porphyriacus                                                                                21\n# Demansia vestigiata                                                                                    20\n# Cryptophis nigrescens                                                                                  15\n# Philodryas olfersii                                                                                    11\n# Pseudoferania polylepis                                                                                10\n# Erythrolamprus poecilogyrus                                                                             6\n# Naja nigricollis                                                                                        5\n# Rhabdophis tigrinus tigrinus                                                                            3\n# Thrasops jacksonii                                                                                      3\n\n# Fungi \n# Debaryomyces hansenii (strain ATCC 36239 / CBS 767 / BCRC 21394 / JCM 1990 / NBRC 0083 / IGC 2968)    653\n# Schwanniomyces occidentalis[1] in uska species han Fungi in\n# Marssonina brunnea f. sp. multigermtubi (strain MB_m1)                                                  4\n# Bipolaris zeicola 26-R-13                                                                               1\n\n# Aneomone (actinia:)\n# Heteractis magnifica, also known by the common names magnificent sea anemone or Ritteri anemone,\n\n# Like a mouse: \n# Bathyergus suillus                                                                                      1\n\nprint(dt['organism'].nunique() )\ns = set(dt['organism']) - set(df['organism'])\nprint(len(s), s)\nprint('n_nan=', (dt['organism'].isna() ).sum() )\nm = dt['organism'].isin(s)\ndt[m]['organism'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:42.25178Z","iopub.execute_input":"2023-05-27T20:00:42.252526Z","iopub.status.idle":"2023-05-27T20:00:42.352932Z","shell.execute_reply.started":"2023-05-27T20:00:42.252483Z","shell.execute_reply":"2023-05-27T20:00:42.351623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Genes in test not in train","metadata":{}},{"cell_type":"code","source":"col = 'gene name lower'\n\n# print(dt['organism'].nunique() )\ns = set(dt[col]) - set(df[col])\nprint(len(s), dt[col].nunique(),df[col].nunique(), list(s)[:200] )\nm = dt[col].isin(s)\ndt[m][col].value_counts().head(50)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:42.354797Z","iopub.execute_input":"2023-05-27T20:00:42.355999Z","iopub.status.idle":"2023-05-27T20:00:42.620369Z","shell.execute_reply.started":"2023-05-27T20:00:42.355954Z","shell.execute_reply":"2023-05-27T20:00:42.618951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = 'gene name lower'\n\n# print(dt['organism'].nunique() )\ns = set(dt['organism']) - set(df['organism'])\n# print(len(s), s)\n# print('n_nan=', (dt['organism'].isna() ).sum() )\nm = dt['organism'].isin(s)\n# dt[m]['organism'].value_counts()\nprint( dt[m][col].nunique())\nprint( dt[m][col].value_counts().head(50) )\nprint( list( dt[m][col].value_counts().index[:500] ) )","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:42.622269Z","iopub.execute_input":"2023-05-27T20:00:42.623398Z","iopub.status.idle":"2023-05-27T20:00:42.696591Z","shell.execute_reply.started":"2023-05-27T20:00:42.623346Z","shell.execute_reply":"2023-05-27T20:00:42.695084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Human genes not in train","metadata":{"execution":{"iopub.status.busy":"2023-05-26T20:16:25.271227Z","iopub.execute_input":"2023-05-26T20:16:25.271797Z","iopub.status.idle":"2023-05-26T20:16:25.278095Z","shell.execute_reply.started":"2023-05-26T20:16:25.271752Z","shell.execute_reply":"2023-05-26T20:16:25.276691Z"}}},{"cell_type":"code","source":"m = dt['common train test'] == 0\nprint(m.sum() )\n\nm2 = dt['organism']=='Homo sapiens'\nprint(m2.sum())\nprint((m&m2).sum())\n\nprint( dt[ m&m2 ][ 'gene name lower' ].nunique() )\ndisplay( dt[ m&m2 ][ 'gene name lower' ].value_counts().head(5) )\n\nll = dt[ m&m2 ][ 'gene name lower' ].unique()\n\nm3 = df['gene name lower'].isin(ll)\nprint(m3.sum())\nprint( df[m3]['gene name lower'].value_counts().head(10) )\ndf[m3]\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:42.70052Z","iopub.execute_input":"2023-05-27T20:00:42.700955Z","iopub.status.idle":"2023-05-27T20:00:42.809724Z","shell.execute_reply.started":"2023-05-27T20:00:42.700919Z","shell.execute_reply":"2023-05-27T20:00:42.808264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Organism-wise n_common/not","metadata":{}},{"cell_type":"code","source":"m = dt['common train test'] == 0\nprint(m.sum() )\nd = pd.concat( (dt[m]['organism'].value_counts(), dt['organism'].value_counts(),  dt[~m]['organism'].value_counts() ), axis = 1 )\nd.columns = ['not in train', 'total', 'common train test']\ndisplay( d.sort_values(d.columns[0], ascending = False).head(60) ) #.astype(int)\ndisplay( d.sort_values(d.columns[0], ascending = False).iloc[60:,:] )#.astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:42.811518Z","iopub.execute_input":"2023-05-27T20:00:42.812005Z","iopub.status.idle":"2023-05-27T20:00:42.944013Z","shell.execute_reply.started":"2023-05-27T20:00:42.811967Z","shell.execute_reply":"2023-05-27T20:00:42.942911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display ( dt['organism'].value_counts().head(50) )\ndisplay ( dt['organism'].value_counts().tail(41) )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:42.945414Z","iopub.execute_input":"2023-05-27T20:00:42.945777Z","iopub.status.idle":"2023-05-27T20:00:43.0011Z","shell.execute_reply.started":"2023-05-27T20:00:42.945747Z","shell.execute_reply":"2023-05-27T20:00:42.999479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(dt['organism'])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:43.008407Z","iopub.execute_input":"2023-05-27T20:00:43.008852Z","iopub.status.idle":"2023-05-27T20:00:43.038302Z","shell.execute_reply.started":"2023-05-27T20:00:43.008818Z","shell.execute_reply":"2023-05-27T20:00:43.036944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# No virus-keywords in test, 1399  sample in train with \"virus\"","metadata":{}},{"cell_type":"code","source":"print('Test:')\n\nl = [t for t in dt['organism'] if 'virus' in str(t).lower() ] \nprint(len(l), l)\n\nprint('\\n Train:')\n\nl = [t for t in df['organism'] if 'virus' in str(t).lower() ] \nprint(len(l), len(set(l)), set(l)  )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:43.040592Z","iopub.execute_input":"2023-05-27T20:00:43.042626Z","iopub.status.idle":"2023-05-27T20:00:43.156064Z","shell.execute_reply.started":"2023-05-27T20:00:43.042474Z","shell.execute_reply":"2023-05-27T20:00:43.154458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Test')\nl = [t for t in dt['organism'] if 'Leishmania'.lower() in str(t).lower() ] \nmask1 = [ 'Leishmania'.lower() in str(t).lower() for t in dt['organism'] ]\nprint(len(l),len(set(l)), set(l))\n# display(dt[mask])\nm2 = dt['common train test'] == False\n\nprint('Train')\nl = [t for t in df['organism'] if 'Leishmania'.lower() in str(t).lower() ] \nmask = ['Leishmania'.lower() in str(t).lower() for t in df['organism'] ] \nprint(len(l),len(set(l)), set(l))\n# display(dt[mask])\n\ndisplay(dt[mask1][m2])\n\nm = dt['common train test'] == 0\nprint( 'Desulfovibrio vulgaris (strain ATCC 29579 / DSM 644 / NCIMB 8303 / VKM B-1760 / Hildenborough)' in list( dt[m]['organism']) )\nprint(     'Streptococcus pneumoniae serotype 4 (strain ATCC BAA-334 / TIGR4)' in list(dt[m]['organism'] ) )\nprint(     'Helicobacter pylori (strain ATCC 700392 / 26695)' in list(dt[m]['organism'] ) )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:43.157902Z","iopub.execute_input":"2023-05-27T20:00:43.159121Z","iopub.status.idle":"2023-05-27T20:00:43.57286Z","shell.execute_reply.started":"2023-05-27T20:00:43.159076Z","shell.execute_reply":"2023-05-27T20:00:43.571483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nl = [t for t in dt['organism'] if 'Leishmania'.lower() in str(t).lower() ] \nprint(len(l), l)\n\nprint('Test')\nl = [t for t in dt['organism'] if 'fusarium' in str(t).lower() ] \nmask1 = ['fusarium' in str(t).lower() for t in dt['organism'] ]\nprint(len(l),len(set(l)), set(l))\n# display(dt[mask])\nm2 = dt['common train test'] == False\n\nprint('Train')\nl = [t for t in df['organism'] if 'fusarium' in str(t).lower() ] \nmask = ['fusarium' in str(t).lower() for t in df['organism'] ] \nprint(len(l),len(set(l)), set(l))\n# display(dt[mask])\n\ndisplay(dt[mask1][m2])\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:43.574624Z","iopub.execute_input":"2023-05-27T20:00:43.574972Z","iopub.status.idle":"2023-05-27T20:00:43.899468Z","shell.execute_reply.started":"2023-05-27T20:00:43.574943Z","shell.execute_reply":"2023-05-27T20:00:43.898161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train data EDA","metadata":{}},{"cell_type":"code","source":"%%time\ndf = pd.read_csv('/kaggle/input/cafa5-data-selected/df_train_eda.csv', index_col = 0)\nprint(df.shape)\ndisplay(df.describe())\ndf","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:43.900991Z","iopub.execute_input":"2023-05-27T20:00:43.901328Z","iopub.status.idle":"2023-05-27T20:00:46.065447Z","shell.execute_reply.started":"2023-05-27T20:00:43.9013Z","shell.execute_reply":"2023-05-27T20:00:46.064148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v = df['Terms Count']\nprint(np.mean(v), np.median(v), np.percentile(v, 5),  np.percentile(v, 10),  np.percentile(v, 25),   np.percentile(v, 75), np.percentile(v, 90), np.percentile(v, 95))\nfor k in [5,10, 20, 80, 90,95]:\n    print(k, np.percentile(v, k) )\nprint()\nfor k in range(10):\n    print(k, (v == k).sum() , (v <= k).sum() )\nplt.hist(v,bins = 1000 )\nplt.show() \n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:46.067134Z","iopub.execute_input":"2023-05-27T20:00:46.068386Z","iopub.status.idle":"2023-05-27T20:00:48.70056Z","shell.execute_reply.started":"2023-05-27T20:00:46.068302Z","shell.execute_reply":"2023-05-27T20:00:48.699308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Terms counts per ogranism ","metadata":{}},{"cell_type":"code","source":"d = df.groupby('organism')['Terms Count'].mean().sort_values(ascending = False)\nprint( set(d.index[:50] ) & set(dt['organism']) )\nprint( set(d.index[:75] ) & set(dt['organism']) )\nprint( set(d.index[:100] ) & set(dt['organism']) )\ns1 = set(d.index[:75] ) & set(dt['organism']) \nmask1=d.index.isin(s1)\ns2 = set(d.index[:150] ) & set(dt['organism']) \nmask2=d.index.isin(s2)\n\nplt.figure(figsize = (20,5))\nplt.plot(d.values[:300],'*-')\nplt.title('mean term count per ogranism')\nplt.show()\n\nd2 = df.groupby('organism')['Terms Count'].count().sort_values(ascending = False)\n# d2.head(30)\nd.name = 'mean terms'; d2.name = 'count samples'\nd = d.to_frame().join(d2).sort_values('mean terms', ascending = False)\ndisplay(d.corr(method = 'spearman'))\nmask1=d.index.isin(s1)\ndisplay(d[mask1])\nmask2=d.index.isin(s2)\ndisplay(d[mask2])\ndisplay(d.head(50) )\ndisplay(d.tail(50) )\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:48.702105Z","iopub.execute_input":"2023-05-27T20:00:48.702512Z","iopub.status.idle":"2023-05-27T20:00:49.254605Z","shell.execute_reply.started":"2023-05-27T20:00:48.702479Z","shell.execute_reply":"2023-05-27T20:00:49.253337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d.sort_values('count samples', ascending = False).head(50)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:49.256802Z","iopub.execute_input":"2023-05-27T20:00:49.257186Z","iopub.status.idle":"2023-05-27T20:00:49.276749Z","shell.execute_reply.started":"2023-05-27T20:00:49.257145Z","shell.execute_reply":"2023-05-27T20:00:49.275154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 'bacter', 'virus' etc in train and test","metadata":{}},{"cell_type":"code","source":"list_kw =  ['virus', 'bacter' , 'Streptococcus', 'Aspergillus', 'Leishmania', 'strain' ]\nfor kw in list_kw:\n    m1 =np.array( [ (kw.lower() in str(t).lower() ) for t in df['organism'] ] )\n    m2 =np.array( [ (kw.lower() in str(t).lower() ) for t in dt['organism'] ] )\n    print(kw, 'train:',m1.sum() , 'test:',m2.sum() )\n    print( )\n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:49.279065Z","iopub.execute_input":"2023-05-27T20:00:49.279496Z","iopub.status.idle":"2023-05-27T20:00:50.430273Z","shell.execute_reply.started":"2023-05-27T20:00:49.279462Z","shell.execute_reply":"2023-05-27T20:00:50.428967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# terms count: 9 - 58 for 20-80 percentiles, 5-85 percentiles for 10-90 ","metadata":{}},{"cell_type":"code","source":"v = df['Terms Count']\nprint(np.mean(v), np.median(v), np.percentile(v, 5),  np.percentile(v, 10),  np.percentile(v, 25),   np.percentile(v, 75), np.percentile(v, 90), np.percentile(v, 95))\nprint()\nfor k in [5,10, 20, 80, 90,95]:\n    print(k, np.percentile(v, k) )\nprint()\nfor k in range(10):\n    print(k, (v == k).sum() , (v <= k).sum() )\nplt.hist(v,bins = 1000 )\nplt.show() \n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:50.431895Z","iopub.execute_input":"2023-05-27T20:00:50.432335Z","iopub.status.idle":"2023-05-27T20:00:52.268165Z","shell.execute_reply.started":"2023-05-27T20:00:50.432301Z","shell.execute_reply":"2023-05-27T20:00:52.267034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns, dt.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:52.269695Z","iopub.execute_input":"2023-05-27T20:00:52.270071Z","iopub.status.idle":"2023-05-27T20:00:52.280339Z","shell.execute_reply.started":"2023-05-27T20:00:52.270039Z","shell.execute_reply":"2023-05-27T20:00:52.278857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Genes ","metadata":{}},{"cell_type":"code","source":"s1 = set(df['gene name lower']) \ns2 = set(dt['gene name lower']) \ns = s1& s2\nm2 = dt['gene name lower'].isin(s)\nprint(len(s1),len(s2), len(s), m2.sum() )\n\n\nm = dt['common train test'] == False\nprint(m.sum(), len(dt) - m.sum()  )\ns3 = set(dt['gene name lower'][m]) \ns = s1& s3\nm2 = dt['gene name lower'][m].isin(s)\nprint(len(s1),len(s3), len(s), m2.sum() )","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:52.282756Z","iopub.execute_input":"2023-05-27T20:00:52.283152Z","iopub.status.idle":"2023-05-27T20:00:52.484811Z","shell.execute_reply.started":"2023-05-27T20:00:52.283115Z","shell.execute_reply":"2023-05-27T20:00:52.483554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Y","metadata":{}},{"cell_type":"code","source":"%%time\nn_labels_to_consider = 100\n\n# import scipy\nimport scipy.sparse\nY = scipy.sparse.load_npz('/kaggle/input/cafa5-data-selected/Y_31466_sparse_float32.npz')\nprint(Y.shape)\n# Y = Y[:,:n_labels_to_consider].toarray()\nY","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:52.486845Z","iopub.execute_input":"2023-05-27T20:00:52.487225Z","iopub.status.idle":"2023-05-27T20:00:52.677127Z","shell.execute_reply.started":"2023-05-27T20:00:52.487191Z","shell.execute_reply":"2023-05-27T20:00:52.675395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Targets and genes","metadata":{}},{"cell_type":"code","source":"df['gene name lower'].value_counts().head(20)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:52.679643Z","iopub.execute_input":"2023-05-27T20:00:52.680434Z","iopub.status.idle":"2023-05-27T20:00:52.770215Z","shell.execute_reply.started":"2023-05-27T20:00:52.68039Z","shell.execute_reply":"2023-05-27T20:00:52.768738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TP53","metadata":{}},{"cell_type":"code","source":"gene = 'tp53'\nm = df['gene name lower'] == gene\nprint(m.sum() )\ndisplay( df[m].drop('sequence', axis=1) )\n\nm = dt['gene name lower'] == gene\nprint(m.sum() )\ndisplay( dt[m])# .drop('sequence', axis=1) )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:52.77171Z","iopub.execute_input":"2023-05-27T20:00:52.772076Z","iopub.status.idle":"2023-05-27T20:00:52.883525Z","shell.execute_reply.started":"2023-05-27T20:00:52.772043Z","shell.execute_reply":"2023-05-27T20:00:52.882391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TP73","metadata":{}},{"cell_type":"code","source":"m = df['gene name lower'] == 'tp73'\nprint(m.sum() )\ndisplay( df[m].drop('sequence', axis=1) )\n\nm = dt['gene name lower'] == 'tp73'\nprint(m.sum() )\ndisplay( dt[m])# .drop('sequence', axis=1) )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:52.885598Z","iopub.execute_input":"2023-05-27T20:00:52.886089Z","iopub.status.idle":"2023-05-27T20:00:52.981679Z","shell.execute_reply.started":"2023-05-27T20:00:52.886046Z","shell.execute_reply":"2023-05-27T20:00:52.980393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SRC gene analysis","metadata":{}},{"cell_type":"code","source":"m = df['gene name lower'] == 'src'\nprint(m.sum() )\n\ndf[m]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:52.983035Z","iopub.execute_input":"2023-05-27T20:00:52.983387Z","iopub.status.idle":"2023-05-27T20:00:53.037891Z","shell.execute_reply.started":"2023-05-27T20:00:52.983357Z","shell.execute_reply":"2023-05-27T20:00:53.036446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i, j = 44965, 39676\nfor i, j in [ (39676, 44965 ), (39676, 53907 ), (44965, 53907 ), (53907, 51434) ]:\n    v = Y[i,:].toarray() - Y[j,:].toarray()\n    a = (Y[i,:].toarray() == Y[j,:].toarray()) &  ( ( Y[i,:].toarray() != 0  ) | ( Y[j,:].toarray() != 0  ) )\n    print( a.sum(), (v>0).sum(), (v<0).sum() , Y[i,:].toarray().sum(), Y[j,:].toarray().sum()  )","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:53.039313Z","iopub.execute_input":"2023-05-27T20:00:53.03989Z","iopub.status.idle":"2023-05-27T20:00:53.074221Z","shell.execute_reply.started":"2023-05-27T20:00:53.039853Z","shell.execute_reply":"2023-05-27T20:00:53.073119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i,j,k = (44965, 39676,  53907 )\n\nm = (Y[i,:].toarray() == Y[j,:].toarray()) &  (Y[i,:].toarray() == Y[k,:].toarray())  & (Y[i,:].toarray() != 0 ) \nprint(m.sum())\nprint()\n\nt = (44965, 39676,  53907 )\nfor s in [0,1,2] :\n    i,j,k = np.roll(t,s)\n    v = Y[i,:].toarray() + Y[j,:].toarray() - Y[k,:].toarray()\n    print( (v<0).sum(), df['organism'].iloc[i],'+',df['organism'].iloc[j],'-',df['organism'].iloc[k] , Y[k,:].toarray().sum())\nprint()\n\nt = (44965, 39676,  53907 )\nfor n in [50049 , 51434, 118522] :\n    i,j,k = np.roll(t,0)\n    v = Y[i,:].toarray() + Y[j,:].toarray() + Y[k,:].toarray() - Y[n,:].toarray()\n    print( (v<0).sum(), df['organism'].iloc[i],'+', df['organism'].iloc[j],'+',df['organism'].iloc[k], '-', df['organism'].iloc[n],  Y[n,:].toarray().sum() )\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:53.080936Z","iopub.execute_input":"2023-05-27T20:00:53.081347Z","iopub.status.idle":"2023-05-27T20:00:53.104934Z","shell.execute_reply.started":"2023-05-27T20:00:53.081316Z","shell.execute_reply":"2023-05-27T20:00:53.103725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ranks of target matrices for proteins from the same gene ","metadata":{}},{"cell_type":"code","source":"%%time\n\nlist_uv = df['gene name lower'].value_counts().index # unique()\nprint(len(list_uv), list_uv[:75])\n\nl = []\nfor uv in list_uv: \n    m = df['gene name lower'] ==  uv\n#     print(uv, m.sum() )\n    \n    l1 = df['index'][m]\n#     print(l1)\n#     print(uv, m.sum() ,  np.linalg.matrix_rank( Y[l1,:30].toarray()) )\n    r = np.linalg.matrix_rank( Y[l1,:10000].toarray())\n    # print(  r , m.sum() )\n    \n    l.append(r)\n    if m.sum() < 10: break\n\nprint(len(l),l)\nfor t in range(10):\n    m = np.array(l) <=t \n    print(t, m.sum() )\nprint( pd.Series(l).describe() )\nplt.hist(l, bins = 100 )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:00:53.10642Z","iopub.execute_input":"2023-05-27T20:00:53.106851Z","iopub.status.idle":"2023-05-27T20:01:42.152603Z","shell.execute_reply.started":"2023-05-27T20:00:53.106816Z","shell.execute_reply":"2023-05-27T20:01:42.151389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load distances train to train ","metadata":{}},{"cell_type":"code","source":"%%time\ndl = pd.read_csv('/kaggle/input/cafa5-data-selected/_Levenshtein_stat_data/df_Levenshtein_TOP100_distances_train_to_train.csv', index_col = 0)\ndl","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:42.154384Z","iopub.execute_input":"2023-05-27T20:01:42.154801Z","iopub.status.idle":"2023-05-27T20:01:51.935612Z","shell.execute_reply.started":"2023-05-27T20:01:42.154762Z","shell.execute_reply":"2023-05-27T20:01:51.934259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dl['index'] = range(len(dl))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:51.937703Z","iopub.execute_input":"2023-05-27T20:01:51.939015Z","iopub.status.idle":"2023-05-27T20:01:51.94541Z","shell.execute_reply.started":"2023-05-27T20:01:51.938961Z","shell.execute_reply":"2023-05-27T20:01:51.944122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"c = dl.columns[16]\nm = dl[c] == 0\nprint(m.sum() )\ndl[m]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:51.946872Z","iopub.execute_input":"2023-05-27T20:01:51.947224Z","iopub.status.idle":"2023-05-27T20:01:52.110715Z","shell.execute_reply.started":"2023-05-27T20:01:51.947195Z","shell.execute_reply":"2023-05-27T20:01:52.109329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dl.loc['P0C415',:].iloc[100:116]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.112171Z","iopub.execute_input":"2023-05-27T20:01:52.113434Z","iopub.status.idle":"2023-05-27T20:01:52.154113Z","shell.execute_reply.started":"2023-05-27T20:01:52.11339Z","shell.execute_reply":"2023-05-27T20:01:52.152635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v = dl[m]['index']\nv","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.156219Z","iopub.execute_input":"2023-05-27T20:01:52.156778Z","iopub.status.idle":"2023-05-27T20:01:52.174501Z","shell.execute_reply.started":"2023-05-27T20:01:52.156727Z","shell.execute_reply":"2023-05-27T20:01:52.173178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.linalg.matrix_rank(Y[v.values,:])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.175865Z","iopub.execute_input":"2023-05-27T20:01:52.177Z","iopub.status.idle":"2023-05-27T20:01:52.213427Z","shell.execute_reply.started":"2023-05-27T20:01:52.176955Z","shell.execute_reply":"2023-05-27T20:01:52.212006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dl.iloc[1,99]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.215114Z","iopub.execute_input":"2023-05-27T20:01:52.215498Z","iopub.status.idle":"2023-05-27T20:01:52.225009Z","shell.execute_reply.started":"2023-05-27T20:01:52.215466Z","shell.execute_reply":"2023-05-27T20:01:52.223587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dl.iloc[41,:5], dl.iloc[41,100:105], dl.iloc[91372,:5], dl.iloc[91372,100:105], ","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.226687Z","iopub.execute_input":"2023-05-27T20:01:52.227124Z","iopub.status.idle":"2023-05-27T20:01:52.244855Z","shell.execute_reply.started":"2023-05-27T20:01:52.227088Z","shell.execute_reply":"2023-05-27T20:01:52.243644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc['Q02338','sequence']  == df.loc['A0A384MTY4','sequence']  ","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.246957Z","iopub.execute_input":"2023-05-27T20:01:52.247348Z","iopub.status.idle":"2023-05-27T20:01:52.285915Z","shell.execute_reply.started":"2023-05-27T20:01:52.247314Z","shell.execute_reply":"2023-05-27T20:01:52.284024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y[41,:20].toarray()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.287777Z","iopub.execute_input":"2023-05-27T20:01:52.288184Z","iopub.status.idle":"2023-05-27T20:01:52.300314Z","shell.execute_reply.started":"2023-05-27T20:01:52.288143Z","shell.execute_reply":"2023-05-27T20:01:52.298898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y[91372,:20].toarray()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.302585Z","iopub.execute_input":"2023-05-27T20:01:52.303098Z","iopub.status.idle":"2023-05-27T20:01:52.316613Z","shell.execute_reply.started":"2023-05-27T20:01:52.303049Z","shell.execute_reply":"2023-05-27T20:01:52.315119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_A0A384MTY4 = 'MLATRLSRPLSRLPGKTLSACDRENGARRPLLLGSTSFIPIGRRTYASAAEPVGSKAVLVTGCDSGFGFSLAKHLHSKGFLVFAGCLMKDKGHDGVKELDSLNSDRLRTVQLNVCSSEEVEKVVEIVRSSLKDPEKGMWGLVNNAGISTFGEVEFTSLETYKQVAEVNLWGTVRMTKSFLPLIRRAKGRVVNISSMLGRMANPARSPYCITKFGVEAFSDCLRYEMYPLGVKVSVVEPGNFIAATSLYSPESIQAIAKKMWEELPEVVRKDYGKKYFDEKIAKMETYCSSGSTDTSPVIDAVTHALTATTPYTRYHPMDYYWWLRMQIMTHLPGAISDMIYIR'\ns_Q02338 =     'MLATRLSRPLSRLPGKTLSACDRENGARRPLLLGSTSFIPIGRRTYASAAEPVGSKAVLVTGCDSGFGFSLAKHLHSKGFLVFAGCLMKDKGHDGVKELDSLNSDRLRTVQLNVCSSEEVEKVVEIVRSSLKDPEKGMWGLVNNAGISTFGEVEFTSLETYKQVAEVNLWGTVRMTKSFLPLIRRAKGRVVNISSMLGRMANPARSPYCITKFGVEAFSDCLRYEMYPLGVKVSVVEPGNFIAATSLYSPESIQAIAKKMWEELPEVVRKDYGKKYFDEKIAKMETYCSSGSTDTSPVIDAVTHALTATTPYTRYHPMDYYWWLRMQIMTHLPGAISDMIYIR'\ns_A0A384MTY4 == s_Q02338","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.318888Z","iopub.execute_input":"2023-05-27T20:01:52.319426Z","iopub.status.idle":"2023-05-27T20:01:52.3301Z","shell.execute_reply.started":"2023-05-27T20:01:52.319382Z","shell.execute_reply":"2023-05-27T20:01:52.328555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count1 = 0\nfor I,id in enumerate(dl.index):\n    for J in range(100):\n        if dl.loc[id,:].iat[J] == 0:\n            id2 = dl.loc[id,:].iat[100+J]\n            I2 = dl.loc[id2,'index']\n            if (Y[I,:] - Y[I2,:]).sum()>0: # np.any(Y[I,:] != Y[J,:]):\n                count1 += 1\n                print(id,I,id2,I2, 'Diffr', (Y[I,:] - Y[J,:]).sum(), 'count:', count1 )","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:01:52.332006Z","iopub.execute_input":"2023-05-27T20:01:52.332439Z","iopub.status.idle":"2023-05-27T20:02:32.77142Z","shell.execute_reply.started":"2023-05-27T20:01:52.332404Z","shell.execute_reply":"2023-05-27T20:02:32.769439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}