{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nFeatures. In particular preparing one-hots from previously saved. \n\n## Versions \n\n#### 2  Keyword one-hot features -- feature set 07\n\n#### 1 Genes names related one-hot encoded features - feature set 06\n    Genes related one hot features \n    /kaggle/input/cafa5-data-selected/features/train_features06_genes_names_related_onehot.csv\n    /kaggle/input/cafa5-data-selected/features/test_features06_genes_names_related_onehot.csv    \n    /kaggle/input/cafa5-data-selected/embeds_43k_Y1850_etc/train_features06_genes_names_related_onehot_cut43k.csv","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time()\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns # there is no one for TPU machine \n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-09T18:54:44.027617Z","iopub.execute_input":"2023-07-09T18:54:44.028553Z","iopub.status.idle":"2023-07-09T18:54:45.925866Z","shell.execute_reply.started":"2023-07-09T18:54:44.028503Z","shell.execute_reply":"2023-07-09T18:54:45.9248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load preliminary Data","metadata":{}},{"cell_type":"code","source":"fn = '/kaggle/input/cafa5-data-selected/df_train_eda.csv'\ndf_train_eda = pd.read_csv(fn,index_col = 0)\nprint( df_train_eda.shape)\nlist_train_ids = list( df_train_eda.index)\nprint(len(list_train_ids), list_train_ids[:10])\ndisplay( df_train_eda.head(4) )\n\nfn = '/kaggle/input/cafa5-data-selected/df_test_eda.csv'\ndf_test_eda = pd.read_csv(fn,index_col = 0)\nprint( df_test_eda.shape)\nlist_test_ids = list( df_test_eda['EntryID'])\nprint(len(list_test_ids), list_test_ids[:10])\ndf_test_eda.head(4)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T18:59:31.763393Z","iopub.execute_input":"2023-07-09T18:59:31.76384Z","iopub.status.idle":"2023-07-09T18:59:36.424202Z","shell.execute_reply.started":"2023-07-09T18:59:31.763796Z","shell.execute_reply":"2023-07-09T18:59:36.422738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids_43k = np.load('/kaggle/input/cafa5-data-selected/embeds_43k_Y1850_etc/train_ids_cut43k.npy')\nprint(len(train_ids_43k), train_ids_43k[:10])\ndf43 = pd.DataFrame(index =train_ids_43k , data = range(len(train_ids_43k)), columns = ['IX'] )\nprint(df43.shape)\ndf43.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T20:36:59.294928Z","iopub.execute_input":"2023-07-09T20:36:59.295317Z","iopub.status.idle":"2023-07-09T20:36:59.330192Z","shell.execute_reply.started":"2023-07-09T20:36:59.295289Z","shell.execute_reply":"2023-07-09T20:36:59.329146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Keyword one-hot features -- feature set 07","metadata":{}},{"cell_type":"code","source":"%%time \n\nN_top_keywords_to_take = 100\n\n# fn = '/kaggle/input/cafa5-data-selected/features/train_features02_genes.csv'\n# /kaggle/input/cafa5-data-selected/features/train_features03_description.csv\n# /kaggle/input/cafa5-data-selected/features/test_features03_description.csv\nfn1 = '/kaggle/input/cafa5-data-selected/features/train_features03_description.csv'\nfn2 = '/kaggle/input/cafa5-data-selected/features/test_features03_description.csv'\n\ndict_df = {}\nfor k,fn in ([('train',fn1),('test',fn2) ]):\n    print(k,fn)\n    df = pd.read_csv(fn, index_col = 0)\n    dict_df[k] = df\n    print(df.shape)\n    display( df.head(3) )\n    \nv = pd.concat(dict_df.values(), axis =0 )[df.columns[0]]\nprint(v)\nl = []\nfor s in v:\n    l += str(s).lower().split(' ')\n\nv = pd.concat(dict_df.values(), axis =0 )[df.columns[0]]\nprint(v)\nl = []\nfor s in v:\n    l += str(s).lower().split(' ')\n\nsr = pd.Series(l).value_counts()\nprint(sr.head(10))\nlist_top_words = list( sr.index) # .head(50)\nlist_top_words = [ t for t in list_top_words if len(t) > 2 ]\nlist_top_words = [ t for t in list_top_words if not t.isdigit()  ]\nlist_top_words = [ t for t in list_top_words if t not in ['uncharacterized', 'homolog','protein',  'and', 'with', 'member', \n                                                         'family',  'isoform', 'cell', 'subfamily', 'systerm', ]  ]\nprint(len(list_top_words))\nprint(list_top_words[:100])\nprint()\nprint(list_top_words[100:150])\nprint()\nprint(list_top_words[150:250])\n\nm = sr.index.isin(list_top_words)\nsr[m].head(50)\nprint( (sr[m] > 100).sum() )\n\nfor k in dict_df:\n    df = dict_df[k]\n    v = df['Description Cleaned']\n    df['Set'] = [set(str(s).lower().split(' ') ) for s in v ]\n\ndict_df_feat = {}\nfor k in dict_df:\n    df = dict_df[k]\n    dict_df_feat[k] = pd.DataFrame() \n    v = df['Description Cleaned']\n    df['Set'] = [set(str(s).lower().split(' ') ) for s in v ]\n    for w in list_top_words[:N_top_keywords_to_take]:\n        l  = [w in t for t in df['Set'] ]\n        dict_df_feat[k][w] = np.array(l).astype(np.float32)\n        \nfor k in dict_df:\n    df = dict_df_feat[k]\n    if k == 'train':\n        df.index = list_train_ids\n    elif k == 'test':\n        df.index = list_test_ids\n\nfor k in dict_df:\n    df = dict_df_feat[k]\n    print()\n    print(k, df.shape)\n    display(df)\n    display(df.describe())\n    plt.plot(df.describe().loc['mean',:], label = k)\nplt.legend()\nplt.grid()\nplt.show()    \n\nfor k in dict_df:\n    df = dict_df_feat[k]\n    fn =  k+'_features07_keywords_onehot.csv' \n    print()\n    print(k,fn, df.shape)\n    df.to_csv(fn)\n\ndf = dict_df_feat['train']\nm = df.index.isin( train_ids_43k )\nprint(m.sum() )\ndf[m].to_csv('train_features07_keywords_onehot_cut43k.csv' )\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T20:47:31.075971Z","iopub.execute_input":"2023-07-09T20:47:31.076486Z","iopub.status.idle":"2023-07-09T20:48:17.2031Z","shell.execute_reply.started":"2023-07-09T20:47:31.076442Z","shell.execute_reply":"2023-07-09T20:48:17.201969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Genes names related one-hot encoded features - feature set 06","metadata":{}},{"cell_type":"code","source":"%%time\n\nfn = '/kaggle/input/cafa5-data-selected/features/train_features02_genes.csv'\ndf = pd.read_csv(fn, index_col = 0)\nprint(df.shape)\ndisplay( df.head(3) )\nv0train = df[df.columns[1]]\nvtrain = df[df.columns[1]].value_counts()\nprint( vtrain.head(20) )\nplt.plot(vtrain.values[:1000])\nplt.show()\ndf1 = df.copy()\n\nfn = '/kaggle/input/cafa5-data-selected/features/test_features02_genes.csv'\ndf = pd.read_csv(fn, index_col = 0)\nprint(df.shape)\ndisplay( df.head(3 ) )\nv0 = df[df.columns[1]]\nv = df[df.columns[1]].value_counts()\nprint( v.head(50) )\nplt.plot(v.values[:1000])\nplt.show()\ndf2 = df.copy()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:13:10.424281Z","iopub.execute_input":"2023-07-09T17:13:10.424721Z","iopub.status.idle":"2023-07-09T17:13:10.904244Z","shell.execute_reply.started":"2023-07-09T17:13:10.424688Z","shell.execute_reply":"2023-07-09T17:13:10.903228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Top 50 genes names present in both train and test ","metadata":{}},{"cell_type":"code","source":"m = v0train.isin(v.head(200).index)\n_t = v0train[m].value_counts().head(50)\nprint(_t)\nlist_top_frequent_genes_names = list(_t.index)\nprint(len(list_top_frequent_genes_names), list_top_frequent_genes_names)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:13:10.905595Z","iopub.execute_input":"2023-07-09T17:13:10.905912Z","iopub.status.idle":"2023-07-09T17:13:10.927922Z","shell.execute_reply.started":"2023-07-09T17:13:10.905885Z","shell.execute_reply":"2023-07-09T17:13:10.926737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf = pd.DataFrame(index = list_train_ids)\nfor t in list_top_frequent_genes_names:\n    df[t] =  (df1['gene name lower'] == t).values.astype(np.float32)\n    \nprint(df.shape)    \ndisplay( df.head() )\ndisplay( df.describe() )\n\ndf_train_feat = df.copy()\n\n\ndf = pd.DataFrame(index = list_test_ids)\n\nfor t in list_top_frequent_genes_names:\n    df[t] =  (df2['gene name lower'] == t).values.astype(np.float32)\n    \nprint(df.shape)    \ndisplay( df.head() )\ndisplay( df.describe() )\n\ndf_test_feat = df.copy()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:13:10.92982Z","iopub.execute_input":"2023-07-09T17:13:10.930175Z","iopub.status.idle":"2023-07-09T17:13:14.365375Z","shell.execute_reply.started":"2023-07-09T17:13:10.930137Z","shell.execute_reply":"2023-07-09T17:13:14.36457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gene names without digits ","metadata":{}},{"cell_type":"code","source":"%%time\nimport re\n\ndef delete_digits_at_end(string):\n    pattern = r'\\d+$'  # Matches one or more digits at the end of the string\n    return re.sub(pattern, '', string)\n\nl = [delete_digits_at_end(t) for t in v0]\nv2 = pd.Series(l).value_counts()\nplt.plot(v2.values[:1000])\nplt.show()\nprint( v2.head(50) )\nprint(v2.iloc[50:100])\nl_train = [delete_digits_at_end(str(t)) for t in v0train ]\nsr = pd.Series(l_train)\nm = sr.isin(l)\ndisplay( sr[m].value_counts().head(50) )\ndisplay( sr[m].value_counts().iloc[50:100] )\n\nlist_genes_names_without_digits = list( sr[m].value_counts().index[:100] )\nprint(len(list_genes_names_without_digits ), list_genes_names_without_digits[:10])\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:13:14.366456Z","iopub.execute_input":"2023-07-09T17:13:14.367158Z","iopub.status.idle":"2023-07-09T17:13:15.588113Z","shell.execute_reply.started":"2023-07-09T17:13:14.367121Z","shell.execute_reply":"2023-07-09T17:13:15.586943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nprint( len( list_genes_names_without_digits ) , list_genes_names_without_digits[:30] )\n\nl = [delete_digits_at_end(str(t))  for t in df1['gene name lower'] ]\nsr_loc = pd.Series(l)\ndf = pd.DataFrame(index = list_train_ids)\nfor t in list_genes_names_without_digits:\n    dt = (sr_loc  == t).values.astype(np.float32)\n    if t != '':\n        df[t] =  dt\n    else:\n        df['EMPTY'] = dt\n    if dt.sum() == 0: print('Warning - zero sum ')\n        \nprint(df.shape)    \ndisplay( df.head() )\ndisplay( df.describe() )\n\n#df_train_feat = df.copy()\ndf_train_feat = pd.concat([ df_train_feat,df],axis = 1) #  df.copy()\n\n\ndf = pd.DataFrame(index = list_test_ids)\n\nl = [delete_digits_at_end(t) for t in df2['gene name lower'] ]\nsr_loc = pd.Series(l)\nfor t in list_genes_names_without_digits:\n    dt = (sr_loc  == t).values.astype(np.float32)\n    if t != '':\n        df[t] =  dt\n    else:\n        df['EMPTY'] = dt\n    if dt.sum() == 0: print(t, 'Warning - zero sum ')\n    \nprint(df.shape)    \ndisplay( df.head() )\ndisplay( df.describe() )\n\ndf_test_feat = pd.concat([ df_test_feat,df],axis = 1) #  df.copy()\n\n\nsns.clustermap(df_train_feat.corr().round(2) )\nplt.show()\nsns.clustermap(df_test_feat.corr().round(2) )\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:13:15.589469Z","iopub.execute_input":"2023-07-09T17:13:15.589798Z","iopub.status.idle":"2023-07-09T17:13:43.08677Z","shell.execute_reply.started":"2023-07-09T17:13:15.589771Z","shell.execute_reply":"2023-07-09T17:13:43.08591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Special gene names ","metadata":{}},{"cell_type":"code","source":"'ywha' in df_train_feat.columns, 'znf' in df_train_feat.columns, ","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:13:43.089912Z","iopub.execute_input":"2023-07-09T17:13:43.090763Z","iopub.status.idle":"2023-07-09T17:13:43.097988Z","shell.execute_reply.started":"2023-07-09T17:13:43.09073Z","shell.execute_reply":"2023-07-09T17:13:43.096627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nlist_current = ['mt-', 'ywha']\n\nprint( len( list_current ) , list_current[:30] )\n\nsr_loc = df1['gene name lower'] \ndf = pd.DataFrame(index = list_train_ids)\nfor symb in list_current:\n    l_ = [str(t).startswith(symb) for t in sr_loc]\n    dt = (pd.Series(l_)).values.astype(np.float32)\n    df[symb] =  dt\n    if dt.sum() == 0: print(symb, 'Warning - zero sum ')\n        \nprint(df.shape)    \ndisplay( df.head() )\ndisplay( df.describe() )\n\ndf_train_feat = pd.concat([ df_train_feat,df],axis = 1) #  df.copy()\n\nsns.clustermap(df.corr().round(2) )\nplt.show()\n\ndf = pd.DataFrame(index = list_test_ids)\n\nsr_loc = df2['gene name lower'] \ndf = pd.DataFrame(index = list_test_ids)\nfor symb in list_current:\n    l_ = [str(t).startswith(symb) for t in sr_loc]\n    dt = (pd.Series(l_)).values.astype(np.float32)\n    df[symb] =  dt\n    if dt.sum() == 0: print(symb, 'Warning - zero sum ')\n    \nprint(df.shape)    \ndisplay( df.head() )\ndisplay( df.describe() )\n\nsns.clustermap(df.corr().round(2) )\nplt.show()\n\ndf_test_feat = pd.concat([ df_test_feat,df],axis = 1) #  df.copy()\n\n\nsns.clustermap(df_train_feat.corr().round(2) )\nplt.show()\nsns.clustermap(df_test_feat.corr().round(2) )\nplt.show()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:13:43.099349Z","iopub.execute_input":"2023-07-09T17:13:43.099788Z","iopub.status.idle":"2023-07-09T17:14:04.921705Z","shell.execute_reply.started":"2023-07-09T17:13:43.099756Z","shell.execute_reply":"2023-07-09T17:14:04.920696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nlist_current = ['atp']\n\nprint( len( list_current ) , list_current[:30] )\n\nsr_loc = df1['gene name lower'] \ndf = pd.DataFrame(index = list_train_ids)\nfor symb in list_current:\n    l_ = [symb in str(t) for t in sr_loc]\n    dt = (pd.Series(l_)).values.astype(np.float32)\n    df[symb] =  dt\n    if dt.sum() == 0: print(symb, 'Warning - zero sum ')\n        \nprint(df.shape)    \ndisplay( df.head() )\ndisplay( df.describe() )\n\ndf_train_feat = pd.concat([ df_train_feat,df],axis = 1) #  df.copy()\n\n# sns.clustermap(df.corr().round(2) )\n# plt.show()\n\ndf = pd.DataFrame(index = list_test_ids)\n\nsr_loc = df2['gene name lower'] \ndf = pd.DataFrame(index = list_test_ids)\nfor symb in list_current:\n    l_ = [symb in str(t) for t in sr_loc]\n    dt = (pd.Series(l_)).values.astype(np.float32)\n    df[symb] =  dt\n    if dt.sum() == 0: print(symb, 'Warning - zero sum ')\n    \nprint(df.shape)    \ndisplay( df.head() )\ndisplay( df.describe() )\n\n# sns.clustermap(df.corr().round(2) )\n# plt.show()\n\ndf_test_feat = pd.concat([ df_test_feat,df],axis = 1) #  df.copy()\n\n\nsns.clustermap(df_train_feat.corr().round(2) )\nplt.show()\nsns.clustermap(df_test_feat.corr().round(2) )\nplt.show()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:14:04.923024Z","iopub.execute_input":"2023-07-09T17:14:04.923325Z","iopub.status.idle":"2023-07-09T17:14:26.156632Z","shell.execute_reply.started":"2023-07-09T17:14:04.923299Z","shell.execute_reply":"2023-07-09T17:14:26.155512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save results","metadata":{}},{"cell_type":"code","source":"print(df_train_feat.shape)\nprint(df_test_feat.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:14:26.158155Z","iopub.execute_input":"2023-07-09T17:14:26.158561Z","iopub.status.idle":"2023-07-09T17:14:26.164823Z","shell.execute_reply.started":"2023-07-09T17:14:26.158527Z","shell.execute_reply":"2023-07-09T17:14:26.163929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_train_feat.to_csv('train_features06_genes_names_related_onehot.csv')\ndf_test_feat.to_csv('test_features06_genes_names_related_onehot.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:14:26.165981Z","iopub.execute_input":"2023-07-09T17:14:26.167056Z","iopub.status.idle":"2023-07-09T17:14:58.477472Z","shell.execute_reply.started":"2023-07-09T17:14:26.167025Z","shell.execute_reply":"2023-07-09T17:14:58.476499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save for 43k","metadata":{}},{"cell_type":"code","source":"m = df_train_feat.index.isin( train_ids_43k )\nprint(m.sum() )\ndf_train_feat[m].to_csv('train_features06_genes_names_related_onehot_cut43k.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:14:58.478672Z","iopub.execute_input":"2023-07-09T17:14:58.478972Z","iopub.status.idle":"2023-07-09T17:15:03.497988Z","shell.execute_reply.started":"2023-07-09T17:14:58.478947Z","shell.execute_reply":"2023-07-09T17:15:03.49683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_feat[m].describe()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:19:31.855217Z","iopub.execute_input":"2023-07-09T17:19:31.856114Z","iopub.status.idle":"2023-07-09T17:19:32.418265Z","shell.execute_reply.started":"2023-07-09T17:19:31.856068Z","shell.execute_reply":"2023-07-09T17:19:32.417199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v_ = df_train_feat[m].sum(axis = 0)\nplt.plot(v.values)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:19:57.512836Z","iopub.execute_input":"2023-07-09T17:19:57.51326Z","iopub.status.idle":"2023-07-09T17:19:57.79249Z","shell.execute_reply.started":"2023-07-09T17:19:57.513227Z","shell.execute_reply":"2023-07-09T17:19:57.791395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(v_ == 0).sum(), (v_ >1).sum(), (v_ == 2).sum(),  (v_ < 10).sum()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:21:13.319256Z","iopub.execute_input":"2023-07-09T17:21:13.319626Z","iopub.status.idle":"2023-07-09T17:21:13.328515Z","shell.execute_reply.started":"2023-07-09T17:21:13.319597Z","shell.execute_reply":"2023-07-09T17:21:13.327378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# /kaggle/input/cafa5-data-selected/features/train_features01_SV_PE.csv\n# /kaggle/input/cafa5-data-selected/features/test_features01_SV_PE.npy\n# /kaggle/input/cafa5-data-selected/features/train_features01_SV_PE.npy\n# /kaggle/input/cafa5-data-selected/features/train_features02_genes.csv\n# /kaggle/input/cafa5-data-selected/features/test_features05_taxons31_onehot.csv\n# /kaggle/input/cafa5-data-selected/features/test_features02_genes.npy\n# /kaggle/input/cafa5-data-selected/features/train_features03_description.csv\n# /kaggle/input/cafa5-data-selected/features/train_features02_genes.npy\n# /kaggle/input/cafa5-data-selected/features/test_features03_description.csv\n# /kaggle/input/cafa5-data-selected/features/train_features03_description.npy\n# /kaggle/input/cafa5-data-selected/features/test_features04_dbase_fragment.csv\n# /kaggle/input/cafa5-data-selected/features/train_features04_dbase_fragment.csv\n# /kaggle/input/cafa5-data-selected/features/train_features05_taxons31_onehot.csv\n# /kaggle/input/cafa5-data-selected/features/test_features04_dbase_fragment.npy\n# /kaggle/input/cafa5-data-selected/features/test_features02_genes.csv\n# /kaggle/input/cafa5-data-selected/features/test_features01_SV_PE.csv\n# /kaggle/input/cafa5-data-selected/features/train_features04_dbase_fragment.npy\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T17:15:03.499845Z","iopub.execute_input":"2023-07-09T17:15:03.500284Z","iopub.status.idle":"2023-07-09T17:15:03.506256Z","shell.execute_reply.started":"2023-07-09T17:15:03.500246Z","shell.execute_reply":"2023-07-09T17:15:03.505134Z"},"trusted":true},"execution_count":null,"outputs":[]}]}