{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import roc_auc_score\n\nimport numpy as np\nimport pandas as pd\n\nfrom tqdm import tqdm\ntqdm.pandas()\n\nfrom keras.models import Sequential\nfrom keras.layers import Dense\n# measure roc auc score metric \nfrom tensorflow.keras.metrics import AUC","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.290929,"end_time":"2023-04-28T16:11:40.612349","exception":false,"start_time":"2023-04-28T16:11:39.32142","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:31:10.760289Z","iopub.execute_input":"2023-05-17T23:31:10.761153Z","iopub.status.idle":"2023-05-17T23:31:19.016653Z","shell.execute_reply.started":"2023-05-17T23:31:10.761115Z","shell.execute_reply":"2023-05-17T23:31:19.015582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Assigning labels","metadata":{"papermill":{"duration":0.004422,"end_time":"2023-04-28T16:11:40.621902","exception":false,"start_time":"2023-04-28T16:11:40.61748","status":"completed"},"tags":[]}},{"cell_type":"code","source":"DATA_DIR = '/kaggle/input/cafa-5-protein-function-prediction'\nMAX_LABELS = 500","metadata":{"papermill":{"duration":0.01492,"end_time":"2023-04-28T16:11:40.6416","exception":false,"start_time":"2023-04-28T16:11:40.62668","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:31:19.018485Z","iopub.execute_input":"2023-05-17T23:31:19.019808Z","iopub.status.idle":"2023-05-17T23:31:19.024551Z","shell.execute_reply.started":"2023-05-17T23:31:19.01977Z","shell.execute_reply":"2023-05-17T23:31:19.023587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(os.path.join(DATA_DIR, 'Train', 'train_terms.tsv'), sep='\\t')\n\nterms = train_terms.groupby(['aspect', 'term'])['term'].count().reset_index(name='frequency')\nprint(terms.groupby('aspect')['term'].nunique())","metadata":{"papermill":{"duration":5.386139,"end_time":"2023-04-28T16:11:46.032438","exception":false,"start_time":"2023-04-28T16:11:40.646299","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:31:19.025762Z","iopub.execute_input":"2023-05-17T23:31:19.026428Z","iopub.status.idle":"2023-05-17T23:31:23.957129Z","shell.execute_reply.started":"2023-05-17T23:31:19.026397Z","shell.execute_reply":"2023-05-17T23:31:23.956087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fractions = (terms.groupby('aspect')['term'].nunique() / terms['term'].nunique() * MAX_LABELS).apply(round)\nprint(fractions)\n\nselected_terms = set()\nfor aspect, number in fractions.items():\n    selection = terms.loc[(terms.aspect == aspect)]\n    selection = selection.nlargest(number, columns='frequency', keep='first')\n    selected_terms.update(selection.term.to_list())","metadata":{"papermill":{"duration":0.07447,"end_time":"2023-04-28T16:11:46.112388","exception":false,"start_time":"2023-04-28T16:11:46.037918","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:31:23.960234Z","iopub.execute_input":"2023-05-17T23:31:23.960686Z","iopub.status.idle":"2023-05-17T23:31:24.008897Z","shell.execute_reply.started":"2023-05-17T23:31:23.960659Z","shell.execute_reply":"2023-05-17T23:31:24.007898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(selected_terms)","metadata":{"papermill":{"duration":0.016051,"end_time":"2023-04-28T16:11:46.133324","exception":false,"start_time":"2023-04-28T16:11:46.117273","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:31:24.010407Z","iopub.execute_input":"2023-05-17T23:31:24.010752Z","iopub.status.idle":"2023-05-17T23:31:24.016236Z","shell.execute_reply.started":"2023-05-17T23:31:24.01072Z","shell.execute_reply":"2023-05-17T23:31:24.014972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def assign_labels(annotations, selected_terms=selected_terms):\n    \n    intersection = selected_terms.intersection(annotations)\n    labels = np.isin(np.array(list(selected_terms)), np.array(list(intersection)))\n    \n    return list(labels.astype('int'))\n\nannotations = train_terms.groupby('EntryID')['term'].apply(set)\nlabels = annotations.progress_apply(assign_labels)\n\nlabels.head()","metadata":{"papermill":{"duration":141.942432,"end_time":"2023-04-28T16:14:08.08098","exception":false,"start_time":"2023-04-28T16:11:46.138548","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:31:24.01808Z","iopub.execute_input":"2023-05-17T23:31:24.018427Z","iopub.status.idle":"2023-05-17T23:32:25.318546Z","shell.execute_reply.started":"2023-05-17T23:31:24.018396Z","shell.execute_reply":"2023-05-17T23:32:25.317745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading train embeddings","metadata":{"papermill":{"duration":0.082986,"end_time":"2023-04-28T16:14:08.247278","exception":false,"start_time":"2023-04-28T16:14:08.164292","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\n\nx_train = np.load('/kaggle/input/t5embeds/train_embeds.npy')\ny_train = np.array(labels[train_ids].to_list())","metadata":{"papermill":{"duration":20.577499,"end_time":"2023-04-28T16:14:28.907666","exception":false,"start_time":"2023-04-28T16:14:08.330167","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:32:25.322518Z","iopub.execute_input":"2023-05-17T23:32:25.329098Z","iopub.status.idle":"2023-05-17T23:32:41.003659Z","shell.execute_reply.started":"2023-05-17T23:32:25.32907Z","shell.execute_reply":"2023-05-17T23:32:41.00271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{"papermill":{"duration":0.083305,"end_time":"2023-04-28T16:14:29.076391","exception":false,"start_time":"2023-04-28T16:14:28.993086","status":"completed"},"tags":[]}},{"cell_type":"code","source":"x_train, x_valid, y_train, y_valid = train_test_split(x_train, y_train, shuffle=True, random_state=42)","metadata":{"papermill":{"duration":1.175161,"end_time":"2023-04-28T16:14:30.339394","exception":false,"start_time":"2023-04-28T16:14:29.164233","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:32:41.005081Z","iopub.execute_input":"2023-05-17T23:32:41.005421Z","iopub.status.idle":"2023-05-17T23:32:41.541099Z","shell.execute_reply.started":"2023-05-17T23:32:41.00539Z","shell.execute_reply":"2023-05-17T23:32:41.540119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build a simple MLP model in Keras with ReLU activation and nothing else\nnfeats = x_train.shape[1]\nnlabels = y_train.shape[1]\nmodel = Sequential()\nmodel.add(Dense(256, activation='relu', input_dim=nfeats))\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dense(nlabels, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy',\n                optimizer='adam',\n                metrics=[AUC()])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T23:32:41.542725Z","iopub.execute_input":"2023-05-17T23:32:41.543128Z","iopub.status.idle":"2023-05-17T23:32:43.967686Z","shell.execute_reply.started":"2023-05-17T23:32:41.543091Z","shell.execute_reply":"2023-05-17T23:32:43.966947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(x_train, y_train, epochs=15, batch_size=128, validation_data=(x_valid, y_valid))","metadata":{"papermill":{"duration":6.259531,"end_time":"2023-04-28T16:14:36.681742","exception":false,"start_time":"2023-04-28T16:14:30.422211","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:32:43.970938Z","iopub.execute_input":"2023-05-17T23:32:43.971288Z","iopub.status.idle":"2023-05-17T23:34:08.607893Z","shell.execute_reply.started":"2023-05-17T23:32:43.971253Z","shell.execute_reply":"2023-05-17T23:34:08.606874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat = model.predict(x_valid)\n\nscores = pd.DataFrame(columns=list(selected_terms), index=['roc_auc'])\n\nfor i, term in enumerate(selected_terms):\n    score = roc_auc_score(y_valid[:, i], y_hat[:, i])\n    scores[term] = score\n\nscores.mean(axis=1)","metadata":{"papermill":{"duration":20.422772,"end_time":"2023-04-28T16:14:57.256109","exception":false,"start_time":"2023-04-28T16:14:36.833337","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:34:30.191773Z","iopub.execute_input":"2023-05-17T23:34:30.192259Z","iopub.status.idle":"2023-05-17T23:34:39.203669Z","shell.execute_reply.started":"2023-05-17T23:34:30.19222Z","shell.execute_reply":"2023-05-17T23:34:39.20277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{"papermill":{"duration":0.08303,"end_time":"2023-04-28T16:14:57.422033","exception":false,"start_time":"2023-04-28T16:14:57.339003","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\nx_test = np.load('/kaggle/input/t5embeds/test_embeds.npy')","metadata":{"papermill":{"duration":6.254099,"end_time":"2023-04-28T16:15:03.759386","exception":false,"start_time":"2023-04-28T16:14:57.505287","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:34:57.444331Z","iopub.execute_input":"2023-05-17T23:34:57.44469Z","iopub.status.idle":"2023-05-17T23:35:08.131936Z","shell.execute_reply.started":"2023-05-17T23:34:57.444662Z","shell.execute_reply":"2023-05-17T23:35:08.130895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del x_train, y_train, x_valid, y_valid, labels\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(x_test)\ndel x_test\ngc.collect()\n\nchunk_size = 5_000\nchunks = [range(i, min(i + chunk_size, len(predictions))) for i in range(0, len(predictions), chunk_size)]\n\nfinal_sub = pd.DataFrame()  # Create an empty DataFrame to hold the final result\n\nprint(f\"processing {len(chunks)} chunks of {chunk_size} predictions each\")\n\nfor chunk in chunks:\n    print(f\"processing chunk {chunk}\")\n    sub = pd.DataFrame(data=predictions[chunk], columns=list(selected_terms), index=test_ids[chunk])\n    sub = sub.T.unstack().reset_index(name='prediction')\n    sub = sub.loc[sub['prediction'] > 0]\n    final_sub = pd.concat([final_sub, sub])  # Concatenate current chunk DataFrame to the final DataFrame\n\nfinal_sub.head()","metadata":{"papermill":{"duration":48.344071,"end_time":"2023-04-28T16:15:52.186228","exception":false,"start_time":"2023-04-28T16:15:03.842157","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-05-17T23:35:08.133872Z","iopub.execute_input":"2023-05-17T23:35:08.134254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_sub.to_csv('submission.tsv', sep='\\t', index=False, header=False)","metadata":{"papermill":{"duration":296.790114,"end_time":"2023-04-28T16:20:49.060174","exception":false,"start_time":"2023-04-28T16:15:52.27006","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]}]}