{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom sklearn.linear_model import LogisticRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-22T21:41:03.608075Z","iopub.execute_input":"2023-01-22T21:41:03.608394Z","iopub.status.idle":"2023-01-22T21:41:03.713763Z","shell.execute_reply.started":"2023-01-22T21:41:03.608367Z","shell.execute_reply":"2023-01-22T21:41:03.713055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = '/kaggle/input/rsna-breast-cancer-detection'","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:41:04.148523Z","iopub.execute_input":"2023-01-22T21:41:04.148821Z","iopub.status.idle":"2023-01-22T21:41:04.152672Z","shell.execute_reply.started":"2023-01-22T21:41:04.148797Z","shell.execute_reply":"2023-01-22T21:41:04.151877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create the feature dataset with meta data features","metadata":{}},{"cell_type":"code","source":"def make_dataset(is_train=True):\n    # load metadata file\n    if is_train:\n        meta_df = pd.read_csv(f'{data_path}/train.csv')\n    else:\n        meta_df = pd.read_csv(f'{data_path}/test.csv')\n\n    # replace NaN age with interger mean age\n    meta_df['age'] = meta_df['age'].replace(to_replace=np.nan, value=int(meta_df['age'].mean()))\n\n    # remove views with too few samples\n    valid_views = {'MLO', 'CC'}\n    is_valid_view = meta_df['view'].isin(valid_views)\n    meta_df = meta_df.loc[is_valid_view, :]\n    meta_df['view'] = meta_df['view'].map({'MLO': 0, 'CC': 1})\n    \n    # map laterality to integers\n    meta_df['laterality'] = meta_df['laterality'].map({'L': 0, 'R': 1})\n    \n    # retain only relevant columns\n    if is_train:\n        meta_df = meta_df[['patient_id', 'age', 'laterality', 'cancer']]\n    else:\n        meta_df = meta_df[['patient_id', 'image_id', 'age', 'laterality', 'prediction_id']]\n\n    # group by patient id\n    meta_df = meta_df.groupby('patient_id').agg(list)\n\n    # finalize df\n    if is_train:\n        meta_df = meta_df.explode(['age', 'laterality', 'cancer'])\n        meta_df = meta_df.reset_index()\n        meta_df = meta_df.drop(['patient_id'], axis=1)\n        meta_df = meta_df.astype(int)\n#         meta_df = meta_df.drop_duplicates()\n    else:\n        meta_df = meta_df.explode(['image_id', 'age', 'laterality', 'prediction_id'])\n        meta_df = meta_df.reset_index()\n        meta_df['image_path'] = meta_df[['patient_id', 'image_id']].apply(\n            lambda x: f'{data_path}/test_images/{x[0]}/{x[1]}.dcm', axis=1\n        )\n        meta_df = meta_df.drop(['patient_id', 'image_id'], axis=1)\n    \n    return meta_df","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:24.10875Z","iopub.execute_input":"2023-01-22T21:50:24.109088Z","iopub.status.idle":"2023-01-22T21:50:24.120074Z","shell.execute_reply.started":"2023-01-22T21:50:24.109063Z","shell.execute_reply":"2023-01-22T21:50:24.118429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = make_dataset(is_train=True)\ntest_df = make_dataset(is_train=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:24.330061Z","iopub.execute_input":"2023-01-22T21:50:24.330389Z","iopub.status.idle":"2023-01-22T21:50:24.706644Z","shell.execute_reply.started":"2023-01-22T21:50:24.330364Z","shell.execute_reply":"2023-01-22T21:50:24.705813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### NOTE: Here we would add the training image predictions to the train_df when we have them, as an additional feature.","metadata":{}},{"cell_type":"code","source":"# load training set predictions\n# train_df['img_pred'] = train_image_preds\n\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:25.775453Z","iopub.execute_input":"2023-01-22T21:50:25.775772Z","iopub.status.idle":"2023-01-22T21:50:25.784943Z","shell.execute_reply.started":"2023-01-22T21:50:25.775747Z","shell.execute_reply":"2023-01-22T21:50:25.78407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:26.064407Z","iopub.execute_input":"2023-01-22T21:50:26.064716Z","iopub.status.idle":"2023-01-22T21:50:26.074479Z","shell.execute_reply.started":"2023-01-22T21:50:26.064692Z","shell.execute_reply":"2023-01-22T21:50:26.073578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### NOTE: Here we will get the test set image predictions and add them as an additional feature","metadata":{}},{"cell_type":"code","source":"# def load_dicom(image_path):\n#     # load in dicom image\n#     image_bytes = tf.io.read_file(image_path)\n#     # load in image array\n#     image = tfio.image.decode_dicom_image(\n#         image_bytes, \n#         dtype=tf.uint16,\n#         on_error='lossy')\n    \n#     return image[0]\n\n# def pipeline(sample):\n#     # load in dicom image\n#     image_path = sample['image_path']\n#     image = load_dicom(image_path)\n\n#     return image\n\n# # create tensorflow datasets\n# test_ds = tf.data.Dataset.from_tensor_slices(dict(test_df))\n# processed_test_ds = test_ds.map(pipeline, num_parallel_calls=2)\n\n# def load_trained_model(path):\n#     model = tf.keras.models.load_model(path)\n#     return model\n\n# model_path = '/kaggle/input/rsna-trained-models/best_model.h5'\n# model = load_trained_model(model_path)\n\n# batched_test_ds = processed_test_ds.batch(1, num_parallel_calls=tf.data.AUTOTUNE).prefetch(10)\n# test_img_preds = model.predict(batched_test_ds)\n\n# test_img_preds = np.squeeze(test_img_preds)\n\n# test_df['img_pred'] = test_img_preds","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:31.263916Z","iopub.execute_input":"2023-01-22T21:50:31.264218Z","iopub.status.idle":"2023-01-22T21:50:31.268744Z","shell.execute_reply.started":"2023-01-22T21:50:31.264195Z","shell.execute_reply":"2023-01-22T21:50:31.267994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Here we will train the tabular classifier","metadata":{}},{"cell_type":"code","source":"train_X = train_df.drop(['cancer'], axis=1).values\ntrain_y = train_df['cancer'].values\n\ntest_X = test_df.drop(['prediction_id', 'image_path'], axis=1).values","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:33.511316Z","iopub.execute_input":"2023-01-22T21:50:33.511673Z","iopub.status.idle":"2023-01-22T21:50:33.518443Z","shell.execute_reply.started":"2023-01-22T21:50:33.511643Z","shell.execute_reply":"2023-01-22T21:50:33.51759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = LogisticRegression(max_iter=500)\nclf.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:36.723309Z","iopub.execute_input":"2023-01-22T21:50:36.723612Z","iopub.status.idle":"2023-01-22T21:50:36.824976Z","shell.execute_reply.started":"2023-01-22T21:50:36.723589Z","shell.execute_reply":"2023-01-22T21:50:36.824013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = clf.predict_proba(test_X)\ntest_pred = test_pred[:, 1]","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:38.399205Z","iopub.execute_input":"2023-01-22T21:50:38.399514Z","iopub.status.idle":"2023-01-22T21:50:38.40384Z","shell.execute_reply.started":"2023-01-22T21:50:38.399491Z","shell.execute_reply":"2023-01-22T21:50:38.403055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Here we will find the cutoff resulting in maximum pfbeta for the training data","metadata":{}},{"cell_type":"code","source":"def pfbeta_np(labels, preds, beta=1):\n    preds = preds.clip(0, 1)\n    y_true_count = labels.sum()\n    ctp = preds[labels==1].sum()\n    cfp = preds[labels==0].sum()\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0.0","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:41.944127Z","iopub.execute_input":"2023-01-22T21:50:41.944446Z","iopub.status.idle":"2023-01-22T21:50:41.950216Z","shell.execute_reply.started":"2023-01-22T21:50:41.944422Z","shell.execute_reply":"2023-01-22T21:50:41.949404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred = clf.predict_proba(train_X)\ntrain_pred = train_pred[:, 1]","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:43.22625Z","iopub.execute_input":"2023-01-22T21:50:43.226836Z","iopub.status.idle":"2023-01-22T21:50:43.232252Z","shell.execute_reply.started":"2023-01-22T21:50:43.2268Z","shell.execute_reply":"2023-01-22T21:50:43.231546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(train_pred)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:44.164757Z","iopub.execute_input":"2023-01-22T21:50:44.165062Z","iopub.status.idle":"2023-01-22T21:50:44.327818Z","shell.execute_reply.started":"2023-01-22T21:50:44.165039Z","shell.execute_reply":"2023-01-22T21:50:44.327099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_pfbeta = []\nfor i in range(1, 100):\n    cutoff = np.percentile(train_pred, i)\n    train_pred_bin = (train_pred > cutoff).astype(float)\n    pfbeta = pfbeta_np(train_y, train_pred_bin)\n    all_pfbeta.append(pfbeta)\nall_pfbeta = np.array(all_pfbeta)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:46.30026Z","iopub.execute_input":"2023-01-22T21:50:46.300552Z","iopub.status.idle":"2023-01-22T21:50:46.369791Z","shell.execute_reply.started":"2023-01-22T21:50:46.300529Z","shell.execute_reply":"2023-01-22T21:50:46.369143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(all_pfbeta)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:47.37818Z","iopub.execute_input":"2023-01-22T21:50:47.378496Z","iopub.status.idle":"2023-01-22T21:50:47.512877Z","shell.execute_reply.started":"2023-01-22T21:50:47.378471Z","shell.execute_reply":"2023-01-22T21:50:47.512201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_pfbeta.argmax()","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:50.531214Z","iopub.execute_input":"2023-01-22T21:50:50.531523Z","iopub.status.idle":"2023-01-22T21:50:50.537127Z","shell.execute_reply.started":"2023-01-22T21:50:50.531499Z","shell.execute_reply":"2023-01-22T21:50:50.536348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_cutoff = np.percentile(train_pred, all_pfbeta.argmax())","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:51.698603Z","iopub.execute_input":"2023-01-22T21:50:51.698955Z","iopub.status.idle":"2023-01-22T21:50:51.704016Z","shell.execute_reply.started":"2023-01-22T21:50:51.698928Z","shell.execute_reply":"2023-01-22T21:50:51.703018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred_bin = (test_pred > best_cutoff).astype(float)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:53.014949Z","iopub.execute_input":"2023-01-22T21:50:53.01527Z","iopub.status.idle":"2023-01-22T21:50:53.019162Z","shell.execute_reply.started":"2023-01-22T21:50:53.015244Z","shell.execute_reply":"2023-01-22T21:50:53.018396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['cancer'] = test_pred_bin\nsubmission_df = test_df.groupby('prediction_id')['cancer'].agg(max)\nsubmission_df = submission_df.reset_index()\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:55.866699Z","iopub.execute_input":"2023-01-22T21:50:55.867038Z","iopub.status.idle":"2023-01-22T21:50:55.874695Z","shell.execute_reply.started":"2023-01-22T21:50:55.867013Z","shell.execute_reply":"2023-01-22T21:50:55.87388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2023-01-22T21:50:57.472361Z","iopub.execute_input":"2023-01-22T21:50:57.472669Z","iopub.status.idle":"2023-01-22T21:50:57.480374Z","shell.execute_reply.started":"2023-01-22T21:50:57.472645Z","shell.execute_reply":"2023-01-22T21:50:57.479626Z"},"trusted":true},"execution_count":null,"outputs":[]}]}