{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-05T09:39:28.904269Z","iopub.execute_input":"2024-01-05T09:39:28.904629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pylibjpeg\n!pip install python-gdcm\n!pip install plotly==5.11.0","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:06:28.189295Z","iopub.execute_input":"2024-01-05T09:06:28.18973Z","iopub.status.idle":"2024-01-05T09:07:28.507352Z","shell.execute_reply.started":"2024-01-05T09:06:28.189704Z","shell.execute_reply":"2024-01-05T09:07:28.506265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom collections import Counter\nfrom pathlib import Path\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nfrom mpl_toolkits.axes_grid1 import ImageGrid\nimport pydicom\nimport pylibjpeg\nsns.set_style(\"darkgrid\")","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:38.53045Z","iopub.execute_input":"2024-01-05T09:39:38.53115Z","iopub.status.idle":"2024-01-05T09:39:38.536879Z","shell.execute_reply.started":"2024-01-05T09:39:38.531122Z","shell.execute_reply":"2024-01-05T09:39:38.535681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = Path(\"/kaggle/input/rsna-breast-cancer-detection/train_images\")\ntest_images = Path(\"/kaggle/input/rsna-breast-cancer-detection/test_images\")\ntrain_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\nsample_submission_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:40.575829Z","iopub.execute_input":"2024-01-05T09:39:40.576646Z","iopub.status.idle":"2024-01-05T09:39:40.652636Z","shell.execute_reply.started":"2024-01-05T09:39:40.576615Z","shell.execute_reply":"2024-01-05T09:39:40.651793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:41.317034Z","iopub.execute_input":"2024-01-05T09:39:41.31737Z","iopub.status.idle":"2024-01-05T09:39:41.340758Z","shell.execute_reply.started":"2024-01-05T09:39:41.317345Z","shell.execute_reply":"2024-01-05T09:39:41.339765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:42.270792Z","iopub.execute_input":"2024-01-05T09:39:42.271159Z","iopub.status.idle":"2024-01-05T09:39:42.282677Z","shell.execute_reply.started":"2024-01-05T09:39:42.271131Z","shell.execute_reply":"2024-01-05T09:39:42.281584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:42.982524Z","iopub.execute_input":"2024-01-05T09:39:42.983391Z","iopub.status.idle":"2024-01-05T09:39:43.011247Z","shell.execute_reply.started":"2024-01-05T09:39:42.983357Z","shell.execute_reply":"2024-01-05T09:39:43.010186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:43.640048Z","iopub.execute_input":"2024-01-05T09:39:43.640412Z","iopub.status.idle":"2024-01-05T09:39:43.646503Z","shell.execute_reply.started":"2024-01-05T09:39:43.640383Z","shell.execute_reply":"2024-01-05T09:39:43.645568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:44.67543Z","iopub.execute_input":"2024-01-05T09:39:44.676141Z","iopub.status.idle":"2024-01-05T09:39:44.681785Z","shell.execute_reply.started":"2024-01-05T09:39:44.676108Z","shell.execute_reply":"2024-01-05T09:39:44.680869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_counter = train_df[\"patient_id\"].value_counts().sort_index()\nfig = px.histogram(images_counter, text_auto=True, title=\"Number of images per patient\")\nfig.update_layout(bargap=0.2)\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:45.466268Z","iopub.execute_input":"2024-01-05T09:39:45.467136Z","iopub.status.idle":"2024-01-05T09:39:45.536293Z","shell.execute_reply.started":"2024-01-05T09:39:45.467103Z","shell.execute_reply":"2024-01-05T09:39:45.535382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"is_cancer_person = train_df.groupby(\"patient_id\")['cancer'].max().sort_index()\nis_cancer_person.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:45.890233Z","iopub.execute_input":"2024-01-05T09:39:45.891037Z","iopub.status.idle":"2024-01-05T09:39:45.904607Z","shell.execute_reply.started":"2024-01-05T09:39:45.891005Z","shell.execute_reply":"2024-01-05T09:39:45.903556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.age.isna()","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:50.785377Z","iopub.execute_input":"2024-01-05T09:39:50.785763Z","iopub.status.idle":"2024-01-05T09:39:50.794593Z","shell.execute_reply.started":"2024-01-05T09:39:50.785734Z","shell.execute_reply":"2024-01-05T09:39:50.793556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_age_values = train_df[train_df['age'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:51.295312Z","iopub.execute_input":"2024-01-05T09:39:51.296159Z","iopub.status.idle":"2024-01-05T09:39:51.301559Z","shell.execute_reply.started":"2024-01-05T09:39:51.296126Z","shell.execute_reply":"2024-01-05T09:39:51.30055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_age_values","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:51.820045Z","iopub.execute_input":"2024-01-05T09:39:51.820986Z","iopub.status.idle":"2024-01-05T09:39:51.85029Z","shell.execute_reply.started":"2024-01-05T09:39:51.820951Z","shell.execute_reply":"2024-01-05T09:39:51.849307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_values = {}\n\nfor patient_id, group in train_df.groupby('patient_id')['age']:\n    age_values[patient_id] = group.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:52.240491Z","iopub.execute_input":"2024-01-05T09:39:52.240882Z","iopub.status.idle":"2024-01-05T09:39:52.530584Z","shell.execute_reply.started":"2024-01-05T09:39:52.240853Z","shell.execute_reply":"2024-01-05T09:39:52.529787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\nfor patient_id, ages in age_values.items():\n    for age in ages:\n        if math.isnan(age):\n            print(f\"Patient ID: {patient_id}, Age: {age}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:52.735455Z","iopub.execute_input":"2024-01-05T09:39:52.735867Z","iopub.status.idle":"2024-01-05T09:39:52.754355Z","shell.execute_reply.started":"2024-01-05T09:39:52.735837Z","shell.execute_reply":"2024-01-05T09:39:52.75342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"patient ids with null age are 51500,49020,47764,45891,40791,27212,11995","metadata":{}},{"cell_type":"code","source":"type(train_df['patient_id'][0])","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:54.642163Z","iopub.execute_input":"2024-01-05T09:39:54.642583Z","iopub.status.idle":"2024-01-05T09:39:54.649452Z","shell.execute_reply.started":"2024-01-05T09:39:54.642545Z","shell.execute_reply":"2024-01-05T09:39:54.6483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for patient_id , ages in age_values.items():\n    if patient_id in [51500,49020,47764,45891,40791,27212,23752,11995]:\n        print(ages)","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:56.180255Z","iopub.execute_input":"2024-01-05T09:39:56.180955Z","iopub.status.idle":"2024-01-05T09:39:56.189129Z","shell.execute_reply.started":"2024-01-05T09:39:56.180925Z","shell.execute_reply":"2024-01-05T09:39:56.188209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_ids_to_drop =[51500,49020,47764,45891,40791,27212,23752,11995]\n\n# Dropping rows with specific patient IDs\ntrain_df2 = train_df.drop(train_df[train_df['patient_id'].isin(patient_ids_to_drop)].index)","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:56.580068Z","iopub.execute_input":"2024-01-05T09:39:56.580401Z","iopub.status.idle":"2024-01-05T09:39:56.591986Z","shell.execute_reply.started":"2024-01-05T09:39:56.580375Z","shell.execute_reply":"2024-01-05T09:39:56.591009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df2.age.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:57.679878Z","iopub.execute_input":"2024-01-05T09:39:57.680244Z","iopub.status.idle":"2024-01-05T09:39:57.687528Z","shell.execute_reply.started":"2024-01-05T09:39:57.680214Z","shell.execute_reply":"2024-01-05T09:39:57.68642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"person_age = train_df2.groupby(\"patient_id\")['age'].max().sort_index().astype('int64')\n\nfig = px.histogram(person_age, title=\"Distribution of patients' age\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:39:58.490028Z","iopub.execute_input":"2024-01-05T09:39:58.490385Z","iopub.status.idle":"2024-01-05T09:39:58.558665Z","shell.execute_reply.started":"2024-01-05T09:39:58.490356Z","shell.execute_reply":"2024-01-05T09:39:58.557791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ages = train_df2.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\ncancer_ages = train_df2[train_df2['cancer'] == 1].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\nno_cancer_ages = train_df2[train_df2['cancer'] == 0].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nplt.figure(figsize=(14, 7))\n\nplt.subplot(1, 2, 2)\nsn.histplot(cancer_ages, bins=51, color='mediumvioletred', kde=True)  \nsn.histplot(no_cancer_ages, bins=63, color='indigo', kde=True)  \nplt.title(\"Patients with/without cancer\", fontsize=14)\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"Count\", fontsize=12)\nplt.xticks(fontsize=10)\nplt.yticks(fontsize=10)\nplt.xlim(33, 89)\nplt.legend([\"Cancer\", \"No cancer\"], fontsize=10)\n\nplt.suptitle(\"Age distribution of the patients\", fontsize=16)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:44:45.870227Z","iopub.execute_input":"2024-01-05T09:44:45.870717Z","iopub.status.idle":"2024-01-05T09:44:48.255084Z","shell.execute_reply.started":"2024-01-05T09:44:45.870688Z","shell.execute_reply":"2024-01-05T09:44:48.254072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_correlation_heatmap(df):\n    df = pd.get_dummies(df, columns=['laterality', 'view', 'density'])\n    df['difficult_negative_case'] = df['difficult_negative_case'].astype(int)\n\n    corr = df.corr()\n#     mask = np.triu(np.ones_like(corr, dtype=bool))\n    fig, ax = plt.subplots(figsize=(10, 7))\n    cmap = sns.diverging_palette(230, 20, as_cmap=True)\n    sns.heatmap(corr, cmap=cmap, vmax=.3, center=0, square=True, linewidths=.5, cbar_kws={\"shrink\": .5})\nplot_correlation_heatmap(train_df2)","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:40:00.059931Z","iopub.execute_input":"2024-01-05T09:40:00.060291Z","iopub.status.idle":"2024-01-05T09:40:01.078479Z","shell.execute_reply.started":"2024-01-05T09:40:00.060264Z","shell.execute_reply":"2024-01-05T09:40:01.077589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df2.view.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:07:32.446081Z","iopub.execute_input":"2024-01-05T09:07:32.446366Z","iopub.status.idle":"2024-01-05T09:07:32.459856Z","shell.execute_reply.started":"2024-01-05T09:07:32.446342Z","shell.execute_reply":"2024-01-05T09:07:32.45885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[test_df['patient_id']==10008]","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:13:23.549852Z","iopub.execute_input":"2024-01-05T09:13:23.550206Z","iopub.status.idle":"2024-01-05T09:13:23.563098Z","shell.execute_reply.started":"2024-01-05T09:13:23.550173Z","shell.execute_reply":"2024-01-05T09:13:23.562106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_images_of_patient(patient_id):\n    patient_df = test_df[test_df[\"patient_id\"] == patient_id] if patient_id in test_df[\"patient_id\"].values else train_df[train_df[\"patient_id\"] == patient_id] if patient_id in train_df[\"patient_id\"].values else None\n    if patient_df is None:\n        print(f\"No patient with id {patient_id}\")\n        return\n    \n    patient_folder = test_images / f\"{patient_id}\" if patient_id in test_df[\"patient_id\"].values else train_images / f\"{patient_id}\"\n    num_images = patient_df.shape[0]\n    n_rows, n_cols = (2, 2) if num_images <= 4 else (4, 4) if num_images <= 8 else (8, 8)\n    \n    fig, ax = plt.subplots(figsize=(24., 24.), nrows=n_rows, ncols=n_cols, gridspec_kw={'wspace': 0.1, 'hspace': 0.1})\n    fig.suptitle(f\"{patient_id} - {num_images} images\", fontsize=16)\n    \n    for idx, curr_image in enumerate(patient_folder.iterdir()):\n        curr_image = Path(curr_image)\n        image_data = patient_df[patient_df[\"image_id\"] == int(curr_image.stem)].squeeze()\n        ds = pydicom.dcmread(curr_image)\n        image_as_np = ds.pixel_array.astype(np.float32)\n        curr_ax = ax[idx // n_cols, idx % n_cols]\n        curr_ax.imshow(image_as_np)\n        curr_ax.set_title(\"image_id: {}\\n laterality: {}\\n view {}\\n machine_id {}\".format(image_data[\"image_id\"], image_data[\"laterality\"], image_data[\"view\"], image_data[\"machine_id\"]), fontdict={\"fontsize\": 9})","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:09:31.436356Z","iopub.execute_input":"2024-01-05T09:09:31.437369Z","iopub.status.idle":"2024-01-05T09:09:31.451733Z","shell.execute_reply.started":"2024-01-05T09:09:31.437327Z","shell.execute_reply":"2024-01-05T09:09:31.450572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" plot_images_of_patient(10008)","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:34:09.194397Z","iopub.execute_input":"2024-01-05T09:34:09.194793Z","iopub.status.idle":"2024-01-05T09:34:14.917185Z","shell.execute_reply.started":"2024-01-05T09:34:09.194764Z","shell.execute_reply":"2024-01-05T09:34:14.915878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_images_of_patient(2518)","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:32:50.454429Z","iopub.execute_input":"2024-01-05T09:32:50.455167Z","iopub.status.idle":"2024-01-05T09:33:06.884677Z","shell.execute_reply.started":"2024-01-05T09:32:50.455131Z","shell.execute_reply":"2024-01-05T09:33:06.883662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_images_of_patient(10179)","metadata":{"execution":{"iopub.status.busy":"2024-01-05T09:33:43.918351Z","iopub.execute_input":"2024-01-05T09:33:43.918777Z","iopub.status.idle":"2024-01-05T09:33:49.499628Z","shell.execute_reply.started":"2024-01-05T09:33:43.918744Z","shell.execute_reply":"2024-01-05T09:33:49.498063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}