{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install -qU \"python-gdcm\" pydicom pylibjpeg \"opencv-python-headless\"","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:16:43.551113Z","iopub.execute_input":"2023-03-16T21:16:43.551595Z","iopub.status.idle":"2023-03-16T21:16:56.99321Z","shell.execute_reply.started":"2023-03-16T21:16:43.551545Z","shell.execute_reply":"2023-03-16T21:16:56.991944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# General Libraries\nimport os\nimport re\nimport gc\nimport cv2\nimport wandb\nimport random\nimport math\nfrom glob import glob\nfrom tqdm import tqdm\nfrom pprint import pprint\nfrom time import time\nimport datetime as dtime\nfrom datetime import datetime\nimport itertools\nimport warnings\nimport pandas as pd\nimport numpy as np\nimport pydicom # for DICOM images\nfrom skimage.transform import resize\nfrom sklearn.preprocessing import LabelEncoder, normalize\n\n# For the Visuals\nimport seaborn as sns\nimport matplotlib as mpl\nfrom matplotlib import cm\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom matplotlib.offsetbox import AnnotationBbox, OffsetImage\nfrom matplotlib.colors import ListedColormap, LinearSegmentedColormap\nfrom matplotlib.patches import Rectangle\nfrom IPython.display import display_html\nplt.rcParams.update({'font.size': 16})\n\n# Environment check\nwarnings.filterwarnings(\"ignore\")\nos.environ[\"WANDB_SILENT\"] = \"true\"\nCONFIG = {'competition': 'RSNA_Breast_Cancer', '_wandb_kernel': 'aot'}\n\n# Custom colors\nclass clr:\n    S = '\\033[1m' + '\\033[91m'\n    E = '\\033[0m'\n    \nmy_colors = [\"#517664\", \"#73AA90\", \"#94DDBC\", \"#DAB06C\", \n             \"#DF928E\", \"#C97973\", \"#B25F57\"]\nCMAP1 = ListedColormap(my_colors)\n\nprint(clr.S+\"Notebook Color Schemes:\"+clr.E)\nsns.palplot(sns.color_palette(my_colors))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:17:10.130345Z","iopub.execute_input":"2023-03-16T21:17:10.13074Z","iopub.status.idle":"2023-03-16T21:17:11.091944Z","shell.execute_reply.started":"2023-03-16T21:17:10.130704Z","shell.execute_reply":"2023-03-16T21:17:11.090327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Secrets\nfrom kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nsecret_value_0 = user_secrets.get_secret(\"wandb\")\n\n! wandb login $secret_value_0","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:21:48.194726Z","iopub.execute_input":"2023-03-16T21:21:48.195285Z","iopub.status.idle":"2023-03-16T21:21:52.276255Z","shell.execute_reply.started":"2023-03-16T21:21:48.195244Z","shell.execute_reply":"2023-03-16T21:21:52.274858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Bonus: Cover Photo\nrun = wandb.init(project='RSNA_Breast_Cancer', name='CoverPhoto', config=CONFIG)\ncover = plt.imread(\"/kaggle/input/rsna-breast-cancer-helper-data/DKn4ofz.png\")\nwandb.log({\"cover\": wandb.Image(cover)})\nwandb.finish()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:23:38.389374Z","iopub.execute_input":"2023-03-16T21:23:38.389766Z","iopub.status.idle":"2023-03-16T21:24:21.578408Z","shell.execute_reply.started":"2023-03-16T21:23:38.389729Z","shell.execute_reply":"2023-03-16T21:24:21.577108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\n\n# Get image path\n# Example path: '/kaggle/input/rsna-breast-cancer-detection/train_images/10706/763186195.dcm'\nbase_path = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\nall_paths = []\nfor k in tqdm(range(len(train))):\n    row = train.iloc[k, :]\n    all_paths.append(base_path + str(row.patient_id) + \"/\" + str(row.image_id) + \".dcm\")\n    \ntrain[\"path\"] = all_paths","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:24:21.580562Z","iopub.execute_input":"2023-03-16T21:24:21.581226Z","iopub.status.idle":"2023-03-16T21:24:29.920402Z","shell.execute_reply.started":"2023-03-16T21:24:21.581183Z","shell.execute_reply":"2023-03-16T21:24:29.919307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Number of TOTAL images:\"+clr.E,\n      len(glob(\"/kaggle/input/rsna-breast-cancer-detection/train_images/*/*\")))\nprint(clr.S+\"Records gathered in Site 1:\"+clr.E, train[\"site_id\"].value_counts().values[0], \"\\n\"+\n      clr.S+\"Records gathered in Site 2:\"+clr.E, train[\"site_id\"].value_counts().values[1])\nprint(\"-------------------------------------------------\")\nprint(clr.S+\"Total unique patients:\"+clr.E, train[\"patient_id\"].nunique())\nprint(\"-------------------------------------------------\")\nprint(clr.S+\"Total unique images:\"+clr.E, train[\"image_id\"].nunique())\nprint(\"-------------------------------------------------\")\nprint(clr.S+\"Statistics: Images per Patient\"+clr.E)\nprint(train.groupby(\"patient_id\")[\"image_id\"].count().reset_index().describe()[\"image_id\"])\nprint(\"-------------------------------------------------\")\nprint(clr.S+\"Image records count per laterality (R):\"+clr.E, train[\"laterality\"].value_counts().values[0], \"\\n\"+\n      clr.S+\"Image records count per laterality (L):\"+clr.E, train[\"laterality\"].value_counts().values[1])\nprint(\"-------------------------------------------------\")\nprint(clr.S+\"Image records count per View:\"+clr.E)\nprint(train[\"view\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:24:29.92504Z","iopub.execute_input":"2023-03-16T21:24:29.925893Z","iopub.status.idle":"2023-03-16T21:25:22.738579Z","shell.execute_reply.started":"2023-03-16T21:24:29.925853Z","shell.execute_reply":"2023-03-16T21:25:22.737577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 New Experiment\nrun = wandb.init(project='RSNA_Breast_Cancer', name='view_sample', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:25:22.740509Z","iopub.execute_input":"2023-03-16T21:25:22.741254Z","iopub.status.idle":"2023-03-16T21:25:53.685058Z","shell.execute_reply.started":"2023-03-16T21:25:22.741214Z","shell.execute_reply":"2023-03-16T21:25:53.683983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_view(view_name, sample_size):\n    \n    if view_name != \"LMO\":\n        # Get image info\n        data = train[train[\"view\"]==view_name].sample(sample_size, random_state=24)\n        image_path = data[\"path\"].to_list()\n\n        # Plot\n        fig, axs = plt.subplots(1, sample_size, figsize=(23, 4))\n        axs = axs.flatten()\n        wandb_images = []\n\n        for k, path in enumerate(image_path):\n            axs[k].set_title(f\"{k+1}. {view_name}\", \n                             fontsize = 16, color = my_colors[0], weight='bold')\n\n            img = pydicom.dcmread(path).pixel_array\n            wandb_images.append(wandb.Image(img))\n            axs[k].imshow(img, cmap=\"turbo\")\n            axs[k].axis(\"off\")\n\n        plt.tight_layout()\n        plt.show()\n\n        # 🐝 Log Image to W&B\n        wandb.log({f\"{view_name}\": wandb_images})\n    else:\n        path = train[train[\"view\"]==\"LMO\"][\"path\"].item()\n        # Plot\n        fig, axs = plt.subplots(1, sample_size, figsize=(23, 4))\n        axs = axs.flatten()\n        wandb_images = []\n        img = pydicom.dcmread(path).pixel_array\n        wandb_images.append(wandb.Image(img))\n        axs[0].imshow(img, cmap=\"turbo\")\n        axs[0].set_title(f\"1. LMO\", \n                         fontsize = 16, color = my_colors[0], weight='bold')\n        axs[0].axis(\"off\")\n        axs[1].axis(\"off\")\n        axs[2].axis(\"off\")\n        axs[3].axis(\"off\")\n        axs[4].axis(\"off\")\n\n        plt.tight_layout()\n        plt.show()\n        \n        wandb.log({f\"LMO\": wandb_images})","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:25:53.687729Z","iopub.execute_input":"2023-03-16T21:25:53.68971Z","iopub.status.idle":"2023-03-16T21:25:53.705085Z","shell.execute_reply.started":"2023-03-16T21:25:53.689672Z","shell.execute_reply":"2023-03-16T21:25:53.703934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for view_name in train[\"view\"].unique().tolist():\n    # Custom function to prin images & log into 🐝W&B\n    show_view(view_name, sample_size=5)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:26:08.248966Z","iopub.execute_input":"2023-03-16T21:26:08.250095Z","iopub.status.idle":"2023-03-16T21:26:57.105402Z","shell.execute_reply.started":"2023-03-16T21:26:08.250038Z","shell.execute_reply":"2023-03-16T21:26:57.101746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:26:57.107243Z","iopub.execute_input":"2023-03-16T21:26:57.107811Z","iopub.status.idle":"2023-03-16T21:27:01.728519Z","shell.execute_reply.started":"2023-03-16T21:26:57.107773Z","shell.execute_reply":"2023-03-16T21:27:01.727125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 New Experiment\nrun = wandb.init(project='RSNA_Breast_Cancer', name='age_hist', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:27:01.730732Z","iopub.execute_input":"2023-03-16T21:27:01.731448Z","iopub.status.idle":"2023-03-16T21:27:33.143008Z","shell.execute_reply.started":"2023-03-16T21:27:01.731409Z","shell.execute_reply":"2023-03-16T21:27:33.141855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot\nf, (a0, a1) = plt.subplots(2, 1, gridspec_kw={'height_ratios': [3, 1]}, figsize=(24, 15))\nsns.distplot(a=train[\"age\"], rug=True, hist=False, \n             rug_kws={\"color\": my_colors[5]},\n             kde_kws={\"color\": my_colors[5], \"lw\": 5, \"alpha\": 0.7},\n             ax=a0)\n\na0.axvline(x=58, ls=\":\", lw=2, color=\"black\")\na0.text(x=58.5, y=0.018, s=\"mean: 58\", size=17, color=\"black\", weight=\"bold\")\na0.axvline(x=26, ls=\":\", lw=2, color=\"black\")\na0.text(x=26.5, y=0.008, s=\"min: 26\", size=17, color=\"black\", weight=\"bold\")\na0.axvline(x=89, ls=\":\", lw=2, color=\"black\")\na0.text(x=84, y=0.037, s=\"max: 89\", size=17, color=\"black\", weight=\"bold\")\n\nsns.boxenplot(x=train[\"age\"], ax=a1, color=my_colors[2])\n\nplt.suptitle(\"Age Distribution\", weight=\"bold\", size=25)\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:35:33.871582Z","iopub.execute_input":"2023-03-16T21:35:33.872057Z","iopub.status.idle":"2023-03-16T21:35:35.591439Z","shell.execute_reply.started":"2023-03-16T21:35:33.872011Z","shell.execute_reply":"2023-03-16T21:35:35.590307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 🐝 Log into dashboard\n# create_wandb_hist(x_data=train[\"age\"], \n#                   x_name=\"Age\",\n#                   title=\"Age Distribution\",\n#                   log=\"age_hist\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot\nf, (a0, a1) = plt.subplots(1, 2, figsize=(24, 12))\nsns.distplot(a=train[train[\"cancer\"]==0][\"age\"], rug=True, hist=False, \n             rug_kws={\"color\": my_colors[5]},\n             kde_kws={\"color\": my_colors[5], \"lw\": 5, \"alpha\": 0.7},\n             ax=a0)\na0.set_title(\"No Cancer Present\", weight=\"bold\", size=20)\na0.axvline(x=58, ls=\":\", lw=2, color=\"black\")\na0.text(x=58.5, y=0.018, s=\"mean: 58\", size=17, color=\"black\", weight=\"bold\")\na0.axvline(x=26, ls=\":\", lw=2, color=\"black\")\na0.text(x=26.5, y=0.008, s=\"min: 26\", size=17, color=\"black\", weight=\"bold\")\na0.axvline(x=89, ls=\":\", lw=2, color=\"black\")\na0.text(x=79, y=0.037, s=\"max: 89\", size=17, color=\"black\", weight=\"bold\")\n\n\nsns.distplot(a=train[train[\"cancer\"]==1][\"age\"], rug=True, hist=False, \n             rug_kws={\"color\": my_colors[2]},\n             kde_kws={\"color\": my_colors[2], \"lw\": 5, \"alpha\": 0.7},\n             ax=a1)\na1.set_title(\"Cancer Present\", weight=\"bold\", size=20)\na1.axvline(x=63, ls=\":\", lw=2, color=\"black\")\na1.text(x=63.5, y=0.018, s=\"mean: 63\", size=17, color=\"black\", weight=\"bold\")\na1.axvline(x=38, ls=\":\", lw=2, color=\"black\")\na1.text(x=38.5, y=0.008, s=\"min: 38\", size=17, color=\"black\", weight=\"bold\")\na1.axvline(x=89, ls=\":\", lw=2, color=\"black\")\na1.text(x=79, y=0.037, s=\"max: 89\", size=17, color=\"black\", weight=\"bold\")\n\n\nplt.suptitle(\"Age Distribution\", weight=\"bold\", size=25)\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:37:09.484075Z","iopub.execute_input":"2023-03-16T21:37:09.484525Z","iopub.status.idle":"2023-03-16T21:37:11.581212Z","shell.execute_reply.started":"2023-03-16T21:37:09.484489Z","shell.execute_reply":"2023-03-16T21:37:11.580029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 🐝 Log into dashboard\n# create_wandb_hist(x_data=train[train[\"cancer\"]==1][\"age\"], \n#                   x_name=\"Age\",\n#                   title=\"Age Distribution - patients with cancer\",\n#                   log=\"age_hist_cancer\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_images(col, col_flag, sample_size, cancer_flag=0):\n    \n    # Get image info\n    data = train[train[col]==col_flag].sample(sample_size, random_state=24)\n    if cancer_flag==1:\n        data = train[train.cancer==1]\n        data = data[data[col]==col_flag].sample(sample_size, random_state=24)\n    image_path = data[\"path\"].to_list()\n\n    # Plot\n    fig, axs = plt.subplots(1, sample_size, figsize=(23, 4))\n    axs = axs.flatten()\n    wandb_images = []\n\n    for k, path in enumerate(image_path):\n        axs[k].set_title(f\"{k+1}. {col_flag}\", \n                         fontsize = 14, color = my_colors[0], weight='bold')\n\n        img = pydicom.dcmread(path).pixel_array\n        wandb_images.append(wandb.Image(img))\n        axs[k].imshow(img, cmap=\"turbo\")\n        axs[k].axis(\"off\")\n\n    plt.tight_layout()\n    plt.show()\n\n    # 🐝 Log Image to W&B\n    wandb.log({f\"{col}_{col_flag}\": wandb_images})","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:45:00.683867Z","iopub.execute_input":"2023-03-16T21:45:00.68485Z","iopub.status.idle":"2023-03-16T21:45:00.696355Z","shell.execute_reply.started":"2023-03-16T21:45:00.684814Z","shell.execute_reply":"2023-03-16T21:45:00.694902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Records with no implants:\"+clr.E, train[\"implant\"].value_counts().values[0], \"\\n\"+\n      clr.S+\"Records with implants:\"+clr.E, train[\"implant\"].value_counts().values[1], \"\\n\")\n\nfor implant_flag in train[\"implant\"].unique().tolist():\n    # Custom function to prin images & log into 🐝W&B\n    show_images(col=\"implant\", col_flag=implant_flag, sample_size=5)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:45:15.537584Z","iopub.execute_input":"2023-03-16T21:45:15.53821Z","iopub.status.idle":"2023-03-16T21:45:31.879366Z","shell.execute_reply.started":"2023-03-16T21:45:15.538173Z","shell.execute_reply":"2023-03-16T21:45:31.878293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:45:56.382914Z","iopub.execute_input":"2023-03-16T21:45:56.383614Z","iopub.status.idle":"2023-03-16T21:46:02.220522Z","shell.execute_reply.started":"2023-03-16T21:45:56.383576Z","shell.execute_reply":"2023-03-16T21:46:02.219085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 New Experiment\nrun = wandb.init(project='RSNA_Breast_Cancer', name='cancer_explore', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:46:02.224112Z","iopub.execute_input":"2023-03-16T21:46:02.22449Z","iopub.status.idle":"2023-03-16T21:46:33.959652Z","shell.execute_reply.started":"2023-03-16T21:46:02.224444Z","shell.execute_reply":"2023-03-16T21:46:33.958155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Records with no cancer:\"+clr.E, train[\"cancer\"].value_counts().values[0], \"\\n\"+\n      clr.S+\"Records with cancer:\"+clr.E, train[\"cancer\"].value_counts().values[1], \"\\n\")\n\nfor cancer_flag in train[\"cancer\"].unique().tolist():\n    # Custom function to prin images & log into 🐝W&B\n    show_images(col=\"cancer\", col_flag=cancer_flag, sample_size=5)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:46:33.965698Z","iopub.execute_input":"2023-03-16T21:46:33.968106Z","iopub.status.idle":"2023-03-16T21:46:51.478062Z","shell.execute_reply.started":"2023-03-16T21:46:33.968064Z","shell.execute_reply":"2023-03-16T21:46:51.476694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Records with invasion:\"+clr.E, train[train.cancer==1][\"invasive\"].value_counts().values[0], \"\\n\"+\n      clr.S+\"Records with no invasion:\"+clr.E, train[train.cancer==1][\"invasive\"].value_counts().values[1], \"\\n\")\n\nfor invasive_flag in train[\"invasive\"].unique().tolist():\n    # Custom function to prin images & log into 🐝W&B\n    show_images(col=\"invasive\", col_flag=invasive_flag, sample_size=5, cancer_flag=1)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:47:26.79459Z","iopub.execute_input":"2023-03-16T21:47:26.79496Z","iopub.status.idle":"2023-03-16T21:47:41.072324Z","shell.execute_reply.started":"2023-03-16T21:47:26.794927Z","shell.execute_reply":"2023-03-16T21:47:41.071273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Cases not particularly difficult:\"+clr.E, train[\"difficult_negative_case\"].value_counts().values[0], \"\\n\"+\n      clr.S+\"Cases particularly difficult:\"+clr.E, train[\"difficult_negative_case\"].value_counts().values[1], \"\\n\")\n\nfor difficult_flag in train[\"difficult_negative_case\"].unique().tolist():\n    # Custom function to prin images & log into 🐝W&B\n    show_images(col=\"difficult_negative_case\", col_flag=difficult_flag, sample_size=5, cancer_flag=0)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:47:41.074329Z","iopub.execute_input":"2023-03-16T21:47:41.074907Z","iopub.status.idle":"2023-03-16T21:47:56.12248Z","shell.execute_reply.started":"2023-03-16T21:47:41.074869Z","shell.execute_reply":"2023-03-16T21:47:56.121359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:47:56.128284Z","iopub.execute_input":"2023-03-16T21:47:56.131069Z","iopub.status.idle":"2023-03-16T21:48:02.282952Z","shell.execute_reply.started":"2023-03-16T21:47:56.131022Z","shell.execute_reply":"2023-03-16T21:48:02.281304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Keep only columns in test + target variable\ntrain = train[[\"patient_id\", \"image_id\", \"laterality\", \"view\", \"age\", \"implant\", \"path\", \"cancer\"]]\n\n# Encode categorical variables\nle_laterality = LabelEncoder()\nle_view = LabelEncoder()\n\ntrain['laterality'] = le_laterality.fit_transform(train['laterality'])\ntrain['view'] = le_view.fit_transform(train['view'])\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:48:02.288704Z","iopub.execute_input":"2023-03-16T21:48:02.289102Z","iopub.status.idle":"2023-03-16T21:48:02.33898Z","shell.execute_reply.started":"2023-03-16T21:48:02.289065Z","shell.execute_reply":"2023-03-16T21:48:02.338038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Number of missing values in Age:\"+clr.E, train[\"age\"].isna().sum())\ntrain['age'] = train['age'].fillna(58)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:48:02.341773Z","iopub.execute_input":"2023-03-16T21:48:02.342519Z","iopub.status.idle":"2023-03-16T21:48:02.350894Z","shell.execute_reply.started":"2023-03-16T21:48:02.342479Z","shell.execute_reply":"2023-03-16T21:48:02.349829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save new dataset\n# train.to_csv(\"train_path.csv\", index=False)\n\n# # 🐝 Save Artifacts\n# save_dataset_artifact(run_name=\"save_train_prep\", \n#                       artifact_name=\"train_prep\",\n#                       path=\"/kaggle/input/rsna-breast-cancer-helper-data/train_path.csv\",\n#                       data_type=\"dataset\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**- PyTorch Baseline -**","metadata":{}},{"cell_type":"code","source":"!pip install -q efficientnet_pytorch","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:50:38.69982Z","iopub.execute_input":"2023-03-16T21:50:38.700214Z","iopub.status.idle":"2023-03-16T21:50:50.045278Z","shell.execute_reply.started":"2023-03-16T21:50:38.70018Z","shell.execute_reply":"2023-03-16T21:50:50.044004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PyTorch\nimport torch\nimport torchvision\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch import FloatTensor, LongTensor\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\n\n# Data Augmentation for Image Preprocessing\nfrom albumentations import (ToFloat, Normalize, VerticalFlip, HorizontalFlip, Compose, Resize,\n                            RandomBrightnessContrast, HueSaturationValue, Blur, GaussNoise,\n                            Rotate, RandomResizedCrop, Cutout, ShiftScaleRotate, ToGray)\nfrom albumentations.pytorch import ToTensorV2\n\nfrom efficientnet_pytorch import EfficientNet\nfrom torchvision.models import resnet34, resnet50\n\n# SKlearn\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold\nfrom sklearn.metrics import accuracy_score, roc_auc_score, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:50:58.391511Z","iopub.execute_input":"2023-03-16T21:50:58.391912Z","iopub.status.idle":"2023-03-16T21:51:01.566136Z","shell.execute_reply.started":"2023-03-16T21:50:58.391874Z","shell.execute_reply":"2023-03-16T21:51:01.56506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed):\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    np.random.seed(seed)\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:57:41.080698Z","iopub.execute_input":"2023-03-16T21:57:41.081671Z","iopub.status.idle":"2023-03-16T21:57:41.088623Z","shell.execute_reply.started":"2023-03-16T21:57:41.081617Z","shell.execute_reply":"2023-03-16T21:57:41.087358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seed\nset_seed(13)\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint('Device available now:', DEVICE)\n\n# Read in Data\ntrain = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-helper-data/train_path.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:57:43.889637Z","iopub.execute_input":"2023-03-16T21:57:43.890768Z","iopub.status.idle":"2023-03-16T21:57:44.112655Z","shell.execute_reply.started":"2023-03-16T21:57:43.890719Z","shell.execute_reply":"2023-03-16T21:57:44.111448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ----- GLOBAL PARAMS -----\nvertical_flip = 0.5\nhorizontal_flip = 0.5\n\ncsv_columns = ['laterality', 'view', 'age', 'implant']\nno_columns = len(csv_columns)\noutput_size = 1\n# -------------------------","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:57:49.000226Z","iopub.execute_input":"2023-03-16T21:57:49.000646Z","iopub.status.idle":"2023-03-16T21:57:49.006645Z","shell.execute_reply.started":"2023-03-16T21:57:49.000609Z","shell.execute_reply":"2023-03-16T21:57:49.005555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNADataset(Dataset):\n    \n    def __init__(self, dataframe, vertical_flip, horizontal_flip,\n                 is_train=True):\n        self.dataframe, self.is_train = dataframe, is_train\n        self.vertical_flip, self.horizontal_flip = vertical_flip, horizontal_flip\n        \n        # Data Augmentation (custom for each dataset type)\n        if is_train:\n            self.transform = Compose([RandomResizedCrop(height=224, width=224),\n                                      ShiftScaleRotate(rotate_limit=90, scale_limit = [0.8, 1.2]),\n                                      HorizontalFlip(p = self.horizontal_flip),\n                                      VerticalFlip(p = self.vertical_flip),\n                                      ToTensorV2()])\n        else:\n            self.transform = Compose([ToTensorV2()])\n            \n    def __len__(self):\n        return len(self.dataframe)\n    \n    def __getitem__(self, index):\n        '''Take each row in batcj at a time.'''\n        \n        # Select path and read image\n        image_path = self.dataframe['path'][index]\n        image = pydicom.dcmread(image_path).pixel_array.astype(np.float32)\n        \n        # For this image also import .csv information\n        csv_data = np.array(self.dataframe.iloc[index][csv_columns].values, \n                            dtype=np.float32)\n        # Apply transforms\n        transf_image = self.transform(image=image)['image']\n        # Change image from 1 channel (B&W) to 3 channels\n        transf_image = np.concatenate([transf_image, transf_image, transf_image], axis=0)\n        \n        # Return info\n        if self.is_train:\n            return {\"image\": transf_image, \n                    \"meta\": csv_data, \n                    \"target\": self.dataframe['cancer'][index]}\n        else:\n            return {\"image\": transf_image, \n                    \"meta\": csv_data}","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:57:50.643686Z","iopub.execute_input":"2023-03-16T21:57:50.64412Z","iopub.status.idle":"2023-03-16T21:57:50.657524Z","shell.execute_reply.started":"2023-03-16T21:57:50.644082Z","shell.execute_reply":"2023-03-16T21:57:50.656061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_to_device(data):\n    \n    image, metadata, targets = data.values()\n    return image.to(DEVICE), metadata.to(DEVICE), targets.to(DEVICE)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:57:56.670927Z","iopub.execute_input":"2023-03-16T21:57:56.671668Z","iopub.status.idle":"2023-03-16T21:57:56.676845Z","shell.execute_reply.started":"2023-03-16T21:57:56.67163Z","shell.execute_reply":"2023-03-16T21:57:56.675768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample data\nsample_df = train.head(6)\n\n# Instantiate Dataset object\ndataset = RSNADataset(sample_df, vertical_flip, horizontal_flip,\n                      is_train=True)\n# The Dataloader\ndataloader = DataLoader(dataset, batch_size=3, shuffle=False)\n\n# Output of the Dataloader\nfor k, data in enumerate(dataloader):\n    image, meta, targets = data_to_device(data)\n    print(clr.S + f\"Batch: {k}\" + clr.E, \"\\n\" +\n          clr.S + \"Image:\" + clr.E, image.shape, \"\\n\" +\n          clr.S + \"Meta:\" + clr.E, meta, \"\\n\" +\n          clr.S + \"Targets:\" + clr.E, targets, \"\\n\" +\n          \"=\"*50)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:57:58.321545Z","iopub.execute_input":"2023-03-16T21:57:58.322269Z","iopub.status.idle":"2023-03-16T21:58:07.244379Z","shell.execute_reply.started":"2023-03-16T21:57:58.322228Z","shell.execute_reply":"2023-03-16T21:58:07.243336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ResNet50Network(nn.Module):\n    def __init__(self, output_size, no_columns):\n        super().__init__()\n        self.no_columns, self.output_size = no_columns, output_size\n        \n        # Define Feature part (IMAGE)\n        self.features = resnet50(pretrained=True) # 1000 neurons out\n        # (metadata)\n        self.csv = nn.Sequential(nn.Linear(self.no_columns, 500),\n                                 nn.BatchNorm1d(500),\n                                 nn.ReLU(),\n                                 nn.Dropout(p=0.2))\n        \n        # Define Classification part\n        self.classification = nn.Linear(1000 + 500, output_size)\n        \n        \n    def forward(self, image, meta, prints=False):\n        if prints: print('Input Image shape:', image.shape, '\\n'+\n                         'Input metadata shape:', meta.shape)\n        \n        # Image CNN\n        image = self.features(image)\n        if prints: print('Features Image shape:', image.shape)\n        \n        # CSV FNN\n        meta = self.csv(meta)\n        if prints: print('Meta Data:', meta.shape)\n            \n        # Concatenate layers from image with layers from csv_data\n        image_meta_data = torch.cat((image, meta), dim=1)\n        if prints: print('Concatenated Data:', image_meta_data.shape)\n        \n        # CLASSIF\n        out = self.classification(image_meta_data)\n        if prints: print('Out shape:', out.shape)\n        \n        return out","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:58:09.096884Z","iopub.execute_input":"2023-03-16T21:58:09.097659Z","iopub.status.idle":"2023-03-16T21:58:09.107095Z","shell.execute_reply.started":"2023-03-16T21:58:09.097618Z","shell.execute_reply":"2023-03-16T21:58:09.105872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Model\nmodel_example = ResNet50Network(output_size=output_size, no_columns=no_columns).to(DEVICE)\n\n# Outputs\nout = model_example(image, meta, prints=True)\n\n# Criterion example\ncriterion_example = nn.BCEWithLogitsLoss()\n# Unsqueeze(1) from shape=[3] to shape=[3, 1]\nloss = criterion_example(out, targets.unsqueeze(1).float()) \nprint(\"=\"*50)\nprint(clr.S+'Loss:'+clr.E, loss.item())","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:58:35.218849Z","iopub.execute_input":"2023-03-16T21:58:35.219467Z","iopub.status.idle":"2023-03-16T21:58:41.639718Z","shell.execute_reply.started":"2023-03-16T21:58:35.219428Z","shell.execute_reply":"2023-03-16T21:58:41.638509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EffNetNetwork(nn.Module):\n    def __init__(self, output_size, no_columns):\n        super().__init__()\n        self.no_columns, self.output_size = no_columns, output_size\n        \n        # Define Feature part (IMAGE)\n        self.features = EfficientNet.from_pretrained('efficientnet-b2')\n        \n        # (CSV)\n        self.csv = nn.Sequential(nn.Linear(self.no_columns, 250),\n                                 nn.BatchNorm1d(250),\n                                 nn.ReLU(),\n                                 nn.Dropout(p=0.2),\n                                 \n                                 nn.Linear(250, 250),\n                                 nn.BatchNorm1d(250),\n                                 nn.ReLU(),\n                                 nn.Dropout(p=0.2))\n        \n        # Define Classification part\n        self.classification = nn.Sequential(nn.Linear(1408 + 250, self.output_size))\n        \n        \n    def forward(self, image, meta, prints=False):   \n        \n        if prints: print('Input Image shape:', image.shape, '\\n'+\n                         'Input metadata shape:', meta.shape)\n        \n        # Image CNN\n        image = self.features.extract_features(image)\n        image = F.avg_pool2d(image, image.size()[2:]).reshape(-1, 1408)\n        if prints: print('Features Image shape:', image.shape)\n        \n        # CSV FNN\n        meta = self.csv(meta)\n        if prints: print('Meta Data:', meta.shape)\n            \n        # Concatenate layers from image with layers from csv_data\n        image_meta_data = torch.cat((image, meta), dim=1)\n        if prints: print('Concatenated Data:', image_meta_data.shape)\n        \n        # CLASSIF\n        out = self.classification(image_meta_data)\n        if prints: print('Out shape:', out.shape)\n        \n        return out","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:59:15.523099Z","iopub.execute_input":"2023-03-16T21:59:15.524183Z","iopub.status.idle":"2023-03-16T21:59:15.535604Z","shell.execute_reply.started":"2023-03-16T21:59:15.524144Z","shell.execute_reply":"2023-03-16T21:59:15.534021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Model\nmodel_example2 = EffNetNetwork(output_size=output_size, no_columns=no_columns).to(DEVICE)\n\n# Outputs\nout = model_example2(image, meta, prints=True)\n\n# Criterion example\ncriterion_example = nn.BCEWithLogitsLoss()\n# Unsqueeze(1) from shape=[3] to shape=[3, 1]\nloss = criterion_example(out, targets.unsqueeze(1).float()) \nprint(\"=\"*50)\nprint(clr.S+'Loss:'+clr.E, loss.item())","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:59:32.31917Z","iopub.execute_input":"2023-03-16T21:59:32.319869Z","iopub.status.idle":"2023-03-16T21:59:33.5226Z","shell.execute_reply.started":"2023-03-16T21:59:32.319832Z","shell.execute_reply":"2023-03-16T21:59:33.521465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"1408+250","metadata":{"execution":{"iopub.status.busy":"2023-03-16T21:59:51.597723Z","iopub.execute_input":"2023-03-16T21:59:51.598122Z","iopub.status.idle":"2023-03-16T21:59:51.605085Z","shell.execute_reply.started":"2023-03-16T21:59:51.598089Z","shell.execute_reply":"2023-03-16T21:59:51.603662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_in_file(text, f):\n    \n    with open(f'logs_{VERSION}.txt', 'a+') as f:\n        print(text, file=f)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T22:00:40.75096Z","iopub.execute_input":"2023-03-16T22:00:40.751963Z","iopub.status.idle":"2023-03-16T22:00:40.758679Z","shell.execute_reply.started":"2023-03-16T22:00:40.751923Z","shell.execute_reply":"2023-03-16T22:00:40.757494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_folds(model, train_original):\n    # Creates a .txt file that will contain the logs\n    # logs == what we also print to console\n    f = open(f\"logs_{VERSION}.txt\", \"w+\")\n    \n    # Split in folds\n    group_fold = GroupKFold(n_splits = FOLDS)\n\n    # Generate indices to split data into training and test set.\n    k_folds = group_fold.split(X = np.zeros(len(train_original)), \n                               y = train_original['cancer'], \n                               groups = train_original['patient_id'].tolist())\n    \n    # For each fold\n    for i, (train_index, valid_index) in enumerate(k_folds):\n        \n        print(clr.S+f\"---------- Fold: {i+1} ----------\"+clr.E)\n        add_in_file(f\"---------- Fold: {i+1} ----------\", f)\n        \n        # 🐝 W&B Tracking\n        RUN_CONFIG = CONFIG.copy()\n        params = dict(model=MODEL, \n                      version=VERSION,\n                      fold=i,\n                      epochs=EPOCHS, \n                      batch=BATCH_SIZE1,\n                      lr=LR,\n                      weight_decay=WD)\n        RUN_CONFIG.update(params)\n        run = wandb.init(project='RSNA_Breast_Cancer', config=RUN_CONFIG)\n\n        wandb.watch(model, log_freq=100) # 🐝\n\n        # --- Create Instances ---\n        # Best ROC score in this fold\n        best_roc = None\n        # Reset patience before every fold\n        patience_f = PATIENCE\n\n        # Optimizer/ Scheduler/ Criterion\n        optimizer = torch.optim.Adam(model.parameters(), lr = LR, \n                                     weight_decay=WD)\n        scheduler = ReduceLROnPlateau(optimizer=optimizer, mode='max', \n                                      patience=LR_PATIENCE, verbose=True, factor=LR_FACTOR)\n        criterion = nn.BCEWithLogitsLoss()\n\n\n        # --- Read in Data ---\n        train_data = train_original.iloc[train_index].reset_index(drop=True)\n        valid_data = train_original.iloc[valid_index].reset_index(drop=True)\n\n        # Create Data instances\n        train = RSNADataset(train_data, vertical_flip, horizontal_flip, \n                            is_train=True)\n        valid = RSNADataset(valid_data, vertical_flip, horizontal_flip,\n                            is_train=True)\n\n        # Dataloaders\n        train_loader = DataLoader(train, batch_size=BATCH_SIZE1, \n                                  shuffle=True, num_workers=WORKERS)\n        valid_loader = DataLoader(valid, batch_size=BATCH_SIZE2, \n                                  shuffle=False, num_workers=WORKERS)\n\n\n        # === EPOCHS ===\n        for epoch in range(EPOCHS):\n            start_time = time()\n            correct = 0\n            train_losses = 0\n\n            # === TRAIN ===\n            # Sets the module in training mode.\n            model.train()\n\n            # For each batch\n            for k, data in tqdm(enumerate(train_loader)):\n                # Save them to device\n                image, meta, targets = data_to_device(data)\n\n                # Clear gradients first; very important\n                # usually done BEFORE prediction\n                optimizer.zero_grad()\n\n                # Log Probabilities & Backpropagation\n                out = model(image, meta)\n                loss = criterion(out, targets.unsqueeze(1).float())\n                loss.backward()\n                optimizer.step()\n\n                # --- Save information after this batch ---\n                # Save loss\n                train_losses += loss.item()\n                wandb.log({\"train_loss\": loss.item()}, step=epoch) # 🐝\n                # From log probabilities to actual probabilities\n                train_preds = torch.round(torch.sigmoid(out)) # 0 and 1\n                # Number of correct predictions\n                correct += (train_preds.cpu() == targets.cpu().unsqueeze(1)).sum().item()\n\n            # Compute Train Accuracy\n            train_acc = correct / len(train_index)\n            wandb.log({\"train_acc\": train_acc}) # 🐝\n\n\n            # === EVAL ===\n            # Sets the model in evaluation mode.\n            model.eval()\n\n            # Create matrix to store evaluation predictions (for accuracy)\n            valid_preds = torch.zeros(size = (len(valid_index), 1), \n                                      device=DEVICE, dtype=torch.float32)\n\n\n            # Disables gradients (we need to be sure no optimization happens)\n            with torch.no_grad():\n                for k, data in tqdm(enumerate(valid_loader)):\n                    # Save them to device\n                    image, meta, targets = data_to_device(data)\n\n                    out = model(image, meta)\n                    pred = torch.sigmoid(out)\n                    valid_preds[k*image.shape[0] : k*image.shape[0] + image.shape[0]] = pred\n\n                # Calculate accuracy\n                valid_acc = accuracy_score(valid_data['cancer'].values, \n                                           torch.round(valid_preds.cpu()))\n                wandb.log({\"valid_acc\": valid_acc}) # 🐝\n                # Calculate ROC\n                valid_roc = roc_auc_score(valid_data['cancer'].values, \n                                          valid_preds.cpu())\n                wandb.log({\"valid_roc\": valid_roc}) # 🐝\n\n                # Calculate time on Train + Eval\n                duration = str(dtime.timedelta(seconds=time() - start_time))[:7]\n\n\n                # PRINT INFO\n                final_logs = '{} | Epoch: {}/{} | Loss: {:.4} | Acc_tr: {:.3} | Acc_vd: {:.3} | ROC: {:.3}'.\\\n                                format(duration, epoch+1, EPOCHS, \n                                       train_losses, train_acc, valid_acc, valid_roc)\n                add_in_file(final_logs,f)\n                print(final_logs)\n\n\n                # === SAVE MODEL ===\n\n                # Update scheduler (for learning_rate)\n                scheduler.step(valid_roc)\n                # Name the model\n                model_name = f\"Fold{i+1}_Epoch{epoch+1}_ValidAcc{valid_acc:.3f}_ROC{valid_roc:.3f}.pth\"\n\n                # Update best_roc\n                if not best_roc: # If best_roc = None\n                    best_roc = valid_roc\n                    torch.save(model.state_dict(), model_name)\n                    continue\n\n                if valid_roc > best_roc:\n                    best_roc = valid_roc\n                    # Reset patience (because we have improvement)\n                    patience_f = PATIENCE\n                    torch.save(model.state_dict(), model_name)\n                else:\n                    # Decrease patience (no improvement in ROC)\n                    patience_f = patience_f - 1\n                    if patience_f == 0:\n                        stop_logs = 'Early stopping (no improvement since 3 models) | Best ROC: {}'.\\\n                                    format(best_roc)\n                        add_in_file(stop_logs, f)\n                        print(stop_logs)\n                        break\n\n\n        # === CLEANING ===\n        # Clear memory\n        del train, valid, train_loader, valid_loader, image, targets\n        gc.collect()\n        \n        # 🐝 Experiment End for this fold\n        wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2023-03-16T22:01:25.168964Z","iopub.execute_input":"2023-03-16T22:01:25.16936Z","iopub.status.idle":"2023-03-16T22:01:25.194118Z","shell.execute_reply.started":"2023-03-16T22:01:25.169327Z","shell.execute_reply":"2023-03-16T22:01:25.192985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDS = 3\nEPOCHS = 3\nPATIENCE = 3\nWORKERS = 8\nLR = 0.0005\nWD = 0.0\nLR_PATIENCE = 1            # 1 model not improving until lr is decreasing\nLR_FACTOR = 0.4            # by how much the lr is decreasing\n\nBATCH_SIZE1 = 32           # for train\nBATCH_SIZE2 = 16           # for valid\n\nVERSION = 'v1'\nMODEL = 'resnet50'\n\nmodel1 = ResNet50Network(output_size=output_size, no_columns=no_columns).to(DEVICE)\n\n# ------------------\n\n# Run the cell below to train\n# Ran it locally on all data, see the results below\n# train_folds(model=model1, train_original=train)\n\n# Print the logs during training\nf = open('/kaggle/input/rsna-breast-cancer-helper-data/logs_v1.txt', \"r\")\ncontents = f.read()\nprint(contents)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T22:01:43.01091Z","iopub.execute_input":"2023-03-16T22:01:43.011312Z","iopub.status.idle":"2023-03-16T22:01:43.727848Z","shell.execute_reply.started":"2023-03-16T22:01:43.01128Z","shell.execute_reply":"2023-03-16T22:01:43.726872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDS = 3\nEPOCHS = 3\nPATIENCE = 3\nWORKERS = 8\nLR = 0.0005\nWD = 0.0\nLR_PATIENCE = 1            # 1 model not improving until lr is decreasing\nLR_FACTOR = 0.4            # by how much the lr is decreasing\n\nBATCH_SIZE1 = 32           # for train\nBATCH_SIZE2 = 16           # for valid\n\nVERSION = 'v2'\nMODEL = 'effnet'\n\nmodel2 = EffNetNetwork(output_size=output_size, no_columns=no_columns).to(DEVICE)\n\n# ------------------\n\n# Run the cell below to train\n# Ran it locally on all data, see the results below\n# train_folds(model=model2, train_original=train)\n\n# Print the logs during training\nf = open('/kaggle/input/rsna-breast-cancer-helper-data/logs_v2.txt', \"r\")\ncontents = f.read()\nprint(contents)","metadata":{"execution":{"iopub.status.busy":"2023-03-16T22:02:03.894064Z","iopub.execute_input":"2023-03-16T22:02:03.894427Z","iopub.status.idle":"2023-03-16T22:02:04.09041Z","shell.execute_reply.started":"2023-03-16T22:02:03.894398Z","shell.execute_reply":"2023-03-16T22:02:04.08927Z"},"trusted":true},"execution_count":null,"outputs":[]}]}