{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!unzip -q ../input/timm-with-dependencies/timm_all -d timm-with-dependencies\n!pip install --no-index --find-links timm-with-dependencies timm\n!pip install /kaggle/input/dicomsdl-offline-installer/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2023-02-03T21:38:14.317424Z","iopub.execute_input":"2023-02-03T21:38:14.318746Z","iopub.status.idle":"2023-02-03T21:38:41.690713Z","shell.execute_reply.started":"2023-02-03T21:38:14.318681Z","shell.execute_reply":"2023-02-03T21:38:41.688949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importing necessary libraries\nimport numpy as np\nimport pandas as pd\nimport os\nfrom fastai.vision.learner import *\nfrom fastai.data.all import *\nfrom fastai.vision.all import *\nfrom fastai.metrics import ActivationType\nfrom sklearn.model_selection import StratifiedKFold\nfrom collections import defaultdict\nfrom pdb import set_trace","metadata":{"execution":{"iopub.status.busy":"2023-02-03T22:08:42.195416Z","iopub.execute_input":"2023-02-03T22:08:42.195856Z","iopub.status.idle":"2023-02-03T22:08:42.287379Z","shell.execute_reply.started":"2023-02-03T22:08:42.195824Z","shell.execute_reply":"2023-02-03T22:08:42.286167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-03T21:42:54.709735Z","iopub.execute_input":"2023-02-03T21:42:54.710468Z","iopub.status.idle":"2023-02-03T21:42:54.859968Z","shell.execute_reply.started":"2023-02-03T21:42:54.71043Z","shell.execute_reply":"2023-02-03T21:42:54.858692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-03T21:43:35.305741Z","iopub.execute_input":"2023-02-03T21:43:35.30618Z","iopub.status.idle":"2023-02-03T21:43:35.327648Z","shell.execute_reply.started":"2023-02-03T21:43:35.306142Z","shell.execute_reply":"2023-02-03T21:43:35.326705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/input/rsna-breast-cancer-detection/train_images')[:10]","metadata":{"execution":{"iopub.status.busy":"2023-02-03T21:45:06.493036Z","iopub.execute_input":"2023-02-03T21:45:06.493485Z","iopub.status.idle":"2023-02-03T21:45:06.507782Z","shell.execute_reply.started":"2023-02-03T21:45:06.493449Z","shell.execute_reply":"2023-02-03T21:45:06.506481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_id = '10706'\nos.listdir(f'/kaggle/input/rsna-breast-cancer-detection/train_images/{patient_id}')","metadata":{"execution":{"iopub.status.busy":"2023-02-03T21:46:32.23908Z","iopub.execute_input":"2023-02-03T21:46:32.240548Z","iopub.status.idle":"2023-02-03T21:46:32.249357Z","shell.execute_reply.started":"2023-02-03T21:46:32.240489Z","shell.execute_reply":"2023-02-03T21:46:32.248447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# building the path to get a single image. can get different paths by changing idx\nidx = 42\n\nbase_img_dir = '/kaggle/input/rsna-breast-cancer-detection/train_images'\n\nimg_id = str(train_df['image_id'].iloc[idx])\nfull_img_id = img_id + '.dcm'\npat_id = str(train_df['patient_id'].iloc[idx])\n\nlabel = train_df['cancer'].iloc[idx]\n\nos.path.join(base_img_dir, pat_id, full_img_id)","metadata":{"execution":{"iopub.status.busy":"2023-02-03T21:59:05.513449Z","iopub.execute_input":"2023-02-03T21:59:05.513862Z","iopub.status.idle":"2023-02-03T21:59:05.523436Z","shell.execute_reply.started":"2023-02-03T21:59:05.51383Z","shell.execute_reply":"2023-02-03T21:59:05.522504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving one path into the img_path variable\nimg_path = os.path.join(base_img_dir, pat_id, full_img_id)","metadata":{"execution":{"iopub.status.busy":"2023-02-03T21:59:05.946096Z","iopub.execute_input":"2023-02-03T21:59:05.946837Z","iopub.status.idle":"2023-02-03T21:59:05.952864Z","shell.execute_reply.started":"2023-02-03T21:59:05.946768Z","shell.execute_reply":"2023-02-03T21:59:05.951635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reading this file using pydicom, viewing the metadata\ndcm_img = pydicom.dcmread(img_path, force=True)\ndcm_img","metadata":{"execution":{"iopub.status.busy":"2023-02-03T21:59:06.34546Z","iopub.execute_input":"2023-02-03T21:59:06.346773Z","iopub.status.idle":"2023-02-03T21:59:06.363693Z","shell.execute_reply.started":"2023-02-03T21:59:06.346703Z","shell.execute_reply":"2023-02-03T21:59:06.362607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_array = dcm_img.pixel_array\n\nplt.imshow(img_array)\nif label == 0:\n    category = \"doesn't have cancer\"\nelif label == 1:\n    category = \"has cancer\"\n    \nplt.title(f'Patient {pat_id} {category} in image: {img_id}');","metadata":{"execution":{"iopub.status.busy":"2023-02-03T21:59:06.848919Z","iopub.execute_input":"2023-02-03T21:59:06.850406Z","iopub.status.idle":"2023-02-03T21:59:09.35794Z","shell.execute_reply.started":"2023-02-03T21:59:06.850356Z","shell.execute_reply":"2023-02-03T21:59:09.352999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def random_9img_sample(\n    df: pd.DataFrame,\n    random_state: int = 101,\n    base_img_dir = base_img_dir):\n    \"\"\"Function shows 9 random images with labels and their img sizes\"\"\"\n    df_sample = df.sample(9,random_state=random_state).reset_index()\n    \n    fig, ax = plt.subplots(3,3,figsize=(14,14))\n    \n    for idx_num, row in df_sample.iterrows():\n        img_id = str(row['image_id'])\n        full_img_id = img_id + '.dcm'\n        pat_id = str(row['patient_id'])\n        \n        label = row['cancer']\n        \n        img_path = os.path.join(base_img_dir,pat_id,full_img_id)\n        q, r = divmod(idx_num, 3)\n        \n        img_array = pydicom.dcmread(img_path, force=True).pixel_array\n        \n        ax[q][r].imshow(img_array)\n        ax[q][r].set_title(f'Image label is: {label}')","metadata":{"execution":{"iopub.status.busy":"2023-02-03T22:06:33.153134Z","iopub.execute_input":"2023-02-03T22:06:33.153599Z","iopub.status.idle":"2023-02-03T22:06:33.16347Z","shell.execute_reply.started":"2023-02-03T22:06:33.153567Z","shell.execute_reply":"2023-02-03T22:06:33.161732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now we can visualize 9 random images!\nrandom_9img_sample(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-02-03T22:07:42.757823Z","iopub.execute_input":"2023-02-03T22:07:42.758234Z","iopub.status.idle":"2023-02-03T22:07:54.45145Z","shell.execute_reply.started":"2023-02-03T22:07:42.758201Z","shell.execute_reply":"2023-02-03T22:07:54.449978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now lets visualize how balanced/unbalanced this dataset is\nsns.countplot(data=train_df,x='cancer')","metadata":{"execution":{"iopub.status.busy":"2023-02-03T22:09:45.704369Z","iopub.execute_input":"2023-02-03T22:09:45.704838Z","iopub.status.idle":"2023-02-03T22:09:45.894231Z","shell.execute_reply.started":"2023-02-03T22:09:45.704801Z","shell.execute_reply":"2023-02-03T22:09:45.892965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# only about 2% of the dataset shows positive cancer cases\ntrain_df['cancer'].value_counts(normalize = True)","metadata":{"execution":{"iopub.status.busy":"2023-02-03T22:10:22.186058Z","iopub.execute_input":"2023-02-03T22:10:22.186508Z","iopub.status.idle":"2023-02-03T22:10:22.197411Z","shell.execute_reply.started":"2023-02-03T22:10:22.186471Z","shell.execute_reply":"2023-02-03T22:10:22.196035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}