{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"print('run')","metadata":{"execution":{"iopub.status.busy":"2023-01-29T15:17:37.211264Z","iopub.execute_input":"2023-01-29T15:17:37.211619Z","iopub.status.idle":"2023-01-29T15:17:37.265639Z","shell.execute_reply.started":"2023-01-29T15:17:37.21152Z","shell.execute_reply":"2023-01-29T15:17:37.264643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install kaggle","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-22T16:12:41.32597Z","iopub.execute_input":"2023-02-22T16:12:41.326928Z","iopub.status.idle":"2023-02-22T16:13:42.831697Z","shell.execute_reply.started":"2023-02-22T16:12:41.326816Z","shell.execute_reply":"2023-02-22T16:13:42.83071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install colabcode","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from colabcode import ColabCode\nColabCode(port = 1234)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing libraries\nimport numpy as np \nimport pandas as pd \nimport shutil\nimport os\nimport cv2\nimport glob\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport random\nimport csv\nimport tensorflow as tf\nimport torch\nimport pydicom as dicom\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom torch import nn, optim\nfrom torchvision import transforms, datasets, models\nimport torch.nn.functional as F\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-02-22T16:14:57.426984Z","iopub.execute_input":"2023-02-22T16:14:57.427364Z","iopub.status.idle":"2023-02-22T16:15:04.428889Z","shell.execute_reply.started":"2023-02-22T16:14:57.427332Z","shell.execute_reply":"2023-02-22T16:15:04.427934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(r\"../input/rsna-breast-cancer-detection/train.csv\")\ntest_df = pd.read_csv(r\"../input/rsna-breast-cancer-detection/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-02-22T16:15:04.430891Z","iopub.execute_input":"2023-02-22T16:15:04.431793Z","iopub.status.idle":"2023-02-22T16:15:04.538396Z","shell.execute_reply.started":"2023-02-22T16:15:04.431755Z","shell.execute_reply":"2023-02-22T16:15:04.537471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-22T16:15:04.539584Z","iopub.execute_input":"2023-02-22T16:15:04.539953Z","iopub.status.idle":"2023-02-22T16:15:04.563118Z","shell.execute_reply.started":"2023-02-22T16:15:04.539917Z","shell.execute_reply":"2023-02-22T16:15:04.562112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA\n## Training data\nLet us understand the data before proceeding","metadata":{}},{"cell_type":"code","source":"print(train_df['laterality'].unique())\nprint(train_df['laterality'].value_counts())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#site_id - ID code for the source hospital.\nprint(\"site_id\")\nprint(train_df[\"site_id\"].unique())\nprint(len(train_df[\"site_id\"].unique()))\nprint(train_df[\"site_id\"].value_counts())\n\n# Number of unique patients\nprint(\"\\n Number of unique patients\")\nprint(len(train_df[\"patient_id\"].unique()))\n\n#image_id - ID code for the image.\nprint(\"\\n ID code for the image\")\nprint(len(train_df[\"image_id\"].unique()))\n\n# laterality - Whether the image is of the left or right breast.\nprint(\"\\n laterality - Whether the image is of the left or right breast.\")\nprint(train_df['laterality'].unique())\nprint(train_df['laterality'].value_counts())\n\n#view - The orientation of the image. The default for a screening exam is to capture two views per breast.\nprint(\"\\n view - The orientation of the image. The default for a screening exam is to capture two views per breast.\")\nprint(train_df['view'].value_counts())\n\n#age - The patient's age in years.\nprint(\"\\n age - The Mean patient's age in years\")\nprint(train_df['age'].mean())\n\n#implant - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\nprint(\"\\n implant\")\nprint(train_df[\"implant\"].value_counts())\n\n#density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train.\nprint(\"\\n density\")\nprint(train_df['density'].value_counts())\n\n#machine_id - An ID code for the imaging device.\nprint(\"\\n machine_id\")\nprint(len(train_df['machine_id'].unique()))\n\n# cancer - Whether or not the breast was positive for malignant cancer. The target value. Only provided for train.\nprint(\"\\n cancer\")\nprint(len(train_df['cancer'].unique()))\nprint(train_df['cancer'].value_counts())\n\n# biopsy - Whether or not a follow-up biopsy was performed on the breast. Only provided for train.\nprint(\"\\n biopsy\")\nprint(train_df['cancer'].value_counts())\n\n# invasive - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\nprint(\"\\n invasive\")\nprint(len(train_df['invasive'].unique()))\nprint(train_df['invasive'].value_counts())\n\n# BIRADS - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\nprint(\"\\n BIRADS\")\nprint(train_df['BIRADS'].value_counts())\n\n# difficult_negative_case - True if the case was unusually difficult. Only provided for train.\nprint(\"\\n True if the case was unusually difficult\")\nprint(train_df['difficult_negative_case'].value_counts())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# seg","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating directories for classes\nos.makedirs(r\"../working/0\",exist_ok=True)\nos.makedirs(r\"../working/1\",exist_ok=True)\n# saving out directories as variables\nout_cancer = r\"../working/1\"\nout_nocancer = r\"../working/0\"\ninput_path = r\"../input/rsna-breast-cancer-detection/train_images\"\n# creating a small training set for initial experimentation\npatient_used = []\nnumber_no_cancer = 0\nnumber_no_cancer_bio = 0\nnumber_no_cancer_no_bio = 0\n# shuffling dataframe\ntrain_df_sh = train_df.sample(frac=1).reset_index(drop=True)\n# creating the dataframe\nfor index, row in train_df_sh.iterrows():\n    # if the image corresponds to a patient with cancer\n    if row[\"cancer\"] == 1:\n        # defining the patient id as variable\n        patient_id = str(row[\"patient_id\"])\n        # defining the image name as variable\n        img_name = str(row[\"image_id\"])+\".dcm\"\n        # output image name - combination of patiend id + image\n        out_name = patient_id+\"_\"+img_name\n        # copy this image\n        shutil.copy(os.path.join(input_path,patient_id,img_name),os.path.join(out_cancer,out_name))\n    else:    \n        # obtaining the patient id  \n        patient_id = str(row[\"patient_id\"])\n        # if we have already extracted one image from this patient\n        if patient_id in patient_used:\n            # skip it\n            continue\n        else:\n            # appending the patient so we do not use it in the future\n            patient_used.append(patient_id)\n            # if we have copied as many non-cancer images as cancer images, break\n            if number_no_cancer >= 1158:\n                continue\n            # grabbing the biopsy variable\n            biopsy = str(row[\"biopsy\"])\n            # if this image corresponds to a biopsy case\n            if biopsy == \"1\":\n                # we increase the biopsy images counter \n                number_no_cancer_bio = number_no_cancer_bio + 1\n                # if the number of biopsy images greater than 50%, we skip it\n                if number_no_cancer_bio > int(1158/2)+1:\n                    continue\n                else:\n                    # else, we include this image in the collection\n                    # defining the img name\n                    img_name = str(row[\"image_id\"])+\".dcm\"\n                    out_name = patient_id+\"_\"+img_name\n                    # copying the image\n                    shutil.copy(os.path.join(input_path,patient_id,img_name),os.path.join(out_nocancer,out_name))\n                    # adding +1 in the non cancer images collected\n                    number_no_cancer += 1\n            # if the image do not contain biopsy, same\n            else:\n                # increasing the counter\n                number_no_cancer_no_bio = number_no_cancer_no_bio + 1\n                # if we have already grabbed more than 50% of the dataset, skip it\n                if number_no_cancer_no_bio > int(1158/2)+1:\n                    continue\n                else:\n                    # else, we include this image in the collection\n                    # defining the image name\n                    img_name = str(row[\"image_id\"])+\".dcm\"\n                    out_name = patient_id+\"_\"+img_name\n                    # copying\n                    shutil.copy(os.path.join(input_path,patient_id,img_name),os.path.join(out_nocancer,out_name))\n                    # adding +1 in the non cancer images collected\n                    number_no_cancer += 1","metadata":{"execution":{"iopub.status.busy":"2023-02-22T16:15:05.848873Z","iopub.execute_input":"2023-02-22T16:15:05.849329Z","iopub.status.idle":"2023-02-22T16:17:35.267943Z","shell.execute_reply.started":"2023-02-22T16:15:05.849288Z","shell.execute_reply":"2023-02-22T16:17:35.266928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of no cancer images: \"+str(len(os.listdir(r\"../working/0\"))))\nprint(\"Number of cancer images: \"+str(len(os.listdir(r\"../working/1\"))))","metadata":{"execution":{"iopub.status.busy":"2023-02-22T16:17:35.269992Z","iopub.execute_input":"2023-02-22T16:17:35.270366Z","iopub.status.idle":"2023-02-22T16:17:35.280964Z","shell.execute_reply.started":"2023-02-22T16:17:35.27033Z","shell.execute_reply":"2023-02-22T16:17:35.27959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs(os.path.join(\"../working/training\",\"0\"))\nos.makedirs(os.path.join(\"../working/training\",\"1\"))\nos.makedirs(os.path.join(\"../working/val\",\"0\"))\nos.makedirs(os.path.join(\"../working/val\",\"1\"))","metadata":{"execution":{"iopub.status.busy":"2023-02-22T16:17:35.282882Z","iopub.execute_input":"2023-02-22T16:17:35.283333Z","iopub.status.idle":"2023-02-22T16:17:35.292976Z","shell.execute_reply.started":"2023-02-22T16:17:35.283275Z","shell.execute_reply":"2023-02-22T16:17:35.292012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = ''\ntrain_dir = data_dir + '/training'\nvalid_dir = data_dir + '/validation'","metadata":{"execution":{"iopub.status.busy":"2023-02-22T16:17:35.296557Z","iopub.execute_input":"2023-02-22T16:17:35.296842Z","iopub.status.idle":"2023-02-22T16:17:35.303327Z","shell.execute_reply.started":"2023-02-22T16:17:35.296818Z","shell.execute_reply":"2023-02-22T16:17:35.30242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -zcvf outputname.tar.gz /kaggle/working","metadata":{"execution":{"iopub.status.busy":"2023-02-22T16:21:11.656211Z","iopub.execute_input":"2023-02-22T16:21:11.656581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}