{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"color:white;\n           display:fill;\n           border-radius:2px;\n           background-color:Black;\n           font-size:250%;\n           font-family:Verdana;\n           letter-spacing:0.5px\">\n<p style=\"padding: 10px;\n          text-align: center;\n          font-size:120%;\n          color:brown;\">\n           🎨RSNA Breast Cancer Detection📊\n</p>\n<style>\n        h1{text-align: center;}\n </style>  \n    \n</div> \n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"<div style=\"color:white;\n           display:fill;\n           border-radius:5px;\n           background-color:skyblue;\n           font-size:250%;\n           font-family:Verdana;\n           letter-spacing:0.5px\">\n\n<p style=\"padding:3px;\n          text-align: center;\n          font-size:150%;\n          color:blue;\">\n           📚📚About The Dataset 🧾📑\n</p>\n<style>\n        h1{text-align: center;}\n </style>  ","metadata":{"execution":{"iopub.status.busy":"2022-11-29T14:29:18.537695Z","iopub.execute_input":"2022-11-29T14:29:18.538124Z","iopub.status.idle":"2022-11-29T14:29:18.546685Z","shell.execute_reply.started":"2022-11-29T14:29:18.538089Z","shell.execute_reply":"2022-11-29T14:29:18.545013Z"}}},{"cell_type":"markdown","source":"<div style=\"color:white;\n           display:fill;\n           border-radius:5px;\n           background-color:#00838F;\n           font-size:110%;\n           font-family:Verdana;\n           letter-spacing:0.5px\">\n\n<p style=\"padding: 10px;\n              color:white;\">\n   🛰😎Goal of the Competition📙📚\n🎲🔈The goal of this competition is to identify breast cancer. You'll train your model with screening mammograms obtained from regular screening.🎆🤸🏽‍♂️Your work improving the automation of detection in screening mammography may enable radiologists to be more accurate and efficient,▶📙🛰improving the quality and safety of patient care. It could also help reduce costs and unnecessary medical procedures.🚀🎭🛰\n</p>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:white;\n           display:fill;\n           border-radius:5px;\n           background-color:black;\n           font-size:110%;\n           font-family:Verdana;\n           letter-spacing:0.5px\">\n\n<p style=\"padding: 10px;\n              color:white;\">\n👩🏽‍💻💎Context📺\nAccording to the WHO,🛰🚀breast cancer is the most commonly occurring cancer worldwide. In 2020 alone, there were 2.3 million new breast cancer diagnoses and 685,000 deaths.🐓⚡Yet breast cancer mortality in high-income countries has dropped by 40% since the 1980s when health authorities implemented regular mammography screening in age groups considered at risk.🎭📕📖 Early detection and treatment are critical to reducing cancer fatalities, and your machine learning skills could help streamline the process radiologists use to evaluate screening mammograms🥽🙏🏽Currently,🏑👞early detection of breast cancer requires the expertise of highly-trained human observers, making screening mammography programs expensive to conduct.💎📺🛰 A looming shortage of radiologists in several countries will likely worsen this problem.🚌🛺Mammography screening also leads to a high incidence of false positive results. This can result in unnecessary anxiety, inconvenient follow-up care, extra imaging tests, and sometimes a need for tissue sampling (often a needle biopsy).🚍🏍The competition host, the Radiological Society of North America (RSNA) is a non-profit organization that represents 31 radiologic subspecialties from 145 countries around the world.🥞🍔RSNA promotes excellence in patient care and health care delivery through education, research, and technological innovation.🥖🧀🍗\n\n\n</p>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:white;\n           display:fill;\n           border-radius:5px;\n           background-color:green;\n           font-size:250%;\n           font-family:Verdana;\n           letter-spacing:0.5px\">\n\n<p style=\"padding:3px;\n          text-align: center;\n          font-size:100%;\n          color:yellow;\">\n          📖Loading Library and Reading dataset💎\n</p>\n<style>\n        h1{text-align: center;}\n </style>  \n","metadata":{"execution":{"iopub.status.busy":"2022-11-29T14:32:28.999392Z","iopub.execute_input":"2022-11-29T14:32:28.999745Z","iopub.status.idle":"2022-11-29T14:32:29.007546Z","shell.execute_reply.started":"2022-11-29T14:32:28.999717Z","shell.execute_reply":"2022-11-29T14:32:29.006037Z"}}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport cv2\nimport glob\nimport pydicom\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:09.944341Z","iopub.execute_input":"2022-12-03T16:10:09.944895Z","iopub.status.idle":"2022-12-03T16:10:11.023489Z","shell.execute_reply.started":"2022-12-03T16:10:09.944789Z","shell.execute_reply":"2022-12-03T16:10:11.022289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image = '/kaggle/input/rsna-breast-cancer-detection/train_images'\ntest_image  = '/kaggle/input/rsna-breast-cancer-detection/test_images'","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:11.028041Z","iopub.execute_input":"2022-12-03T16:10:11.028515Z","iopub.status.idle":"2022-12-03T16:10:11.034174Z","shell.execute_reply.started":"2022-12-03T16:10:11.028452Z","shell.execute_reply":"2022-12-03T16:10:11.032975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"color:white;\n           display:fill;\n           border-radius:5px;\n           background-color:#00838F;\n           font-size:110%;\n           font-family:Verdana;\n           letter-spacing:0.5px\">\n\n<p style=\"padding: 20px;\n              color:white;\">\n 🥙🥪Glob Usage:🍣🍱In Python, the glob module is used to retrieve files/pathnames matching a specified pattern. \n 🤗😎Resources to go Through🥞🏍🚌 https://www.geeksforgeeks.org/how-to-use-glob-function-to-find-files-recursively-in-python/\n</p>\n</div>","metadata":{}},{"cell_type":"code","source":"# Reading the files through glob\n# In Python, the glob module is used to retrieve files/pathnames matching a specified pattern. \n# https://www.geeksforgeeks.org/how-to-use-glob-function-to-find-files-recursively-in-python/\nimg = glob.glob(train_image+'/10**/*.dcm') # Load all the images in the training folder\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:11.035543Z","iopub.execute_input":"2022-12-03T16:10:11.035923Z","iopub.status.idle":"2022-12-03T16:10:14.719119Z","shell.execute_reply.started":"2022-12-03T16:10:11.035892Z","shell.execute_reply":"2022-12-03T16:10:14.718133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To check on the dcom files what it contain can be read and showed as follows\n\nimport pydicom as dicom # Get the dicom library to read .dcm files\nimg_read = dicom.dcmread(img[0]) # Read only one images , can be done to load random image to load from folder\nplt.imshow(img_read.pixel_array)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:14.721147Z","iopub.execute_input":"2022-12-03T16:10:14.721926Z","iopub.status.idle":"2022-12-03T16:10:16.344871Z","shell.execute_reply.started":"2022-12-03T16:10:14.72189Z","shell.execute_reply":"2022-12-03T16:10:16.343641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_read # Here what is the information contained in the .dcm files","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.346142Z","iopub.execute_input":"2022-12-03T16:10:16.34652Z","iopub.status.idle":"2022-12-03T16:10:16.358617Z","shell.execute_reply.started":"2022-12-03T16:10:16.346455Z","shell.execute_reply":"2022-12-03T16:10:16.357509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets write a small function to load a set of images randomly and display them \nfrom pydicom import dcmread\n\ndef dcm_image(file): \n    for files in file:             \n        fig,axes = plt.subplots(2,2,figsize=(10,10)) \n        for i,ax in enumerate(axes.reshape(-1)): \n            ds = dcmread(files)   \n            ax.imshow(ds.pixel_array)\n        plt.show()\n        return ","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.360168Z","iopub.execute_input":"2022-12-03T16:10:16.360777Z","iopub.status.idle":"2022-12-03T16:10:16.369277Z","shell.execute_reply.started":"2022-12-03T16:10:16.360738Z","shell.execute_reply":"2022-12-03T16:10:16.367791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#k = np.random.choice(img)\n#ds = dcmread(k)\n#p = ds.pixel_array\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.371024Z","iopub.execute_input":"2022-12-03T16:10:16.371911Z","iopub.status.idle":"2022-12-03T16:10:16.381352Z","shell.execute_reply.started":"2022-12-03T16:10:16.371872Z","shell.execute_reply":"2022-12-03T16:10:16.380375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.imshow(p)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.382669Z","iopub.execute_input":"2022-12-03T16:10:16.383346Z","iopub.status.idle":"2022-12-03T16:10:16.395808Z","shell.execute_reply.started":"2022-12-03T16:10:16.383311Z","shell.execute_reply":"2022-12-03T16:10:16.394311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf_test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndf_sub = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.397624Z","iopub.execute_input":"2022-12-03T16:10:16.39842Z","iopub.status.idle":"2022-12-03T16:10:16.548314Z","shell.execute_reply.started":"2022-12-03T16:10:16.398385Z","shell.execute_reply":"2022-12-03T16:10:16.546925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.553833Z","iopub.execute_input":"2022-12-03T16:10:16.554211Z","iopub.status.idle":"2022-12-03T16:10:16.579769Z","shell.execute_reply.started":"2022-12-03T16:10:16.554176Z","shell.execute_reply":"2022-12-03T16:10:16.578686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We see from the dataframe we have some nan value, but it can be checked directly\ndf_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.581106Z","iopub.execute_input":"2022-12-03T16:10:16.581538Z","iopub.status.idle":"2022-12-03T16:10:16.599895Z","shell.execute_reply.started":"2022-12-03T16:10:16.581505Z","shell.execute_reply":"2022-12-03T16:10:16.598351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To take care of the nan \ndf_train = df_train.apply(lambda x:x.fillna(x.value_counts().index[0]))","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.601675Z","iopub.execute_input":"2022-12-03T16:10:16.602021Z","iopub.status.idle":"2022-12-03T16:10:16.657704Z","shell.execute_reply.started":"2022-12-03T16:10:16.601991Z","shell.execute_reply":"2022-12-03T16:10:16.656672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check the data again\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.658933Z","iopub.execute_input":"2022-12-03T16:10:16.659811Z","iopub.status.idle":"2022-12-03T16:10:16.678689Z","shell.execute_reply.started":"2022-12-03T16:10:16.659768Z","shell.execute_reply":"2022-12-03T16:10:16.677387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['prediction_id'] = df_train['patient_id'].astype(str) +'_'+ df_train['laterality'].astype(str)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.680381Z","iopub.execute_input":"2022-12-03T16:10:16.680915Z","iopub.status.idle":"2022-12-03T16:10:16.737265Z","shell.execute_reply.started":"2022-12-03T16:10:16.680868Z","shell.execute_reply":"2022-12-03T16:10:16.736094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.73843Z","iopub.execute_input":"2022-12-03T16:10:16.738757Z","iopub.status.idle":"2022-12-03T16:10:16.756822Z","shell.execute_reply.started":"2022-12-03T16:10:16.73873Z","shell.execute_reply":"2022-12-03T16:10:16.755614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting the object type into numeric type for better analysis\nfrom sklearn.preprocessing import LabelEncoder\n\ndef labencoder(file):\n    le = LabelEncoder()\n    label = le.fit_transform(file)\n    return label","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.758337Z","iopub.execute_input":"2022-12-03T16:10:16.758732Z","iopub.status.idle":"2022-12-03T16:10:16.836947Z","shell.execute_reply.started":"2022-12-03T16:10:16.758701Z","shell.execute_reply":"2022-12-03T16:10:16.835876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['machine_id'] = labencoder(df_train['machine_id'])\ndf_train['view'] = labencoder(df_train['view'])\ndf_train['laterality'] = labencoder(df_train['laterality'])\ndf_train['difficult_negative_case'] = labencoder(df_train['difficult_negative_case'])\ndf_train['density'] = labencoder(df_train['density'])\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.838384Z","iopub.execute_input":"2022-12-03T16:10:16.83946Z","iopub.status.idle":"2022-12-03T16:10:16.890077Z","shell.execute_reply.started":"2022-12-03T16:10:16.839421Z","shell.execute_reply":"2022-12-03T16:10:16.889138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.89133Z","iopub.execute_input":"2022-12-03T16:10:16.892291Z","iopub.status.idle":"2022-12-03T16:10:16.919904Z","shell.execute_reply.started":"2022-12-03T16:10:16.892256Z","shell.execute_reply":"2022-12-03T16:10:16.918756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.921335Z","iopub.execute_input":"2022-12-03T16:10:16.921722Z","iopub.status.idle":"2022-12-03T16:10:16.945722Z","shell.execute_reply.started":"2022-12-03T16:10:16.921688Z","shell.execute_reply":"2022-12-03T16:10:16.944528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:16.947106Z","iopub.execute_input":"2022-12-03T16:10:16.947456Z","iopub.status.idle":"2022-12-03T16:10:17.023564Z","shell.execute_reply.started":"2022-12-03T16:10:16.947424Z","shell.execute_reply":"2022-12-03T16:10:17.022288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:17.02474Z","iopub.execute_input":"2022-12-03T16:10:17.025058Z","iopub.status.idle":"2022-12-03T16:10:17.039046Z","shell.execute_reply.started":"2022-12-03T16:10:17.02503Z","shell.execute_reply":"2022-12-03T16:10:17.037807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:17.040735Z","iopub.execute_input":"2022-12-03T16:10:17.04114Z","iopub.status.idle":"2022-12-03T16:10:17.055513Z","shell.execute_reply.started":"2022-12-03T16:10:17.041105Z","shell.execute_reply":"2022-12-03T16:10:17.054362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:17.056941Z","iopub.execute_input":"2022-12-03T16:10:17.057291Z","iopub.status.idle":"2022-12-03T16:10:17.072117Z","shell.execute_reply.started":"2022-12-03T16:10:17.05726Z","shell.execute_reply":"2022-12-03T16:10:17.070737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(df_train['cancer'],label='count')","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:17.073545Z","iopub.execute_input":"2022-12-03T16:10:17.073893Z","iopub.status.idle":"2022-12-03T16:10:17.205588Z","shell.execute_reply.started":"2022-12-03T16:10:17.073863Z","shell.execute_reply":"2022-12-03T16:10:17.204395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(['site_id','patient_id'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:17.207236Z","iopub.execute_input":"2022-12-03T16:10:17.208133Z","iopub.status.idle":"2022-12-03T16:10:17.227851Z","shell.execute_reply.started":"2022-12-03T16:10:17.208066Z","shell.execute_reply":"2022-12-03T16:10:17.226083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The target dataset seems to be higly imbalanced so to make classification better we can follow \n# feature engineering processs to balance the target dataset","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:17.229955Z","iopub.execute_input":"2022-12-03T16:10:17.231363Z","iopub.status.idle":"2022-12-03T16:10:17.243528Z","shell.execute_reply.started":"2022-12-03T16:10:17.231296Z","shell.execute_reply":"2022-12-03T16:10:17.241778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f,ax = plt.subplots(figsize=(18,18))\nmatrix = np.triu(df_train.corr())\nsns.heatmap(df_train.corr(),annot = True,linewidths=.5,fmt='.1f',ax=ax,mask=matrix)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:17.245715Z","iopub.execute_input":"2022-12-03T16:10:17.246683Z","iopub.status.idle":"2022-12-03T16:10:18.128763Z","shell.execute_reply.started":"2022-12-03T16:10:17.246647Z","shell.execute_reply":"2022-12-03T16:10:18.127448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.drop(['site_id','patient_id','laterality'],axis=1)\ndf_test['view'] = labencoder(df_test['view'])\ndf_test['prediction_id'] = labencoder(df_test['prediction_id'])\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.136794Z","iopub.execute_input":"2022-12-03T16:10:18.137524Z","iopub.status.idle":"2022-12-03T16:10:18.153657Z","shell.execute_reply.started":"2022-12-03T16:10:18.137479Z","shell.execute_reply":"2022-12-03T16:10:18.152326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df_train['cancer']","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.15522Z","iopub.execute_input":"2022-12-03T16:10:18.15557Z","iopub.status.idle":"2022-12-03T16:10:18.172612Z","shell.execute_reply.started":"2022-12-03T16:10:18.15554Z","shell.execute_reply":"2022-12-03T16:10:18.171767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(['cancer'],axis=1)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.174034Z","iopub.execute_input":"2022-12-03T16:10:18.17465Z","iopub.status.idle":"2022-12-03T16:10:18.205799Z","shell.execute_reply.started":"2022-12-03T16:10:18.174616Z","shell.execute_reply":"2022-12-03T16:10:18.204523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['prediction_id'] = labencoder(df_train['prediction_id'])","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.207496Z","iopub.execute_input":"2022-12-03T16:10:18.208091Z","iopub.status.idle":"2022-12-03T16:10:18.290582Z","shell.execute_reply.started":"2022-12-03T16:10:18.208058Z","shell.execute_reply":"2022-12-03T16:10:18.289402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.292045Z","iopub.execute_input":"2022-12-03T16:10:18.292391Z","iopub.status.idle":"2022-12-03T16:10:18.315969Z","shell.execute_reply.started":"2022-12-03T16:10:18.29236Z","shell.execute_reply":"2022-12-03T16:10:18.314645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X1 = df_train.drop(['view','laterality','image_id','implant','difficult_negative_case','machine_id','density'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.318241Z","iopub.execute_input":"2022-12-03T16:10:18.31898Z","iopub.status.idle":"2022-12-03T16:10:18.334079Z","shell.execute_reply.started":"2022-12-03T16:10:18.318927Z","shell.execute_reply":"2022-12-03T16:10:18.332942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.335675Z","iopub.execute_input":"2022-12-03T16:10:18.336737Z","iopub.status.idle":"2022-12-03T16:10:18.354743Z","shell.execute_reply.started":"2022-12-03T16:10:18.336697Z","shell.execute_reply":"2022-12-03T16:10:18.353585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.35623Z","iopub.execute_input":"2022-12-03T16:10:18.356602Z","iopub.status.idle":"2022-12-03T16:10:18.374176Z","shell.execute_reply.started":"2022-12-03T16:10:18.35657Z","shell.execute_reply":"2022-12-03T16:10:18.37336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X1 = df_test.drop(['view','image_id','implant','machine_id'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.375503Z","iopub.execute_input":"2022-12-03T16:10:18.37583Z","iopub.status.idle":"2022-12-03T16:10:18.385948Z","shell.execute_reply.started":"2022-12-03T16:10:18.375801Z","shell.execute_reply":"2022-12-03T16:10:18.384831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.387637Z","iopub.execute_input":"2022-12-03T16:10:18.388098Z","iopub.status.idle":"2022-12-03T16:10:18.40248Z","shell.execute_reply.started":"2022-12-03T16:10:18.388051Z","shell.execute_reply":"2022-12-03T16:10:18.401216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train,x_test,y_train,y_test = train_test_split(df_train,y,random_state=100,test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.403967Z","iopub.execute_input":"2022-12-03T16:10:18.404318Z","iopub.status.idle":"2022-12-03T16:10:18.478461Z","shell.execute_reply.started":"2022-12-03T16:10:18.404288Z","shell.execute_reply":"2022-12-03T16:10:18.477188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.479799Z","iopub.execute_input":"2022-12-03T16:10:18.48012Z","iopub.status.idle":"2022-12-03T16:10:18.487872Z","shell.execute_reply.started":"2022-12-03T16:10:18.480092Z","shell.execute_reply":"2022-12-03T16:10:18.486555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import confusion_matrix,classification_report\nmodel = LogisticRegression(solver='liblinear',class_weight='balanced',max_iter=10000)\nmodel.fit(x_train,y_train)\npredict = model.predict(x_test)\nprint(confusion_matrix(y_test,predict))\nprint(classification_report(y_test,predict))","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.489255Z","iopub.execute_input":"2022-12-03T16:10:18.489768Z","iopub.status.idle":"2022-12-03T16:10:18.846588Z","shell.execute_reply.started":"2022-12-03T16:10:18.489659Z","shell.execute_reply":"2022-12-03T16:10:18.844874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.854527Z","iopub.execute_input":"2022-12-03T16:10:18.8588Z","iopub.status.idle":"2022-12-03T16:10:18.879811Z","shell.execute_reply.started":"2022-12-03T16:10:18.858709Z","shell.execute_reply":"2022-12-03T16:10:18.877669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Before OverSampling, counts of label '1': {}\".format(sum(y_train== 1)))\nprint(\"Before OverSampling, counts of label '0': {} \\n\".format(sum(y_train== 0)))\n\n# import SMOTE module from imblearn library\n# pip install imblearn (if you don't have imblearn in your system)\nfrom imblearn.over_sampling import SMOTE\nsm = SMOTE(random_state = 2)\nX_train_res, y_train_res = sm.fit_resample(x_train, y_train.ravel())\n\nprint('After OverSampling, the shape of train_X: {}'.format(X_train_res.shape))\nprint('After OverSampling, the shape of train_y: {} \\n'.format(y_train_res.shape))\n\nprint(\"After OverSampling, counts of label '1': {}\".format(sum(y_train_res == 1)))\nprint(\"After OverSampling, counts of label '0': {}\".format(sum(y_train_res == 0)))\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:18.887966Z","iopub.execute_input":"2022-12-03T16:10:18.889142Z","iopub.status.idle":"2022-12-03T16:10:19.557655Z","shell.execute_reply.started":"2022-12-03T16:10:18.889075Z","shell.execute_reply":"2022-12-03T16:10:19.556385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Work in progress","metadata":{}},{"cell_type":"code","source":"lr1 = LogisticRegression()\nlr1.fit(X_train_res, y_train_res.ravel())\npredictions = lr1.predict(x_test)\n  \n# print classification report\nprint(classification_report(y_test, predictions))","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:19.558832Z","iopub.execute_input":"2022-12-03T16:10:19.559138Z","iopub.status.idle":"2022-12-03T16:10:20.352358Z","shell.execute_reply.started":"2022-12-03T16:10:19.559111Z","shell.execute_reply":"2022-12-03T16:10:20.350716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function which trains models\ndef models(X_train,y_train):\n    \n  \n  #Using Logistic Regression \n    from sklearn.linear_model import LogisticRegression\n    log = LogisticRegression(random_state = 0)\n    log.fit(X_train, y_train)\n\n  #Using SVC linear\n    from sklearn.svm import SVC\n    svc_lin = SVC(kernel = 'linear', random_state = 0)\n    svc_lin.fit(X_train, y_train)\n\n  #Using SVC rbf\n    from sklearn.svm import SVC\n    svc_rbf = SVC(kernel = 'rbf', random_state = 0)\n    svc_rbf.fit(X_train, y_train)\n\n  #Using DecisionTreeClassifier \n    from sklearn.tree import DecisionTreeClassifier\n    tree = DecisionTreeClassifier(criterion = 'entropy', random_state = 0)\n    tree.fit(X_train, y_train)\n\n  #Using RandomForestClassifier method of ensemble class to use Random Forest Classification algorithm\n    from sklearn.ensemble import RandomForestClassifier\n    forest = RandomForestClassifier(n_estimators = 10, criterion = 'entropy', random_state = 0)\n    forest.fit(X_train, y_train)\n  \n  #print model accuracy on the training data.\n    print('[0]Logistic Regression Training Accuracy:', log.score(X_train, y_train))\n    #print('[1]K Nearest Neighbor Training Accuracy:', knn.score(X_train, y_train))\n    print('[1]Support Vector Machine (Linear Classifier) Training Accuracy:', svc_lin.score(X_train, y_train))\n    print('[2]Support Vector Machine (RBF Classifier) Training Accuracy:', svc_rbf.score(X_train, y_train))\n    #print('[4]Gaussian Naive Bayes Training Accuracy:', gauss.score(X_train, y_train))\n    print('[3]Decision Tree Classifier Training Accuracy:', tree.score(X_train, y_train))\n    print('[4]Random Forest Classifier Training Accuracy:', forest.score(X_train, y_train))\n  \n    return log, svc_lin, svc_rbf, tree, forest","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:20.354565Z","iopub.execute_input":"2022-12-03T16:10:20.355463Z","iopub.status.idle":"2022-12-03T16:10:20.376896Z","shell.execute_reply.started":"2022-12-03T16:10:20.3554Z","shell.execute_reply":"2022-12-03T16:10:20.374514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = models(x_train,y_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:10:20.379871Z","iopub.execute_input":"2022-12-03T16:10:20.38111Z","iopub.status.idle":"2022-12-03T16:14:29.503732Z","shell.execute_reply.started":"2022-12-03T16:10:20.381042Z","shell.execute_reply":"2022-12-03T16:14:29.50226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nfor i in range(len(model)):\n    \n    cm = confusion_matrix(y_test, model[i].predict(x_test))\n  \n    TN = cm[0][0]\n    TP = cm[1][1]\n    FN = cm[1][0]\n    FP = cm[0][1]\n  \n    print(cm)\n    print('Model[{}] Testing Accuracy = \"{}\"'.format(i,  (TP + TN) / (TP + TN + FN + FP)))\n    print()# Print a new line","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:14:29.505146Z","iopub.execute_input":"2022-12-03T16:14:29.505615Z","iopub.status.idle":"2022-12-03T16:14:31.573224Z","shell.execute_reply.started":"2022-12-03T16:14:29.505573Z","shell.execute_reply":"2022-12-03T16:14:31.572115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Show other ways to get the classification accuracy & other metrics \n\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import accuracy_score\nypred1=[]\nfor i in range(len(model)):\n    ypred = model[i].predict(x_test)\n    ypred1.append(ypred)\n    print('Model ',i)\n  #Check precision, recall, f1-score\n    print(classification_report(y_test, model[i].predict(x_test)))\n  #Another way to get the models accuracy on the test data\n    print(accuracy_score(y_test, model[i].predict(x_test)))\n    print()#Print a new line\n    print(ypred)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:14:31.574555Z","iopub.execute_input":"2022-12-03T16:14:31.574895Z","iopub.status.idle":"2022-12-03T16:14:37.70904Z","shell.execute_reply.started":"2022-12-03T16:14:31.574863Z","shell.execute_reply":"2022-12-03T16:14:37.707841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ypred","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:14:37.710292Z","iopub.execute_input":"2022-12-03T16:14:37.710756Z","iopub.status.idle":"2022-12-03T16:14:37.717719Z","shell.execute_reply.started":"2022-12-03T16:14:37.710722Z","shell.execute_reply":"2022-12-03T16:14:37.71669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\npreds = np.mean(ypred)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:14:37.719032Z","iopub.execute_input":"2022-12-03T16:14:37.719411Z","iopub.status.idle":"2022-12-03T16:14:37.729015Z","shell.execute_reply.started":"2022-12-03T16:14:37.71938Z","shell.execute_reply":"2022-12-03T16:14:37.727983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_testi  = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndf_samp = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:14:37.730632Z","iopub.execute_input":"2022-12-03T16:14:37.731043Z","iopub.status.idle":"2022-12-03T16:14:37.748556Z","shell.execute_reply.started":"2022-12-03T16:14:37.731013Z","shell.execute_reply":"2022-12-03T16:14:37.747652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_testi[1:2] = df_testi[2:3]","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:14:37.749647Z","iopub.execute_input":"2022-12-03T16:14:37.750715Z","iopub.status.idle":"2022-12-03T16:14:37.759723Z","shell.execute_reply.started":"2022-12-03T16:14:37.750678Z","shell.execute_reply":"2022-12-03T16:14:37.758572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.DataFrame(columns=['prediction_id', 'cancer'])\nsub_df['prediction_id'] = df_testi['prediction_id']\nsub_df = pd.DataFrame(data={'prediction_id': df_testi['prediction_id'],'cancer':preds}).drop_duplicates(subset='prediction_id')","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:14:37.761422Z","iopub.execute_input":"2022-12-03T16:14:37.762587Z","iopub.status.idle":"2022-12-03T16:14:37.780595Z","shell.execute_reply.started":"2022-12-03T16:14:37.762538Z","shell.execute_reply":"2022-12-03T16:14:37.779545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:14:37.781729Z","iopub.execute_input":"2022-12-03T16:14:37.782322Z","iopub.status.idle":"2022-12-03T16:14:37.796942Z","shell.execute_reply.started":"2022-12-03T16:14:37.782287Z","shell.execute_reply":"2022-12-03T16:14:37.795939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2022-12-03T16:14:37.798536Z","iopub.execute_input":"2022-12-03T16:14:37.799526Z","iopub.status.idle":"2022-12-03T16:14:37.813263Z","shell.execute_reply.started":"2022-12-03T16:14:37.799461Z","shell.execute_reply":"2022-12-03T16:14:37.811968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}