{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport pydicom\n\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-29T15:12:57.079771Z","iopub.execute_input":"2022-11-29T15:12:57.080936Z","iopub.status.idle":"2022-11-29T15:12:58.23982Z","shell.execute_reply.started":"2022-11-29T15:12:57.080803Z","shell.execute_reply":"2022-11-29T15:12:58.238637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(1)","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:12:58.242191Z","iopub.execute_input":"2022-11-29T15:12:58.242616Z","iopub.status.idle":"2022-11-29T15:12:58.248683Z","shell.execute_reply.started":"2022-11-29T15:12:58.242573Z","shell.execute_reply":"2022-11-29T15:12:58.247314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data inladen","metadata":{}},{"cell_type":"code","source":"train_filename = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\ntest_filename = '/kaggle/input/rsna-breast-cancer-detection/test.csv'","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:12:58.250253Z","iopub.execute_input":"2022-11-29T15:12:58.250668Z","iopub.status.idle":"2022-11-29T15:12:58.262341Z","shell.execute_reply.started":"2022-11-29T15:12:58.250626Z","shell.execute_reply":"2022-11-29T15:12:58.261151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(train_filename)\ndf_test = pd.read_csv(test_filename)\ndisplay(df_train.head())\ndisplay(df_test.head())","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:12:58.265027Z","iopub.execute_input":"2022-11-29T15:12:58.265382Z","iopub.status.idle":"2022-11-29T15:12:58.470795Z","shell.execute_reply.started":"2022-11-29T15:12:58.265341Z","shell.execute_reply":"2022-11-29T15:12:58.469702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data visualizatie","metadata":{}},{"cell_type":"code","source":"def visualize(patient_id, figsize=(10, 10)):\n    fig, ax = plt.subplots(2, 2, figsize=figsize)\n    ax = ax.flatten()\n    path_to_dcms = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n    for i, dcm_path in enumerate(os.listdir(os.path.join(path_to_dcms, patient_id))):\n        dcm = pydicom.dcmread(os.path.join(path_to_dcms, patient_id, dcm_path))\n        dcm = dcm.pixel_array\n        ax[i].imshow(dcm, cmap=\"bone\")","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:12:58.472341Z","iopub.execute_input":"2022-11-29T15:12:58.472998Z","iopub.status.idle":"2022-11-29T15:12:58.481053Z","shell.execute_reply.started":"2022-11-29T15:12:58.472953Z","shell.execute_reply":"2022-11-29T15:12:58.479886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# patient_id = str(df_train['patient_id'].sample(1).values[0])\n# visualize(patient_id)","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:12:58.482764Z","iopub.execute_input":"2022-11-29T15:12:58.483972Z","iopub.status.idle":"2022-11-29T15:12:59.548697Z","shell.execute_reply.started":"2022-11-29T15:12:58.483927Z","shell.execute_reply":"2022-11-29T15:12:59.547396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Er zijn een aantal hoeken waarop een foto gemaakt kan worden. De \"view\" kolom bevat afkorting voor orientatie van fotografie. Standaard worden er 2 orientaties gebruikt per borst. Ik ben benieuwd naar de verdeling van deze orientaties. Is er een orientatie die bijvoorbeeld altijd wel voorkomt bij een patient? Dit zou ik willen weten want dan kan ik in eerste instantie een model maken die slechts 1 afbeelding bekijkt. Zo maak ik het probleem makkelijker en kan ik het later nog uitbouwen","metadata":{}},{"cell_type":"code","source":"df_train['view'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:12:59.549749Z","iopub.status.idle":"2022-11-29T15:12:59.550648Z","shell.execute_reply.started":"2022-11-29T15:12:59.55044Z","shell.execute_reply":"2022-11-29T15:12:59.550462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=df_train, x='view', stat='percent')","metadata":{"execution":{"iopub.status.busy":"2022-11-29T14:52:12.540182Z","iopub.execute_input":"2022-11-29T14:52:12.540619Z","iopub.status.idle":"2022-11-29T14:52:12.816383Z","shell.execute_reply.started":"2022-11-29T14:52:12.540583Z","shell.execute_reply":"2022-11-29T14:52:12.814833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In deze [thread](https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/369264) wordt uitleg gegeven van de afkortingen. Meest voorkomende orientaties in de dataset zijn:\n* CC - Cranial Caudad, kop staart, dus aanzicht van bovenaf \n* MLO - Medio Lateral Oblique. \"A standard mammographic view taken from an oblique or angled view, which is the most important projection as it allows imaging of the greatest amount of breast tissue\" [bron](https://medical-dictionary.thefreedictionary.com/mediolateral+oblique+view) Vanuit het midden (borst) naar buiten (linker of rechterkant op)\n\nHet is handig om hier een [medisch woordenboek](https://medical-dictionary.thefreedictionary.com/) te raadplegen als je termen tegenkomt die je niet begrijpt. ","metadata":{}},{"cell_type":"code","source":"patient_with_mlo = df_train[df_train.groupby('patient_id')['view'].apply(lambda x: x == 'MLO')]\npatient_with_cc = df_train[df_train.groupby('patient_id')['view'].apply(lambda x: x == 'MLO')]\n\nprint(f'# patients with atleast one MLO view: {}' len(patient_with_mlo['patient_id'].unique()))\nprint(f'# patients with atleast one CC view: {}' len(patient_with_cc['patient_id'].unique()))\nprint(f'# patients: {}' len(df_train['patient_id'].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:16:51.72522Z","iopub.execute_input":"2022-11-29T15:16:51.725645Z","iopub.status.idle":"2022-11-29T15:16:54.059921Z","shell.execute_reply.started":"2022-11-29T15:16:51.725605Z","shell.execute_reply":"2022-11-29T15:16:54.058606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Alle patienten in de training set hebben in ieder geval 1 foto in orientatie MLO en 1 CC. Dus we kunnen een model maken die alleen naar de MLO foto kijkt en dan voorspelt of de patient kanker heeft.","metadata":{}},{"cell_type":"markdown","source":"Als MLO en CC de standaard orientatie is, waarom wordt er dan soms wel een foto in een andere orientatie gemaakt?\n\nHypothese: Als de doktor iets abnormaals ziet bij een patient dan is de kans groter dat hij een foto in een andere orienatie maakt. En aangezien kanker abnormaal is, lijkt het mij dat kanker een correlatie heeft met de orientatie van de foto's.\n\nLaten we gaan kijken of mijn vermoeden juist is.","metadata":{}},{"cell_type":"code","source":"df_train.groupby('view')['cancer'].mean().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:44:43.745507Z","iopub.execute_input":"2022-11-29T15:44:43.746553Z","iopub.status.idle":"2022-11-29T15:44:43.760292Z","shell.execute_reply.started":"2022-11-29T15:44:43.746512Z","shell.execute_reply":"2022-11-29T15:44:43.75932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percantage_patients_with_cancer = sum(df_train.groupby('patient_id').sum('cancer')['cancer'])/len(df_train.groupby('patient_id'))","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:46:22.223354Z","iopub.execute_input":"2022-11-29T15:46:22.223774Z","iopub.status.idle":"2022-11-29T15:46:22.376717Z","shell.execute_reply.started":"2022-11-29T15:46:22.223716Z","shell.execute_reply":"2022-11-29T15:46:22.375644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"has_at_view = df_train.groupby('patient_id')['view'].apply(lambda x: x == 'AT')\ndf_train['has_at_view'] = has_at_view\ndf_train.groupby('cancer')['has_at_view'].sum()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T15:57:11.350264Z","iopub.execute_input":"2022-11-29T15:57:11.350698Z","iopub.status.idle":"2022-11-29T15:57:13.175188Z","shell.execute_reply.started":"2022-11-29T15:57:11.350658Z","shell.execute_reply":"2022-11-29T15:57:13.173829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Util Functies\n\nMoeten functie hebben die de afbeeldingen pakt die bij een patient id hoort en deze omzet van dcm formaat naar arrays","metadata":{}},{"cell_type":"code","source":"def get_mammograms(patient_id):\n    path_to_dcms = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n    images = []\n    for i, dcm_path in enumerate(os.listdir(os.path.join(path_to_dcms, patient_id))):\n            dcm = pydicom.dcmread(os.path.join(path_to_dcms, patient_id, dcm_path))\n            dcm = dcm.pixel_array\n            images.append(dcm)\n    return np.array(images)","metadata":{"execution":{"iopub.status.busy":"2022-11-29T14:12:48.048776Z","iopub.execute_input":"2022-11-29T14:12:48.049244Z","iopub.status.idle":"2022-11-29T14:12:48.0568Z","shell.execute_reply.started":"2022-11-29T14:12:48.049207Z","shell.execute_reply":"2022-11-29T14:12:48.055119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mammograms = get_mammograms(patient_id)\nmammograms.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-29T14:12:48.227986Z","iopub.execute_input":"2022-11-29T14:12:48.228825Z","iopub.status.idle":"2022-11-29T14:12:50.39284Z","shell.execute_reply.started":"2022-11-29T14:12:48.228777Z","shell.execute_reply":"2022-11-29T14:12:50.391324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"De afbeeldingen zijn erg groot (2776 hoog en 2082 breed). Het is een goed idee om voor nu met kleinere afbeeldingen te werken. We zijn de dataset nog aan het ontdekken. Door de afbeeldingen te verkleinen kunnen we sneller data laden en modellen trainen. Zo kunnen we snel een prototype maken. We kunnen altijd later nog kijken of grotere afbeeldingen het model beter maakt.","metadata":{}},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"def preprocces","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission maken","metadata":{}},{"cell_type":"code","source":"has_at_view = df_test.groupby('patient_id')['view'].apply(lambda x: x == 'AT')\ndf_test['has_at_view'] = has_at_view.astype(float)\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-11-29T16:02:03.377479Z","iopub.execute_input":"2022-11-29T16:02:03.377891Z","iopub.status.idle":"2022-11-29T16:02:03.395216Z","shell.execute_reply.started":"2022-11-29T16:02:03.377854Z","shell.execute_reply":"2022-11-29T16:02:03.394418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(data={'prediction_id': df_test['prediction_id'], \n                                'cancer': df_test['has_at_view']}).drop_duplicates(subset='prediction_id')\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T16:02:08.277041Z","iopub.execute_input":"2022-11-29T16:02:08.277464Z","iopub.status.idle":"2022-11-29T16:02:08.291452Z","shell.execute_reply.started":"2022-11-29T16:02:08.277432Z","shell.execute_reply":"2022-11-29T16:02:08.290337Z"},"trusted":true},"execution_count":null,"outputs":[]}]}