{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4619805,"sourceType":"datasetVersion","datasetId":2688675}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#importing the libraries\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sn","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:15.755662Z","iopub.execute_input":"2023-12-13T23:01:15.756029Z","iopub.status.idle":"2023-12-13T23:01:15.76166Z","shell.execute_reply.started":"2023-12-13T23:01:15.755998Z","shell.execute_reply":"2023-12-13T23:01:15.760399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing our cancer dataset\ndf = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf.columns\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:15.765954Z","iopub.execute_input":"2023-12-13T23:01:15.766243Z","iopub.status.idle":"2023-12-13T23:01:15.823192Z","shell.execute_reply.started":"2023-12-13T23:01:15.766219Z","shell.execute_reply":"2023-12-13T23:01:15.822214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing our cancer dataset\ndf2 = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndf2.columns","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:15.824912Z","iopub.execute_input":"2023-12-13T23:01:15.825332Z","iopub.status.idle":"2023-12-13T23:01:15.833154Z","shell.execute_reply.started":"2023-12-13T23:01:15.825309Z","shell.execute_reply":"2023-12-13T23:01:15.832278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Informations from the Dataset Description:**\n- site_id - ID code for the source hospital.\n- patient_id - ID code for the patient.\n- image_id - ID code for the image.\n- laterality - Whether the image is of the left or right breast.\n- view - The orientation of the image. The default for a screening exam is to capture two views per breast.\n- age - The patient's age in years.\n- implant - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\n- density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. (Only provided for train).\n- machine_id - An ID code for the imaging device.\n- cancer - Whether or not the breast was positive for malignant cancer. The target value. (Only provided for train).\n- biopsy - Whether or not a follow-up biopsy was performed on the breast. (Only provided for train).\n- invasive - If the breast is positive for cancer, whether or not the cancer proved to be invasive. (Only provided for train).\n- BIRADS - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n- prediction_id - The ID for the matching submission row. Multiple images will share the same prediction ID. (Test only).\n- difficult_negative_case - True if the case was unusually difficult. (Only provided for train)","metadata":{}},{"cell_type":"code","source":"#find the dimensions of the data set using the panda dataset ‘shape’ attribute.\nprint(\"Cancer data set dimensions : {}\".format(df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:15.834071Z","iopub.execute_input":"2023-12-13T23:01:15.834303Z","iopub.status.idle":"2023-12-13T23:01:15.843877Z","shell.execute_reply.started":"2023-12-13T23:01:15.834283Z","shell.execute_reply":"2023-12-13T23:01:15.842725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:15.845522Z","iopub.execute_input":"2023-12-13T23:01:15.845972Z","iopub.status.idle":"2023-12-13T23:01:15.863361Z","shell.execute_reply.started":"2023-12-13T23:01:15.845935Z","shell.execute_reply":"2023-12-13T23:01:15.862568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sample images:**","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\n\ndef load_dicom_images(directory):\n    dicom_images = []\n    for root, dirs, files in os.walk(directory):\n        for file in files:\n            dicom_images.append(pydicom.dcmread(os.path.join(root, file)))\n    return dicom_images\n\n# Step 2: Visualize DICOM Images in Black and White\ndef visualize_dicom_images(images, num_samples):\n    sample_images = images[:num_samples]\n    for i, image in enumerate(sample_images):\n        plt.subplot(1, num_samples, i + 1)\n        plt.imshow(image.pixel_array, cmap='gray')  # Utilisez 'gray' pour afficher en noir et blanc\n        plt.title(f\"Image {i + 1}\")\n        plt.axis('off')\n    plt.show()\n\n# Changez le chemin du dossier en fonction de votre situation\ntrain_dcm_folder = \"/kaggle/input/rsna-breast-cancer-detection/train_images/10006\"\ndcm_img = load_dicom_images(train_dcm_folder)\nvisualize_dicom_images(dcm_img, 4)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:15.866053Z","iopub.execute_input":"2023-12-13T23:01:15.866368Z","iopub.status.idle":"2023-12-13T23:01:25.71725Z","shell.execute_reply.started":"2023-12-13T23:01:15.866322Z","shell.execute_reply":"2023-12-13T23:01:25.716118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df[['site_id', 'patient_id', 'image_id', 'laterality', 'view', 'age',\n       'cancer', 'biopsy', 'invasive', 'BIRADS', 'implant', 'density',\n       'machine_id', 'difficult_negative_case']]\ndf1[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:25.718415Z","iopub.execute_input":"2023-12-13T23:01:25.718648Z","iopub.status.idle":"2023-12-13T23:01:25.73603Z","shell.execute_reply.started":"2023-12-13T23:01:25.718627Z","shell.execute_reply":"2023-12-13T23:01:25.734898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note :The value NaN (Not a Number) is not equivalent to the null value (0). NaN is used to represent the absence of data or an undefined value in a numeric array","metadata":{}},{"cell_type":"code","source":"#Missing or Null Data points\ndf1.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:25.737103Z","iopub.execute_input":"2023-12-13T23:01:25.73732Z","iopub.status.idle":"2023-12-13T23:01:25.755698Z","shell.execute_reply.started":"2023-12-13T23:01:25.7373Z","shell.execute_reply":"2023-12-13T23:01:25.754573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlation Matrix:**\n\nCorrelation is a statistical measure of the relationship between two variables. The correlation coefficient, ranging from -1 to 1, indicates the strength and direction of this connection.","metadata":{}},{"cell_type":"code","source":"# Select only numeric columns from the DataFrame:\nnumeric_columns = df1.select_dtypes(include=['int64', 'float64'])\n\n# Create the correlation matrix\ncorrelation_matrix = numeric_columns.corr()\n\n# Plot the correlation matrix using seaborn\nplt.figure(figsize=(20, 15))\nsn.heatmap(correlation_matrix, annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:25.757094Z","iopub.execute_input":"2023-12-13T23:01:25.757465Z","iopub.status.idle":"2023-12-13T23:01:26.303949Z","shell.execute_reply.started":"2023-12-13T23:01:25.757433Z","shell.execute_reply":"2023-12-13T23:01:26.302545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The correlation of cancer with another variable:**","metadata":{}},{"cell_type":"code","source":"# Select only numeric columns from the DataFrame\nnumeric_columns = df1.select_dtypes(include=['int64', 'float64'])\n\n# Create the correlation matrix\ncorr_matrix = numeric_columns.corr()\n\n# Sort the correlations between the \"cancer\" column and other numeric columns\ncancer_correlations = corr_matrix['cancer'].sort_values(ascending=False)\n\nprint(cancer_correlations)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:26.305775Z","iopub.execute_input":"2023-12-13T23:01:26.306121Z","iopub.status.idle":"2023-12-13T23:01:26.326936Z","shell.execute_reply.started":"2023-12-13T23:01:26.306091Z","shell.execute_reply":"2023-12-13T23:01:26.325138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note : The matrix demonstrates a positive relationship between cancerand : invasive,biopsy and age","metadata":{}},{"cell_type":"markdown","source":"**Number of patients / the youngest and the oldest patient / negative and positive cases :**","metadata":{}},{"cell_type":"code","source":"num_patients = df1['patient_id'].nunique()\nmin_patient_age = df1['age'].min()\nmax_patient_age = df1['age'].max()\n\ngrouped = df1.groupby('patient_id')['cancer'].max()\nn_negative = grouped.value_counts().get(0, 0)\nn_positive = grouped.value_counts().get(1, 0)\n\nprint(f\"There are {num_patients} different patients in the train set.\\n\")\nprint(f\"The youngest patient is {int(min_patient_age)} years old.\")\nprint(f\"The oldest patient is {int(max_patient_age)} years old.\\n\")\nprint(f\"{n_negative} patients ({n_negative / num_patients:.1%}) are negative to breast cancer.\")\nprint(f\"{n_positive} patients ({n_positive / num_patients:.1%}) are positive to breast cancer.\")","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:26.328373Z","iopub.execute_input":"2023-12-13T23:01:26.328662Z","iopub.status.idle":"2023-12-13T23:01:26.342696Z","shell.execute_reply.started":"2023-12-13T23:01:26.328637Z","shell.execute_reply":"2023-12-13T23:01:26.341259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Figure represents the number of patients with/without cancer:**","metadata":{}},{"cell_type":"code","source":"cancer_per_patient = df1.groupby(\"patient_id\")[\"cancer\"].max().values\nn_negative = (cancer_per_patient == 0).sum()\nn_positive = (cancer_per_patient == 1).sum()\n\nfig, ax = plt.subplots()\nbars = ax.bar([\"No cancer\", \"Cancer\"], [n_negative, n_positive], color='mediumvioletred')\nax.set(xlabel=\"\", ylabel=\"Count\", title=\"Number of patients with/without cancer\")\n\n# Adding labels on top of the bars\nfor bar, count in zip(bars, [n_negative, n_positive]):\n    height = bar.get_height()\n    ax.text(bar.get_x() + bar.get_width() / 2, height, count, ha=\"center\", va=\"bottom\")\n\n# Display the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:26.344655Z","iopub.execute_input":"2023-12-13T23:01:26.345387Z","iopub.status.idle":"2023-12-13T23:01:26.511938Z","shell.execute_reply.started":"2023-12-13T23:01:26.34533Z","shell.execute_reply":"2023-12-13T23:01:26.510828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Age distribution:**\n","metadata":{}},{"cell_type":"code","source":"ages = df1.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\ncancer_ages = df1[df1['cancer'] == 1].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\nno_cancer_ages = df1[df1['cancer'] == 0].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nplt.figure(figsize=(14, 7))\n\nplt.subplot(1, 2, 2)\nsn.histplot(cancer_ages, bins=51, color='mediumvioletred', kde=True)  \nsn.histplot(no_cancer_ages, bins=63, color='indigo', kde=True)  \nplt.title(\"Patients with/without cancer\", fontsize=14)\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"Count\", fontsize=12)\nplt.xticks(fontsize=10)\nplt.yticks(fontsize=10)\nplt.xlim(33, 89)\nplt.legend([\"Cancer\", \"No cancer\"], fontsize=10)\n\nplt.suptitle(\"Age distribution of the patients\", fontsize=16)\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:26.513638Z","iopub.execute_input":"2023-12-13T23:01:26.513907Z","iopub.status.idle":"2023-12-13T23:01:28.738546Z","shell.execute_reply.started":"2023-12-13T23:01:26.513884Z","shell.execute_reply":"2023-12-13T23:01:28.737507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The average age:**","metadata":{}},{"cell_type":"code","source":"# Statistics\nimport pandas as pd\n\nages = df1.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nstats = pd.DataFrame({\n    #moyenne\n    \"Mean\": [ages.mean()],\n    #écart type\n    \"Std\": [ages.std()],\n   \n})\n\nformatted_stats = stats.style.format(\"{:.2f}\")\nformatted_stats.set_caption(\"Statistics for Age\")\nformatted_stats.set_properties(**{'text-align': 'center'})\n\ndisplay(formatted_stats)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:28.740593Z","iopub.execute_input":"2023-12-13T23:01:28.741814Z","iopub.status.idle":"2023-12-13T23:01:29.43189Z","shell.execute_reply.started":"2023-12-13T23:01:28.741767Z","shell.execute_reply":"2023-12-13T23:01:29.430877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**For cancer age :**","metadata":{}},{"cell_type":"code","source":"# Statistics for Cancer Age\nimport pandas as pd\n\nages = df1.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nstats = pd.DataFrame({\n    \"Mean\": [cancer_ages.mean()],\n    \"Std\": [cancer_ages.std()],\n    \"Minimum Cancer Age\": [cancer_ages.min()]\n})\n\nformatted_stats = stats.style.format(\"{:.2f}\")\nformatted_stats.set_caption(\"Statistics for Cancer Age\")\nformatted_stats.set_properties(**{'text-align': 'center'})\n\ndisplay(formatted_stats)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:29.435861Z","iopub.execute_input":"2023-12-13T23:01:29.436135Z","iopub.status.idle":"2023-12-13T23:01:30.108153Z","shell.execute_reply.started":"2023-12-13T23:01:29.436112Z","shell.execute_reply":"2023-12-13T23:01:30.107209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Cancer Positive Age Distribution:**","metadata":{}},{"cell_type":"code","source":"import seaborn as sn\nmy_colors = [\"#8a4dbf\", \"#9f76e6\", \"#9569e6\", \"#aa92e6\", \"#bba1e6\", \"#a991e6\"]\n# Create a figure with two subplots arranged vertically\nf, (a0, a1) = plt.subplots(2, 1, gridspec_kw={'height_ratios': [3, 1]}, figsize=(14, 10))\n# Plot a histogram with a kernel density estimate (KDE) for the 'cancer_ages' data, using the color from 'my_colors'\nsn.histplot(data=cancer_ages, kde=True, color=my_colors[5], ax=a0)\n\n# Add vertical lines and text annotations for mean, min, and max values on the histogram\na0.axvline(x=cancer_ages.mean(), ls=\":\", lw=2, color=\"black\")\na0.text(x=cancer_ages.mean()+1, y=0.018, s=f\"Mean: {cancer_ages.mean():.2f}\", size=17, color=\"black\", weight=\"bold\")\na0.axvline(x=cancer_ages.min(), ls=\":\", lw=2, color=\"black\")\na0.text(x=cancer_ages.min()+1, y=0.008, s=f\"Min: {cancer_ages.min()}\", size=17, color=\"black\", weight=\"bold\")\na0.axvline(x=cancer_ages.max(), ls=\":\", lw=2, color=\"black\")\na0.text(x=cancer_ages.max()-7, y=0.037, s=f\"Max: {cancer_ages.max()}\", size=17, color=\"black\", weight=\"bold\")\n\n# Plot a boxen plot for the 'cancer_ages' data on the second subplot\nsn.boxenplot(x=cancer_ages, ax=a1, color=my_colors[2])\n# Set labels for the x-axis and y-axis on the second subplot\na1.set(xlabel=\"Age\", ylabel=\"\")\n# Set tick label font size on the second subplot\na1.tick_params(labelsize=12)\n\n# Set the overall title for the entire figure\nplt.suptitle(\"Cancer Positive Age Distribution\", weight=\"bold\", size=20)\n# Set labels for the x-axis and y-axis on the first subplot\na0.set(xlabel=\"Age\", ylabel=\"Density\")\n# Set tick label font size on the first subplot\na0.tick_params(labelsize=12)\n# Configure spines (axes borders) for the first subplot\na0.spines[\"top\"].set_visible(False)\na0.spines[\"right\"].set_visible(False)\na0.spines[\"left\"].set_linewidth(2)\na0.spines[\"bottom\"].set_linewidth(2)\n\n# Adjust layout to prevent clipping of titles and labels\nplt.tight_layout()\n# Display the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:30.109293Z","iopub.execute_input":"2023-12-13T23:01:30.109663Z","iopub.status.idle":"2023-12-13T23:01:30.593671Z","shell.execute_reply.started":"2023-12-13T23:01:30.109637Z","shell.execute_reply":"2023-12-13T23:01:30.592627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Number of Images Taken per Patient:**","metadata":{}},{"cell_type":"code","source":"num_images_per_patient = df1['patient_id'].value_counts()\n\nplt.figure(figsize=(14, 8))\nsn.countplot(x=num_images_per_patient, palette=['mediumvioletred'], saturation=0.7)\n\nplt.title(\"Number of Images Taken per Patient\", fontsize=20, fontweight=\"bold\")\nplt.xlabel('Number of Images Taken', fontsize=14, fontweight=\"bold\")\nplt.ylabel('Count of Patients', fontsize=14, fontweight=\"bold\")\n\nplt.xticks(rotation=90)\nplt.tick_params(labelsize=12)\nplt.tight_layout()\nsn.despine(trim=True)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:30.59542Z","iopub.execute_input":"2023-12-13T23:01:30.596437Z","iopub.status.idle":"2023-12-13T23:01:30.896071Z","shell.execute_reply.started":"2023-12-13T23:01:30.596407Z","shell.execute_reply":"2023-12-13T23:01:30.89511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 3, figsize=(18, 12))\nsn.countplot(x=df1['laterality'], palette='Blues_r', ax=ax[0, 0])\nsn.countplot(x=df1['implant'], palette='Greens_r', ax=ax[0, 1])\nsn.countplot(x=df1['difficult_negative_case'], palette='Reds_r', ax=ax[0, 2])\nsn.countplot(x=df1['view'], palette='Oranges_r', ax=ax[1, 0])\nsn.countplot(x=df1['density'], palette='Purples_r', order=['A', 'B', 'C', 'D'], ax=ax[1, 1])\nsn.countplot(x=df1['site_id'], palette='Greys_r', ax=ax[1, 2])\n\nax[0, 0].set_title('Laterality', fontsize=18, fontweight='bold')\nax[0, 1].set_title('Implant', fontsize=18, fontweight='bold')\nax[0, 2].set_title('Difficult Negative Case', fontsize=18, fontweight='bold')\nax[1, 0].set_title('View', fontsize=18, fontweight='bold')\nax[1, 1].set_title('Density', fontsize=18, fontweight='bold')\nax[1, 2].set_title('Site ID', fontsize=18, fontweight='bold')\n\nfor i in range(2):\n    for j in range(3):\n        ax[i, j].set_xlabel('')\n        ax[i, j].set_ylabel('')\n        ax[i, j].tick_params(axis='both', which='major', labelsize=14)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:30.897327Z","iopub.execute_input":"2023-12-13T23:01:30.897577Z","iopub.status.idle":"2023-12-13T23:01:32.04616Z","shell.execute_reply.started":"2023-12-13T23:01:30.897556Z","shell.execute_reply":"2023-12-13T23:01:32.045463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**--->Observations:**\n1. In terms of laterality, the photos are balanced, suggesting that the same amount of images were captured for both the left and right breast.\n2. Implants are uncommon, with just a few photos indicating their presence.\n3. Several photos were difficult to diagnose, which may indicate the existence of breast tissue that is more complicated or unusual.\n4. The bulk of photos consist of two different types of views: CC (craniocaudal) and MLO (mediolateral oblique).\n5. The density of breast tissue in the photos is often in the medium range (B and C), with a minority of photographs showing breast tissue density that is either extremely dense (D) or extremely low (A).\n6. The photos were obtained in a balanced manner at two distinct locations, which may indicate that the data came from two distinct medical institutions or imaging centers.","metadata":{}},{"cell_type":"code","source":"## unstack(): function reshapes the result into a pivot table with 'cancer' as the index and 'biopsy' as columns.\nbiopsy_counts = df1.groupby(['cancer', 'biopsy']).size().unstack(fill_value=0)\nbiopsy_perc = biopsy_counts.apply(lambda x: x / x.sum(), axis=1)\n\n# create fig\nfig, ax = plt.subplots(1, 2, figsize=(12, 6))\n\n# fig1 (countplot)\nsn.countplot(x='biopsy', hue='cancer', data=df1, palette=['mediumvioletred', 'green'], ax=ax[0])\nax[0].bar_label(ax[0].containers[0], label_type='edge')\nax[0].bar_label(ax[0].containers[1], label_type='edge')\n\n# fig2 (heatmap) \nsn.heatmap(biopsy_perc, square=True, annot=True, fmt='.1%', cmap='Purples', ax=ax[1], cbar=False)\nax[1].set_yticklabels(ax[1].get_yticklabels(), rotation=0)\n\nplt.subplots_adjust(wspace=0.3)\nax[0].set_title(\"Number of images resulting in a biopsy\")\nax[1].set_title(\"Percentage of images resulting in a biopsy\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:32.047081Z","iopub.execute_input":"2023-12-13T23:01:32.047957Z","iopub.status.idle":"2023-12-13T23:01:32.362477Z","shell.execute_reply.started":"2023-12-13T23:01:32.047932Z","shell.execute_reply":"2023-12-13T23:01:32.36183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note: A biopsy is a medical procedure in which a sample of tissue or cells is taken from a person's body for further analysis","metadata":{}},{"cell_type":"markdown","source":"**Image Analysis:**","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(16, 4))\n\nsn.countplot(x='invasive', data=df1[df1['cancer'] == True], ax=ax[0], color='mediumvioletred')\nsn.countplot(x='BIRADS', data=df1[df1['cancer'] == False], order=[0, 1, 2], ax=ax[1], color='green')\nsn.countplot(x='BIRADS', data=df1[df1['cancer'] == True], order=[0, 1, 2], ax=ax[2], color='purple')\n\nax[0].set_title(\"Count of Invasive Cancer Images\")\nax[0].set_xlabel(\"Invasive\")\nax[0].set_ylabel(\"Count\")\n\nax[1].set_title(\"BIRADS for Healthy Images\")\nax[1].set_xlabel(\"BIRADS\")\nax[1].set_ylabel(\"Count\")\n\nax[2].set_title(\"BIRADS for Cancer Images\")\nax[2].set_xlabel(\"BIRADS\")\nax[2].set_ylabel(\"Count\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:32.36333Z","iopub.execute_input":"2023-12-13T23:01:32.364199Z","iopub.status.idle":"2023-12-13T23:01:32.798748Z","shell.execute_reply.started":"2023-12-13T23:01:32.364174Z","shell.execute_reply":"2023-12-13T23:01:32.797679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note : BIRADS value signification:\n\n0: required follow-up\n\n1: rated as negative for cancer\n\n2: rated as normal","metadata":{}},{"cell_type":"markdown","source":"Note2 : \"invasive\" means something that aggressively spreads .","metadata":{}},{"cell_type":"markdown","source":"**Machine id:**\n","metadata":{}},{"cell_type":"code","source":"count_machine=df1.groupby(by=\"machine_id\").count()[\"patient_id\"]\ncount_machine.reset_index().head()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:32.799537Z","iopub.execute_input":"2023-12-13T23:01:32.79975Z","iopub.status.idle":"2023-12-13T23:01:32.819811Z","shell.execute_reply.started":"2023-12-13T23:01:32.799729Z","shell.execute_reply":"2023-12-13T23:01:32.818764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nplt.title(\"Number of images per machine_id\")\nsn.barplot(data=count_machine.reset_index(), x=\"machine_id\", y=\"patient_id\", color='mediumvioletred')\nplt.ylabel(\"Number of images\")\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:32.821129Z","iopub.execute_input":"2023-12-13T23:01:32.821431Z","iopub.status.idle":"2023-12-13T23:01:33.072104Z","shell.execute_reply.started":"2023-12-13T23:01:32.821406Z","shell.execute_reply":"2023-12-13T23:01:33.071214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note : \"not-malignant cancer\" cases should be selected from \"patients with biopsy but without malignant cancer.\"\n\nWe took this choice because it seems more logical, but another choice might be correct.","metadata":{}},{"cell_type":"code","source":"# The not-malignant cancer cases were limited into biopsy cases.\nDF_train = df[df['biopsy'] == 1].reset_index(drop = True)\nDF_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.0736Z","iopub.execute_input":"2023-12-13T23:01:33.074115Z","iopub.status.idle":"2023-12-13T23:01:33.089949Z","shell.execute_reply.started":"2023-12-13T23:01:33.074088Z","shell.execute_reply":"2023-12-13T23:01:33.089015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Creating a balanced dataset :**","metadata":{}},{"cell_type":"code","source":"\nDF_train = DF_train.groupby(['cancer']).apply(lambda x: x.sample(1158, replace = True)\n                                                      ).reset_index(drop = True)\nprint('New Data Size:', DF_train.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.091475Z","iopub.execute_input":"2023-12-13T23:01:33.091772Z","iopub.status.idle":"2023-12-13T23:01:33.107903Z","shell.execute_reply.started":"2023-12-13T23:01:33.091749Z","shell.execute_reply":"2023-12-13T23:01:33.10685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.DataFrame(np.concatenate([['Biopsy but Not Malignant'] * len(DF_train[(DF_train['biopsy'] == 1) & (DF_train['cancer'] == 0)]),\n                                    ['Malignant Cancer'] * len(DF_train[DF_train['cancer'] == 1]),\n                                    ['Invasive Cancer'] * len(DF_train[(DF_train['cancer'] == 1) & (DF_train['invasive'] == 1)])]),\n                   columns=[\"class\"])\n\ncolors = [\"mediumvioletred\", \"blue\", \"green\"]\nsn.countplot(x='class', data=data, palette=colors)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.109384Z","iopub.execute_input":"2023-12-13T23:01:33.109737Z","iopub.status.idle":"2023-12-13T23:01:33.260827Z","shell.execute_reply.started":"2023-12-13T23:01:33.109705Z","shell.execute_reply":"2023-12-13T23:01:33.259488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Creating the path to the images:** \n\n( path to the image in the new dataset with .png images : /kaggle/input/rsna-breast-cancer-512-pngs)","metadata":{}},{"cell_type":"code","source":"# the path to the image data\nRSNA_512_path = '/kaggle/input/rsna-breast-cancer-512-pngs'","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.263179Z","iopub.execute_input":"2023-12-13T23:01:33.263859Z","iopub.status.idle":"2023-12-13T23:01:33.27Z","shell.execute_reply.started":"2023-12-13T23:01:33.263828Z","shell.execute_reply":"2023-12-13T23:01:33.267233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the path to each image.\nfor i in range(len(DF_train)):\n    DF_train.loc[i, 'path'] = os.path.join(RSNA_512_path + '/' + str(DF_train.loc[i, 'patient_id']) + '_' + str(DF_train.loc[i, 'image_id']) + '.png')\nDF_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.271706Z","iopub.execute_input":"2023-12-13T23:01:33.271979Z","iopub.status.idle":"2023-12-13T23:01:33.646056Z","shell.execute_reply.started":"2023-12-13T23:01:33.271954Z","shell.execute_reply":"2023-12-13T23:01:33.644777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a sample path\nDF_train.loc[0, 'path']","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.647114Z","iopub.execute_input":"2023-12-13T23:01:33.647379Z","iopub.status.idle":"2023-12-13T23:01:33.653191Z","shell.execute_reply.started":"2023-12-13T23:01:33.647328Z","shell.execute_reply":"2023-12-13T23:01:33.652211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a sample image\nimport cv2\nimg = cv2.imread(DF_train.loc[0, 'path'])\nplt.imshow(img, cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.654687Z","iopub.execute_input":"2023-12-13T23:01:33.655037Z","iopub.status.idle":"2023-12-13T23:01:33.911924Z","shell.execute_reply.started":"2023-12-13T23:01:33.655004Z","shell.execute_reply":"2023-12-13T23:01:33.910855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.914221Z","iopub.execute_input":"2023-12-13T23:01:33.91485Z","iopub.status.idle":"2023-12-13T23:01:33.921305Z","shell.execute_reply.started":"2023-12-13T23:01:33.914814Z","shell.execute_reply":"2023-12-13T23:01:33.920406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Divide the Data into Training ,Test and Validation:**\n\nPs : Normal and cancer images must be equally distrubuted.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# Split into train and temp (combined validation and test) sets\ntrain_df, temp_df = train_test_split(DF_train, \n                                      test_size=0.4, \n                                      random_state=2018, \n                                      stratify=DF_train['cancer'])\n\n# Further split temp_df into validation and test sets\nvalidation_df, test_df = train_test_split(temp_df, \n                                           test_size=0.5, \n                                           random_state=2018, \n                                           stratify=temp_df['cancer'])\n\n# afficher la taille de chaque ensemble\nprint('train:', train_df.shape[0])\nprint('validation:', validation_df.shape[0])\nprint('test:', test_df.shape[0])\nprint('train', train_df['cancer'].value_counts())\nprint('test', test_df['cancer'].value_counts())\nprint('validation', validation_df['cancer'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.924255Z","iopub.execute_input":"2023-12-13T23:01:33.924826Z","iopub.status.idle":"2023-12-13T23:01:33.939282Z","shell.execute_reply.started":"2023-12-13T23:01:33.924801Z","shell.execute_reply":"2023-12-13T23:01:33.938568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training data\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.940061Z","iopub.execute_input":"2023-12-13T23:01:33.940286Z","iopub.status.idle":"2023-12-13T23:01:33.958001Z","shell.execute_reply.started":"2023-12-13T23:01:33.940265Z","shell.execute_reply":"2023-12-13T23:01:33.957035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test data\ntest_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.959206Z","iopub.execute_input":"2023-12-13T23:01:33.959468Z","iopub.status.idle":"2023-12-13T23:01:33.98031Z","shell.execute_reply.started":"2023-12-13T23:01:33.959445Z","shell.execute_reply":"2023-12-13T23:01:33.979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df[train_df['cancer'] == 0])\nlen(train_df[train_df['cancer'] == 1])\nlen(test_df[test_df['cancer'] == 0])\nlen(test_df[test_df['cancer'] == 1])\nlen(validation_df[validation_df['cancer'] == 0])\nlen(validation_df[validation_df['cancer'] == 1])\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.98272Z","iopub.execute_input":"2023-12-13T23:01:33.983094Z","iopub.status.idle":"2023-12-13T23:01:33.995397Z","shell.execute_reply.started":"2023-12-13T23:01:33.983062Z","shell.execute_reply":"2023-12-13T23:01:33.994297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training data:**","metadata":{}},{"cell_type":"code","source":"# Pick up normal images from the training data.\ntrain_df_normal = train_df[train_df['cancer'] == 0].reset_index(drop = True)\nprint(\"normal images:\")\nprint(len(train_df_normal))\ntrain_df_normal.head(5)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:33.998107Z","iopub.execute_input":"2023-12-13T23:01:33.998435Z","iopub.status.idle":"2023-12-13T23:01:34.016236Z","shell.execute_reply.started":"2023-12-13T23:01:33.998411Z","shell.execute_reply":"2023-12-13T23:01:34.015402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up cancer images from the training data.\ntrain_df_cancer = train_df[train_df['cancer'] == 1].reset_index(drop = True)\nprint(\"cancer images:\")\nprint(len(train_df_cancer))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:34.017632Z","iopub.execute_input":"2023-12-13T23:01:34.017942Z","iopub.status.idle":"2023-12-13T23:01:34.024859Z","shell.execute_reply.started":"2023-12-13T23:01:34.017912Z","shell.execute_reply":"2023-12-13T23:01:34.023989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Test:**","metadata":{}},{"cell_type":"code","source":"# Pick up normal images from the test data.\ntest_df_normal = test_df[test_df['cancer'] == 0].reset_index(drop = True)\nprint(\"normal images:\")\nprint(len(test_df_normal))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:34.025945Z","iopub.execute_input":"2023-12-13T23:01:34.026772Z","iopub.status.idle":"2023-12-13T23:01:34.038532Z","shell.execute_reply.started":"2023-12-13T23:01:34.026717Z","shell.execute_reply":"2023-12-13T23:01:34.037004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up cancer images from the test data.\ntest_df_cancer = test_df[test_df['cancer'] == 1].reset_index(drop = True)\nprint(\"cancer images:\")\nprint(len(test_df_cancer))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:34.040192Z","iopub.execute_input":"2023-12-13T23:01:34.040742Z","iopub.status.idle":"2023-12-13T23:01:34.049768Z","shell.execute_reply.started":"2023-12-13T23:01:34.040706Z","shell.execute_reply":"2023-12-13T23:01:34.048621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now , We have to store the datasets in each folder to be used with our model.","metadata":{}},{"cell_type":"code","source":"import shutil\nimport os\n# Define the destination directory.\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/normal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in train_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:34.051449Z","iopub.execute_input":"2023-12-13T23:01:34.051853Z","iopub.status.idle":"2023-12-13T23:01:38.526852Z","shell.execute_reply.started":"2023-12-13T23:01:34.051819Z","shell.execute_reply":"2023-12-13T23:01:38.526141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/cancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in train_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:38.527832Z","iopub.execute_input":"2023-12-13T23:01:38.528122Z","iopub.status.idle":"2023-12-13T23:01:41.660692Z","shell.execute_reply.started":"2023-12-13T23:01:38.528099Z","shell.execute_reply":"2023-12-13T23:01:41.659647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/test'\ndestination_dir_sub = '/kaggle/working/test/normal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in test_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:41.666844Z","iopub.execute_input":"2023-12-13T23:01:41.66718Z","iopub.status.idle":"2023-12-13T23:01:42.911981Z","shell.execute_reply.started":"2023-12-13T23:01:41.667155Z","shell.execute_reply":"2023-12-13T23:01:42.910415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/test'\ndestination_dir_sub = '/kaggle/working/test/cancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in test_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:42.913469Z","iopub.execute_input":"2023-12-13T23:01:42.913763Z","iopub.status.idle":"2023-12-13T23:01:43.6841Z","shell.execute_reply.started":"2023-12-13T23:01:42.913736Z","shell.execute_reply":"2023-12-13T23:01:43.683392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df_normal = validation_df[validation_df['cancer'] == 0].reset_index(drop = True)\n\n# copier les images vers le dossier destination\ndestination_dir = '/kaggle/working/validation'\ndestination_dir_sub = '/kaggle/working/validation/normal'\n\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# copier les images vers le dossier destination\nfor path in val_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:43.685253Z","iopub.execute_input":"2023-12-13T23:01:43.685608Z","iopub.status.idle":"2023-12-13T23:01:44.890605Z","shell.execute_reply.started":"2023-12-13T23:01:43.685577Z","shell.execute_reply":"2023-12-13T23:01:44.889452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df_cancer = validation_df[validation_df['cancer'] == 1].reset_index(drop = True)\n\n# copier les images vers le dossier destination\ndestination_dir = '/kaggle/working/validation'\ndestination_dir_sub = '/kaggle/working/validation/cancer'\n\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# copier les images vers le dossier destination\nfor path in val_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:44.893192Z","iopub.execute_input":"2023-12-13T23:01:44.894145Z","iopub.status.idle":"2023-12-13T23:01:45.508803Z","shell.execute_reply.started":"2023-12-13T23:01:44.894095Z","shell.execute_reply":"2023-12-13T23:01:45.507829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see if it worked and show some samples images :","metadata":{}},{"cell_type":"code","source":"import glob\nnormal_train_images = glob.glob('/kaggle/working/train/normal/*.png')\ncancer_train_images = glob.glob('/kaggle/working/train/cancer/*.png')","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:45.509869Z","iopub.execute_input":"2023-12-13T23:01:45.510185Z","iopub.status.idle":"2023-12-13T23:01:45.519209Z","shell.execute_reply.started":"2023-12-13T23:01:45.510157Z","shell.execute_reply":"2023-12-13T23:01:45.518235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See normal images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(normal_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Normal')\nfig.tight_layout()    \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:45.520438Z","iopub.execute_input":"2023-12-13T23:01:45.52074Z","iopub.status.idle":"2023-12-13T23:01:46.472881Z","shell.execute_reply.started":"2023-12-13T23:01:45.520711Z","shell.execute_reply":"2023-12-13T23:01:46.471502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See cancer images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(cancer_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Cancer')\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-13T23:01:46.474102Z","iopub.execute_input":"2023-12-13T23:01:46.474422Z","iopub.status.idle":"2023-12-13T23:01:47.252665Z","shell.execute_reply.started":"2023-12-13T23:01:46.474396Z","shell.execute_reply":"2023-12-13T23:01:47.251287Z"},"trusted":true},"execution_count":null,"outputs":[]}]}