{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"}],"dockerImageVersionId":30357,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nThis is an exploratory data notebook. I will gradualy refine it, in order to get a better understanding of the data distribution.","metadata":{}},{"cell_type":"markdown","source":"# Analysis preparation","metadata":{}},{"cell_type":"markdown","source":"## Import packages","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tqdm import tqdm\nimport pydicom as dcm\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:39.95381Z","iopub.execute_input":"2025-03-29T19:43:39.954372Z","iopub.status.idle":"2025-03-29T19:43:39.961795Z","shell.execute_reply.started":"2025-03-29T19:43:39.954331Z","shell.execute_reply":"2025-03-29T19:43:39.960179Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Read the data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:39.964614Z","iopub.execute_input":"2025-03-29T19:43:39.965023Z","iopub.status.idle":"2025-03-29T19:43:40.05925Z","shell.execute_reply.started":"2025-03-29T19:43:39.964991Z","shell.execute_reply":"2025-03-29T19:43:40.0579Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# What we can learn about the data at a quick glimpse?","metadata":{}},{"cell_type":"code","source":"print(\"Train / test data shape: \", train_df.shape, test_df.shape)\nprint(\"Train images folders: \", len(os.listdir(\"/kaggle/input/rsna-breast-cancer-detection/train_images\")))\nprint(\"Test images folders: \", len(os.listdir(\"/kaggle/input/rsna-breast-cancer-detection/test_images\")))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.061508Z","iopub.execute_input":"2025-03-29T19:43:40.061846Z","iopub.status.idle":"2025-03-29T19:43:40.073276Z","shell.execute_reply.started":"2025-03-29T19:43:40.061816Z","shell.execute_reply":"2025-03-29T19:43:40.071941Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are 54,706 entries in the train dataset and only 11913 image folders in train_images. As the two numbers are not dividing exactly, it results that we should expect to have different number of images in the subfolders. We will get to this in a moment.","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.074855Z","iopub.execute_input":"2025-03-29T19:43:40.075257Z","iopub.status.idle":"2025-03-29T19:43:40.106586Z","shell.execute_reply.started":"2025-03-29T19:43:40.075207Z","shell.execute_reply":"2025-03-29T19:43:40.105253Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are several columns with missing data:\n* age  \n* BIRADS  (meaning: Breast Imaging - Reporting and Database System)\n* density  \n","metadata":{}},{"cell_type":"markdown","source":"Let's define few auxiliary functions - for missing data and for unique values.","metadata":{}},{"cell_type":"code","source":"def missing_data(data):\n    total = data.isnull().sum().sort_values(ascending = False)\n    percent = (data.isnull().sum()/data.isnull().count()*100).sort_values(ascending = False)\n    return np.transpose(pd.concat([total, percent], axis=1, keys=['Total', 'Percent']))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.110044Z","iopub.execute_input":"2025-03-29T19:43:40.110524Z","iopub.status.idle":"2025-03-29T19:43:40.117913Z","shell.execute_reply.started":"2025-03-29T19:43:40.110486Z","shell.execute_reply":"2025-03-29T19:43:40.116237Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def unique_values(data):\n    total = data.count()\n    tt = pd.DataFrame(total)\n    tt.columns = ['Total']\n    uniques = []\n    for col in data.columns:\n        unique = data[col].nunique()\n        uniques.append(unique)\n    tt['Uniques'] = uniques\n    return np.transpose(tt)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.119527Z","iopub.execute_input":"2025-03-29T19:43:40.119895Z","iopub.status.idle":"2025-03-29T19:43:40.132915Z","shell.execute_reply.started":"2025-03-29T19:43:40.119861Z","shell.execute_reply":"2025-03-29T19:43:40.131516Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We look now to missing data.","metadata":{}},{"cell_type":"code","source":"missing_data(train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.135232Z","iopub.execute_input":"2025-03-29T19:43:40.135688Z","iopub.status.idle":"2025-03-29T19:43:40.206228Z","shell.execute_reply.started":"2025-03-29T19:43:40.135646Z","shell.execute_reply":"2025-03-29T19:43:40.20446Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"And to unique values.","metadata":{}},{"cell_type":"code","source":"unique_values(train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.208256Z","iopub.execute_input":"2025-03-29T19:43:40.208801Z","iopub.status.idle":"2025-03-29T19:43:40.259672Z","shell.execute_reply.started":"2025-03-29T19:43:40.208751Z","shell.execute_reply":"2025-03-29T19:43:40.258011Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's look also to the first data rows in the dataset.","metadata":{}},{"cell_type":"code","source":"train_df.head(6)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.261222Z","iopub.execute_input":"2025-03-29T19:43:40.261618Z","iopub.status.idle":"2025-03-29T19:43:40.279828Z","shell.execute_reply.started":"2025-03-29T19:43:40.261584Z","shell.execute_reply":"2025-03-29T19:43:40.278607Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In the first 4 rows, there is data about the same patient (identified by patient_id), totally 4 images, having laterality and view with values L & R, CC & MLO respectively.","metadata":{}},{"cell_type":"markdown","source":"We look also to test data distribution.","metadata":{}},{"cell_type":"code","source":"test_df.info()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.281315Z","iopub.execute_input":"2025-03-29T19:43:40.281684Z","iopub.status.idle":"2025-03-29T19:43:40.300651Z","shell.execute_reply.started":"2025-03-29T19:43:40.28165Z","shell.execute_reply":"2025-03-29T19:43:40.298992Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's also look for missing data and unique values in test data.","metadata":{}},{"cell_type":"code","source":"missing_data(test_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.304971Z","iopub.execute_input":"2025-03-29T19:43:40.30547Z","iopub.status.idle":"2025-03-29T19:43:40.3356Z","shell.execute_reply.started":"2025-03-29T19:43:40.305435Z","shell.execute_reply":"2025-03-29T19:43:40.333943Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_values(test_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.337411Z","iopub.execute_input":"2025-03-29T19:43:40.337851Z","iopub.status.idle":"2025-03-29T19:43:40.359834Z","shell.execute_reply.started":"2025-03-29T19:43:40.337799Z","shell.execute_reply":"2025-03-29T19:43:40.358521Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.361877Z","iopub.execute_input":"2025-03-29T19:43:40.362474Z","iopub.status.idle":"2025-03-29T19:43:40.385901Z","shell.execute_reply.started":"2025-03-29T19:43:40.362413Z","shell.execute_reply":"2025-03-29T19:43:40.384376Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Test data has info for only one patient. The following fields, present in train data, are not present in test data:\n* cancer\n* biopsy  \n* invasive  \n* BIRADS  \n* density  \n* difficult_negative_case\n","metadata":{}},{"cell_type":"markdown","source":"# Do we have images for all entries in the train data?\n\nTo check this, we compare the list of image ids in the dataset with the list of names of images in the images folders. We will also check that we do have an image folder for each patient_id.","metadata":{}},{"cell_type":"code","source":"file_list = []\ntrain_path = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\nfolder_list = list(os.listdir(train_path))\nfor folder in tqdm(os.listdir(train_path)):\n    file_list += [x.split(\".dcm\")[0] for x in os.listdir(os.path.join(train_path, folder))]\nprint(len(folder_list), len(file_list))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:40.387496Z","iopub.execute_input":"2025-03-29T19:43:40.387872Z","iopub.status.idle":"2025-03-29T19:43:47.259777Z","shell.execute_reply.started":"2025-03-29T19:43:40.38784Z","shell.execute_reply":"2025-03-29T19:43:47.258381Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"diff = list(set(folder_list) - set([str(x) for x in train_df.patient_id.unique()]))\nprint(\"Differences in patient/folder list: \",len(diff))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:47.261754Z","iopub.execute_input":"2025-03-29T19:43:47.262319Z","iopub.status.idle":"2025-03-29T19:43:47.287123Z","shell.execute_reply.started":"2025-03-29T19:43:47.262268Z","shell.execute_reply":"2025-03-29T19:43:47.285901Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"diff = list(set(file_list) - set([str(x) for x in train_df.image_id.unique()]))\nprint(\"Differences in patient/folder list: \",len(diff))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:47.288704Z","iopub.execute_input":"2025-03-29T19:43:47.289169Z","iopub.status.idle":"2025-03-29T19:43:47.355506Z","shell.execute_reply.started":"2025-03-29T19:43:47.289101Z","shell.execute_reply":"2025-03-29T19:43:47.354289Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We can conclude that all the patients and all the images are indexed in the train dataset.  \nLet's look next to the distribution of laterality and view in the patient / images set.","metadata":{}},{"cell_type":"code","source":"train_df.laterality.value_counts()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:47.357165Z","iopub.execute_input":"2025-03-29T19:43:47.357574Z","iopub.status.idle":"2025-03-29T19:43:47.37194Z","shell.execute_reply.started":"2025-03-29T19:43:47.357539Z","shell.execute_reply":"2025-03-29T19:43:47.370152Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.view.value_counts()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:47.373624Z","iopub.execute_input":"2025-03-29T19:43:47.374013Z","iopub.status.idle":"2025-03-29T19:43:47.392053Z","shell.execute_reply.started":"2025-03-29T19:43:47.373964Z","shell.execute_reply":"2025-03-29T19:43:47.390588Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Overall, most of the images have at least a R and L laterality; as per view, majority will present both MLO and CC views, with a small set of patients having as well other 4 views. Let's group laterality and view to get all the combinations.","metadata":{}},{"cell_type":"code","source":"train_df[\"lv\"] = train_df[[\"laterality\", \"view\"]].apply(lambda x: \"_\".join(x), axis=1)","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:47.394368Z","iopub.execute_input":"2025-03-29T19:43:47.394958Z","iopub.status.idle":"2025-03-29T19:43:47.860046Z","shell.execute_reply.started":"2025-03-29T19:43:47.394907Z","shell.execute_reply":"2025-03-29T19:43:47.858599Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[\"lv\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:47.861761Z","iopub.execute_input":"2025-03-29T19:43:47.862199Z","iopub.status.idle":"2025-03-29T19:43:47.880721Z","shell.execute_reply.started":"2025-03-29T19:43:47.862163Z","shell.execute_reply":"2025-03-29T19:43:47.879189Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# What are the data sources?","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"machine_id\", \"site_id\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"machine_id\", \"site_id\", \"count\"]\nsns.barplot(data=train_agg_df, x=\"machine_id\", y=\"count\", hue=\"site_id\")\nplt.title(\"Number of images per machine id, grouped by site id\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-03-29T19:43:47.88263Z","iopub.execute_input":"2025-03-29T19:43:47.883093Z","iopub.status.idle":"2025-03-29T19:43:48.252027Z","shell.execute_reply.started":"2025-03-29T19:43:47.883041Z","shell.execute_reply":"2025-03-29T19:43:48.250804Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Majority of the data sources are machine_id = 49 and site_id = 1","metadata":{}},{"cell_type":"markdown","source":"# What is the distribution of number of images per patient?\n\nWe already noticed that it is not a standard number of images or view and laterality distribution per patient. Let's see what is the distribution of number of images / patient.","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"patient_id\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"patient_id\", \"images\"]\ntrain_agg_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.253564Z","iopub.execute_input":"2025-03-29T19:43:48.253889Z","iopub.status.idle":"2025-03-29T19:43:48.274264Z","shell.execute_reply.started":"2025-03-29T19:43:48.253858Z","shell.execute_reply":"2025-03-29T19:43:48.272721Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_agg_df.images.value_counts()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.276749Z","iopub.execute_input":"2025-03-29T19:43:48.277203Z","iopub.status.idle":"2025-03-29T19:43:48.286612Z","shell.execute_reply.started":"2025-03-29T19:43:48.277159Z","shell.execute_reply":"2025-03-29T19:43:48.285433Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"A majority of 8,233 patients have 4 images. Then there are less and less patients with increasingly number of images. Only 2 patients have 14 images.\n\nLet's also check how many unique laterality/view tuple have each patient.","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"patient_id\"])[\"lv\"].nunique().reset_index()\ntrain_agg_df.columns = [\"patient_id\", \"lat_view\"]","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.289052Z","iopub.execute_input":"2025-03-29T19:43:48.290568Z","iopub.status.idle":"2025-03-29T19:43:48.316314Z","shell.execute_reply.started":"2025-03-29T19:43:48.290502Z","shell.execute_reply":"2025-03-29T19:43:48.314535Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_agg_df.lat_view.value_counts()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.317776Z","iopub.execute_input":"2025-03-29T19:43:48.31817Z","iopub.status.idle":"2025-03-29T19:43:48.328471Z","shell.execute_reply.started":"2025-03-29T19:43:48.318109Z","shell.execute_reply":"2025-03-29T19:43:48.326887Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the 11913 different patients, allmost all (11,885) will have a certain combination of 4 laterality / view (with most frequent L/R , MLO / CC).","metadata":{}},{"cell_type":"markdown","source":"# How old are the patients? And how old are the ones with cancer?","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"patient_id\", \"age\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"patient\", \"age\", \"count\"]\nsns.histplot(train_agg_df.age, bins=20)\nplt.title(\"Number of patients per age groups\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.330207Z","iopub.execute_input":"2025-03-29T19:43:48.330568Z","iopub.status.idle":"2025-03-29T19:43:48.623043Z","shell.execute_reply.started":"2025-03-29T19:43:48.330538Z","shell.execute_reply":"2025-03-29T19:43:48.621474Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's take only the patients with cancer.","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.loc[train_df.cancer==1].groupby([\"patient_id\", \"age\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"patient\", \"age\", \"count\"]\nsns.histplot(train_agg_df.age, bins=20)\nplt.title(\"Number of patients with cancer per age groups\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.624724Z","iopub.execute_input":"2025-03-29T19:43:48.625081Z","iopub.status.idle":"2025-03-29T19:43:48.891922Z","shell.execute_reply.started":"2025-03-29T19:43:48.625048Z","shell.execute_reply":"2025-03-29T19:43:48.890453Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Are there patients that have both cancer and not cancer diagnosis?\n ","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"patient_id\", \"cancer\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"patient_id\", \"cancer\", \"count\"]","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.893539Z","iopub.execute_input":"2025-03-29T19:43:48.893892Z","iopub.status.idle":"2025-03-29T19:43:48.915856Z","shell.execute_reply.started":"2025-03-29T19:43:48.893861Z","shell.execute_reply":"2025-03-29T19:43:48.914572Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Total cases: \", train_agg_df.shape[0])\nprint(\"Total patients: \", train_agg_df.patient_id.nunique())\nprint(\"Total cancer cases: \", train_agg_df.loc[train_agg_df.cancer==1].patient_id.nunique())\npatients_with_cancer = train_agg_df.loc[train_agg_df.cancer==1].patient_id.unique()\npat_with_both = train_agg_df.loc[~train_agg_df.patient_id.isin(patients_with_cancer)].patient_id.nunique()\nprint(\"Total diagnosed without cancer: \", pat_with_both)","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.922031Z","iopub.execute_input":"2025-03-29T19:43:48.922548Z","iopub.status.idle":"2025-03-29T19:43:48.938759Z","shell.execute_reply.started":"2025-03-29T19:43:48.922512Z","shell.execute_reply":"2025-03-29T19:43:48.936661Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_agg2_df = train_agg_df.groupby([\"patient_id\"])[\"cancer\"].count().reset_index()\ntrain_agg2_df.columns = [\"patient_id\", \"diagnoses\"]\ntrain_agg2_df.diagnoses.value_counts()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.940285Z","iopub.execute_input":"2025-03-29T19:43:48.940699Z","iopub.status.idle":"2025-03-29T19:43:48.961192Z","shell.execute_reply.started":"2025-03-29T19:43:48.940661Z","shell.execute_reply":"2025-03-29T19:43:48.959095Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"It appears that there are 480 patients with two dignoses (cancer and not cancer), and 486 patients with cancer only. hen, there are 11433 patients with only one diagnosis, either cancer or not cancer.","metadata":{}},{"cell_type":"markdown","source":"# How about the patients with difficult_negative_case?","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"patient_id\", \"difficult_negative_case\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"patient_id\", \"difficult_negative_case\", \"count\"]\nprint(\"Total cases: \", train_agg_df.shape[0])\nprint(\"Total patients: \", train_agg_df.patient_id.nunique())\nprint(\"Total difficult_negative_case cases: \", train_agg_df.loc[train_agg_df.difficult_negative_case==1].patient_id.nunique())\npatients_with_dnc = train_agg_df.loc[train_agg_df.difficult_negative_case==1].patient_id.unique()\npat_with_both = train_agg_df.loc[~train_agg_df.patient_id.isin(patients_with_dnc)].patient_id.nunique()\nprint(\"Total diagnosed without difficult_negative_case: \", pat_with_both)","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.963507Z","iopub.execute_input":"2025-03-29T19:43:48.963865Z","iopub.status.idle":"2025-03-29T19:43:48.993671Z","shell.execute_reply.started":"2025-03-29T19:43:48.963832Z","shell.execute_reply":"2025-03-29T19:43:48.992172Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# And how about patients with biopsy?","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"patient_id\", \"biopsy\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"patient_id\", \"biopsy\", \"count\"]\nprint(\"Total cases: \", train_agg_df.shape[0])\nprint(\"Total patients: \", train_agg_df.patient_id.nunique())\nprint(\"Total biopsy cases: \", train_agg_df.loc[train_agg_df.biopsy==1].patient_id.nunique())\npatients_with_b = train_agg_df.loc[train_agg_df.biopsy==1].patient_id.unique()\npat_with_both = train_agg_df.loc[~train_agg_df.patient_id.isin(patients_with_b)].patient_id.nunique()\nprint(\"Total diagnosed without biopsy: \", pat_with_both)","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:48.995774Z","iopub.execute_input":"2025-03-29T19:43:48.996156Z","iopub.status.idle":"2025-03-29T19:43:49.023976Z","shell.execute_reply.started":"2025-03-29T19:43:48.996101Z","shell.execute_reply":"2025-03-29T19:43:49.022593Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Patients with implant","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"implant\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"implant\", \"count\"]\nsns.barplot(data=train_agg_df, x=\"implant\", y=\"count\")\nplt.title(\"Number of images for patients with/without implant\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:49.025534Z","iopub.execute_input":"2025-03-29T19:43:49.025851Z","iopub.status.idle":"2025-03-29T19:43:49.232937Z","shell.execute_reply.started":"2025-03-29T19:43:49.025824Z","shell.execute_reply":"2025-03-29T19:43:49.231364Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# BIRADS status\n\nBIRADS stands for Breast Imaging - Reporting and Database System.","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"BIRADS\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"BIRADS\", \"count\"]\nsns.barplot(data=train_agg_df, x=\"BIRADS\", y=\"count\")\nplt.title(\"Number of images per BIRADS value\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:49.23467Z","iopub.execute_input":"2025-03-29T19:43:49.235067Z","iopub.status.idle":"2025-03-29T19:43:49.452801Z","shell.execute_reply.started":"2025-03-29T19:43:49.235029Z","shell.execute_reply":"2025-03-29T19:43:49.451406Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Invasive/non-invasive","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"invasive\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"invasive\", \"count\"]\nsns.barplot(data=train_agg_df, x=\"invasive\", y=\"count\")\nplt.title(\"Number of images for patients with/without invasive\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:49.454866Z","iopub.execute_input":"2025-03-29T19:43:49.455436Z","iopub.status.idle":"2025-03-29T19:43:49.651022Z","shell.execute_reply.started":"2025-03-29T19:43:49.455372Z","shell.execute_reply":"2025-03-29T19:43:49.64963Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Laterality/view statistics","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"laterality\", \"view\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"laterality\", \"view\", \"count\"]\nfig, ax = plt.subplots()\nsns.barplot(data=train_agg_df, x=\"view\", y=\"count\", hue=\"laterality\")\nax.set_yscale('log')\nax.set_ylabel('Number of images (log scale)')\nplt.title(\"Number of images per view, grouped by laterality\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:49.652952Z","iopub.execute_input":"2025-03-29T19:43:49.65337Z","iopub.status.idle":"2025-03-29T19:43:50.601072Z","shell.execute_reply.started":"2025-03-29T19:43:49.653336Z","shell.execute_reply":"2025-03-29T19:43:50.599663Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Biopsy/cancer statistics","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"biopsy\", \"cancer\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"biopsy\", \"cancer\", \"count\"]\nfig, ax = plt.subplots()\nsns.barplot(data=train_agg_df, x=\"biopsy\", y=\"count\", hue=\"cancer\")\nax.set_yscale('log')\nax.set_ylabel('Number of images (log scale)')\nplt.title(\"Number of images per biopsy, grouped by cancer\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:50.60269Z","iopub.execute_input":"2025-03-29T19:43:50.603075Z","iopub.status.idle":"2025-03-29T19:43:50.885812Z","shell.execute_reply.started":"2025-03-29T19:43:50.60304Z","shell.execute_reply":"2025-03-29T19:43:50.884239Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Biopsy/difficult negative cases","metadata":{}},{"cell_type":"code","source":"train_agg_df = train_df.groupby([\"biopsy\", \"difficult_negative_case\"])[\"image_id\"].count().reset_index()\ntrain_agg_df.columns = [\"biopsy\", \"difficult_negative_case\", \"count\"]\nfig, ax = plt.subplots()\nsns.barplot(data=train_agg_df, x=\"biopsy\", y=\"count\", hue=\"difficult_negative_case\")\nax.set_yscale('log')\nax.set_ylabel('Number of images (log scale)')\nplt.title(\"Number of images per biopsy, grouped by difficult_negative_case\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:50.887528Z","iopub.execute_input":"2025-03-29T19:43:50.887983Z","iopub.status.idle":"2025-03-29T19:43:51.278343Z","shell.execute_reply.started":"2025-03-29T19:43:50.887935Z","shell.execute_reply":"2025-03-29T19:43:51.276902Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# What information is in the DICOM files?","metadata":{}},{"cell_type":"markdown","source":"We define a function for extraction of data from DICOM format files.","metadata":{}},{"cell_type":"code","source":"def extract_dicom_data(data_path, patient_id):\n    images_path = os.path.join(data_path,patient_id)\n    for image in os.listdir(images_path):\n        image_id = image.split(\".dcm\")[0]\n        image_path = os.path.join(images_path, image)\n        data_row_img_data = dcm.read_file(image_path)\n        print(\"=================================================\")\n        print(f\"Patient: {patient_id} Image_id: {image_id}\")\n        print(\"=================================================\")\n        print(data_row_img_data)\n        print(\"=================================================\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:51.279979Z","iopub.execute_input":"2025-03-29T19:43:51.280434Z","iopub.status.idle":"2025-03-29T19:43:51.28889Z","shell.execute_reply.started":"2025-03-29T19:43:51.280396Z","shell.execute_reply":"2025-03-29T19:43:51.287193Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"And we test it on one patient (with 4 DICOM files).","metadata":{}},{"cell_type":"code","source":"patient_id = '10006'\nextract_dicom_data(train_path, patient_id)","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:51.290955Z","iopub.execute_input":"2025-03-29T19:43:51.29152Z","iopub.status.idle":"2025-03-29T19:43:51.591812Z","shell.execute_reply.started":"2025-03-29T19:43:51.29147Z","shell.execute_reply":"2025-03-29T19:43:51.590192Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Extract DICOM data and add it to train data\n\nHere is defined a function to process the file in DICOM format and extract data, not only print it.","metadata":{}},{"cell_type":"code","source":"def process_dicom_data(data_path, patient_id, dicom_features):\n    images_path = os.path.join(data_path,str(patient_id))\n    for image in os.listdir(images_path):\n        try:\n            image_id = image.split(\".dcm\")[0]\n            image_path = os.path.join(images_path, image)\n            data_row_img_data = dcm.read_file(image_path)\n            rows = data_row_img_data.Rows\n            columns = data_row_img_data.Columns\n            content_date = data_row_img_data.ContentDate\n            photometric_interpretation = data_row_img_data.PhotometricInterpretation\n            dicom_features.append((image_id, rows, columns, content_date, photometric_interpretation))\n        except Exception as ex:\n            print(ex)\n            continue","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:51.593922Z","iopub.execute_input":"2025-03-29T19:43:51.594468Z","iopub.status.idle":"2025-03-29T19:43:51.603655Z","shell.execute_reply.started":"2025-03-29T19:43:51.594424Z","shell.execute_reply":"2025-03-29T19:43:51.601973Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dicom_features = []\nfor patient_id in tqdm(train_df.patient_id.unique()):\n    process_dicom_data(train_path, patient_id, dicom_features)","metadata":{"execution":{"iopub.status.busy":"2025-03-29T19:43:51.606398Z","iopub.execute_input":"2025-03-29T19:43:51.606926Z","iopub.status.idle":"2025-03-29T20:42:44.428273Z","shell.execute_reply.started":"2025-03-29T19:43:51.606883Z","shell.execute_reply":"2025-03-29T20:42:44.424217Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's run this function for all files.","metadata":{}},{"cell_type":"code","source":"features_df = pd.DataFrame(dicom_features)\nfeatures_df.columns = [\"image_id\", \"rows\", \"columns\", \"content_date\", \"photometric_interpretation\"]\nfeatures_df[\"image_id\"] = features_df[\"image_id\"].apply(lambda x: int(x))\ntrain_add_df = train_df.merge(features_df, on=\"image_id\")","metadata":{"execution":{"iopub.status.busy":"2025-03-29T20:42:44.43567Z","iopub.execute_input":"2025-03-29T20:42:44.436371Z","iopub.status.idle":"2025-03-29T20:42:44.764997Z","shell.execute_reply.started":"2025-03-29T20:42:44.436323Z","shell.execute_reply":"2025-03-29T20:42:44.763692Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_add_df.head()","metadata":{"execution":{"iopub.status.busy":"2025-03-29T20:42:44.767934Z","iopub.execute_input":"2025-03-29T20:42:44.768373Z","iopub.status.idle":"2025-03-29T20:42:44.802958Z","shell.execute_reply.started":"2025-03-29T20:42:44.768335Z","shell.execute_reply":"2025-03-29T20:42:44.80144Z"},"_kg_hide-input":true,"trusted":true},"outputs":[],"execution_count":null}]}