{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA Mammo - Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"%%capture\n# Source: https://www.kaggle.com/code/remekkinas/fast-dicom-processing-1-6-2x-faster?scriptVersionId=113360473\n\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n   !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\n\nimport collections\nimport scipy\nimport pathlib\nimport pydicom\nimport functools\nimport itertools\nimport input_pipeline as ip","metadata":{"execution":{"iopub.status.busy":"2023-01-05T16:17:10.518905Z","iopub.execute_input":"2023-01-05T16:17:10.519349Z","iopub.status.idle":"2023-01-05T16:17:13.079516Z","shell.execute_reply.started":"2023-01-05T16:17:10.519315Z","shell.execute_reply":"2023-01-05T16:17:13.078333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_theme()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T16:17:13.081932Z","iopub.execute_input":"2023-01-05T16:17:13.08265Z","iopub.status.idle":"2023-01-05T16:17:13.088889Z","shell.execute_reply.started":"2023-01-05T16:17:13.082609Z","shell.execute_reply":"2023-01-05T16:17:13.087973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/input/rsna-breast-cancer-detection","metadata":{"execution":{"iopub.status.busy":"2023-01-05T16:17:13.09025Z","iopub.execute_input":"2023-01-05T16:17:13.090812Z","iopub.status.idle":"2023-01-05T16:17:14.21201Z","shell.execute_reply.started":"2023-01-05T16:17:13.090767Z","shell.execute_reply":"2023-01-05T16:17:14.210628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = pathlib.Path('/kaggle/input/rsna-breast-cancer-detection')","metadata":{"execution":{"iopub.status.busy":"2023-01-05T16:17:14.214836Z","iopub.execute_input":"2023-01-05T16:17:14.215349Z","iopub.status.idle":"2023-01-05T16:17:14.221325Z","shell.execute_reply.started":"2023-01-05T16:17:14.215308Z","shell.execute_reply":"2023-01-05T16:17:14.220363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = INPUT_DIR / 'train.csv'\ntrain_df = pd.read_csv(train_csv_path)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T16:17:14.223031Z","iopub.execute_input":"2023-01-05T16:17:14.223611Z","iopub.status.idle":"2023-01-05T16:17:14.343587Z","shell.execute_reply.started":"2023-01-05T16:17:14.22357Z","shell.execute_reply":"2023-01-05T16:17:14.34211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of patients.\ntrain_df.patient_id.nunique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Statistics about the number of images per patient.\ntrain_df.groupby('patient_id').image_id.count().describe(percentiles=[0.5, 0.95, 0.99])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot histogram of number of scans per patient.\ntrain_df.groupby('patient_id').image_id.count().hist(bins=11)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['view'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"views = train_df['view'].value_counts().index\nsns.countplot(x='view', data=train_df, order=views)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The min/max number of L and R breast images per patient.\ntrain_df.groupby(['patient_id', 'laterality']).image_id.agg([len]).agg([min, max])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lateralities = train_df['laterality'].value_counts().index\nsns.countplot(x='laterality', data=train_df, order=lateralities)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Accross the train dataset, patients each has 4 images or more. The majority of patients have the standard 4 images: 2 left (1 MLO and 1 CC) and 2 right (1 MLO and 1 CC).\n\nFor more information read the following article [radiopaedia | mammography views](https://radiopaedia.org/articles/mammography-views).","metadata":{}},{"cell_type":"code","source":"cancer_categories = train_df['cancer'].value_counts().index\nsns.countplot(x='cancer', data=train_df, order=cancer_categories)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_with_cancer = train_df[train_df.cancer == 1].size\nnum_without_cancer = train_df[train_df.cancer == 0].size\n\nnum_without_cancer / num_with_cancer","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is an imbalance between the number of cancerous images and healthy ones: ~46 more healthy images. This should be taken into account during training.","metadata":{}},{"cell_type":"code","source":"# density: A rating for how dense the breast tissue is, with A being the least dense\n# and D being the most dense. Extremely dense tissue can make diagnosis more difficult.\n\ndensity_categories = sorted(train_df['density'].value_counts().index)\nsns.countplot(x='density', data=train_df, order=density_categories)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"healthy_df = train_df[train_df.cancer == 0]\nhealthy_df.groupby('patient_id').age.agg([max]).hist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"healthy_df.groupby('patient_id').age.agg([max]).agg([np.mean])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_df = train_df[train_df.cancer == 1]\ncancer_df.groupby('patient_id').age.agg([max]).hist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_df.groupby('patient_id').age.agg([max]).agg([np.mean])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scipy.stats.spearmanr(\n    train_df.groupby('patient_id').age.agg([max]).to_numpy(),\n    train_df.groupby('patient_id').cancer.agg([max]).to_numpy(),\n    nan_policy='omit',\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image Analysis","metadata":{}},{"cell_type":"code","source":"!ls /kaggle/input/rsna-breast-cancer-detection/train_images/10006","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO: Annotate.\ndef scrape_dcm_headers(root: pathlib.Path):\n  readfn = functools.partial(\n      pydicom.dcmread, \n      stop_before_pixels=True)\n  return (readfn(p) for p in root.rglob('*.dcm'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO: Run in parallel.\n# WARNING: VERY SLOW\ndcms = list(scrape_dcm_headers(INPUT_DIR))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resolutions = ((d.Rows, d.Columns) for d in dcms)\nheights, widths = zip(*resolutions)\n\nfreqs = collections.Counter(zip(heights, widths))\nweights = [freqs[(h, w)] for h, w in zip(heights, widths)]\n\nsns.scatterplot(\n    x=widths, y=heights, \n    size=weights, # The corresponding weight of each point.\n    sizes=(80, 500), # Min/Max pt size to use when plotting.\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO show 2 Left and 2 Right and add captions.\n\nfor _, df in itertools.islice(train_df.groupby('patient_id'), 20):\n  fig = plt.figure(figsize=(20, 5))\n  \n  num_cols = len(df)\n  for i, (_, r) in enumerate(df.iterrows()):\n    p = INPUT_DIR / 'train_images' / f'{r.patient_id}' / f'{r.image_id}.dcm'\n    print(p)\n    plt.subplot(1, num_cols, i + 1)\n    plt.imshow(ip.loadbreastimg(p), cmap='gray')\n  plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T16:17:15.552066Z","iopub.execute_input":"2023-01-05T16:17:15.552519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}