{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Some LB probing results to share\nIn this notebook, I tested some basic assumptions about test dataset.\nI demonstrated that following assumptions are all TRUE.\n* There are no new site ID in test dataset.\n* Patient IDs in train and test sets do not overlap\n* Image IDs in train and test sets do not overlap\n* There are no new laterality values in test dataset.\n* There are new machine IDs in test dataset. (This is already raised by patriot in [here](https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/369362))\n* There are no new view values in test dataset.\n* No of images/patient are all >= 4\n* Site ID is always the same for each patient.\n* Age is always the same for each patient.\n* There are no overlap of machine IDs between two sites in test dataset.\n* Some patients underwent mammography with multiple machines.\n* No. of images in site ID 1 > No. of images in site ID 2. (by @yujiariyasu)\n* All patients have CC and MLO images for both sides\n* More than 40% of images are from machine ID 49 (43% for train dataset) (by @kaggleqrdl)\n* Mean age of patients is between 56-61 (58.6 for train set), and patients in site 1 are younger than those in site 2. (by @kaggleqrdl)\n* 1-2% of patients use implants (1.4% for train set).(by @kaggleqrdl)\n* Age column contains nan, while others do not.\n\nI am happy if anyone correct me if I am wrong.\nI am also very happy if anyone share us other assumtions/hypothesis about test dataset.","metadata":{}},{"cell_type":"code","source":"!cp /kaggle/input/nvjpeg2k/nvjpeg2k.so ./\n!pip install /kaggle/input/dicomsdl-offline-installer/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!pip install -q --disable-pip-version-check /kaggle/input/rsna-2022-whl/pylibjpeg-1.4.0-py3-none-any.whl\n!pip install -q --disable-pip-version-check /kaggle/input/rsna-2022-whl/python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:07.761929Z","iopub.execute_input":"2022-12-23T06:34:07.762312Z","iopub.status.idle":"2022-12-23T06:34:58.56386Z","shell.execute_reply.started":"2022-12-23T06:34:07.762264Z","shell.execute_reply":"2022-12-23T06:34:58.562684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport nvjpeg2k\nimport dicomsdl\n\nimport pydicom\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt","metadata":{"papermill":{"duration":2.803277,"end_time":"2022-11-30T00:42:15.921134","exception":false,"start_time":"2022-11-30T00:42:13.117857","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-12-23T06:35:33.347971Z","iopub.execute_input":"2022-12-23T06:35:33.348475Z","iopub.status.idle":"2022-12-23T06:35:33.356433Z","shell.execute_reply.started":"2022-12-23T06:35:33.348433Z","shell.execute_reply":"2022-12-23T06:35:33.355358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\nsub_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\")\n\nprint(\"train shape:\", train_df.shape)\nprint(\"test shape:\", test_df.shape)\nprint(\"sub_df shape:\", sub_df.shape)\ndisplay(train_df.head())\ndisplay(test_df.head())\ndisplay(sub_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:58.774563Z","iopub.execute_input":"2022-12-23T06:34:58.774906Z","iopub.status.idle":"2022-12-23T06:34:58.876194Z","shell.execute_reply.started":"2022-12-23T06:34:58.774871Z","shell.execute_reply":"2022-12-23T06:34:58.875261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_num_unique(train_df, test_df, col):\n    all_df = pd.concat([train_df, test_df])\n    num_unique_train = len(train_df[col].unique())\n    num_unique_test = len(test_df[col].unique())\n    num_unique_all = len(all_df[col].unique())\n    return num_unique_train, num_unique_test, num_unique_all\n\ndef add_count(df, col):\n    if type(col) == str:\n        aggs = df.groupby(col, as_index=True)[col].count().rename(col + \"_count\")\n    else:\n        aggs = (\n            df.groupby(col, as_index=False)[col[0]]\n            .count()\n            .rename(\"_\".join(col) + \"_count\")\n        )\n    df = df.merge(aggs, on=col, how=\"inner\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:58.878865Z","iopub.execute_input":"2022-12-23T06:34:58.879267Z","iopub.status.idle":"2022-12-23T06:34:58.887967Z","shell.execute_reply.started":"2022-12-23T06:34:58.879228Z","shell.execute_reply":"2022-12-23T06:34:58.885263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hypotheses = []","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:58.889898Z","iopub.execute_input":"2022-12-23T06:34:58.890683Z","iopub.status.idle":"2022-12-23T06:34:58.89812Z","shell.execute_reply.started":"2022-12-23T06:34:58.890633Z","shell.execute_reply":"2022-12-23T06:34:58.897159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There are no new site ID in test dataset.","metadata":{}},{"cell_type":"code","source":"num_unique_train, num_unique_test, num_unique_all = get_num_unique(train_df, test_df, 'site_id')\nprint(f'num_unique_train: {num_unique_train}')\nprint(f'num_unique_test: {num_unique_test}')\nprint(f'num_unique_all: {num_unique_all}')\nhypothesis = (num_unique_all == 2)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:58.899589Z","iopub.execute_input":"2022-12-23T06:34:58.901537Z","iopub.status.idle":"2022-12-23T06:34:58.92187Z","shell.execute_reply.started":"2022-12-23T06:34:58.901506Z","shell.execute_reply":"2022-12-23T06:34:58.921004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Patient IDs in train and test sets do not overlap","metadata":{}},{"cell_type":"code","source":"num_unique_train, num_unique_test, num_unique_all = get_num_unique(train_df, test_df, 'patient_id')\nprint(f'num_unique_train: {num_unique_train}')\nprint(f'num_unique_test: {num_unique_test}')\nprint(f'num_unique_all: {num_unique_all}')\nhypothesis = (num_unique_train + num_unique_test == num_unique_all)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:58.923082Z","iopub.execute_input":"2022-12-23T06:34:58.92452Z","iopub.status.idle":"2022-12-23T06:34:58.941862Z","shell.execute_reply.started":"2022-12-23T06:34:58.924483Z","shell.execute_reply":"2022-12-23T06:34:58.940957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image IDs in train and test sets do not overlap","metadata":{}},{"cell_type":"code","source":"num_unique_train, num_unique_test, num_unique_all = get_num_unique(train_df, test_df, 'image_id')\nprint(f'num_unique_train: {num_unique_train}')\nprint(f'num_unique_test: {num_unique_test}')\nprint(f'num_unique_all: {num_unique_all}')\nhypothesis = (num_unique_train + num_unique_test == num_unique_all)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:58.94312Z","iopub.execute_input":"2022-12-23T06:34:58.943555Z","iopub.status.idle":"2022-12-23T06:34:58.964803Z","shell.execute_reply.started":"2022-12-23T06:34:58.943519Z","shell.execute_reply":"2022-12-23T06:34:58.963948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There are no new laterality values in test dataset.","metadata":{}},{"cell_type":"code","source":"num_unique_train, num_unique_test, num_unique_all = get_num_unique(train_df, test_df, 'laterality')\nprint(f'num_unique_train: {num_unique_train}')\nprint(f'num_unique_test: {num_unique_test}')\nprint(f'num_unique_all: {num_unique_all}')\nhypothesis = (num_unique_test == 2)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:58.966195Z","iopub.execute_input":"2022-12-23T06:34:58.966899Z","iopub.status.idle":"2022-12-23T06:34:58.99008Z","shell.execute_reply.started":"2022-12-23T06:34:58.966843Z","shell.execute_reply":"2022-12-23T06:34:58.989059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There are new machine IDs in test dataset.","metadata":{}},{"cell_type":"code","source":"num_unique_train, num_unique_test, num_unique_all = get_num_unique(train_df, test_df, 'machine_id')\nprint(f'num_unique_train: {num_unique_train}')\nprint(f'num_unique_test: {num_unique_test}')\nprint(f'num_unique_all: {num_unique_all}')\nhypothesis = (num_unique_train != num_unique_all)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:58.994179Z","iopub.execute_input":"2022-12-23T06:34:58.994461Z","iopub.status.idle":"2022-12-23T06:34:59.013849Z","shell.execute_reply.started":"2022-12-23T06:34:58.994435Z","shell.execute_reply":"2022-12-23T06:34:59.012985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There are no new view values in test dataset.","metadata":{}},{"cell_type":"code","source":"num_unique_train, num_unique_test, num_unique_all = get_num_unique(train_df, test_df, 'view')\nprint(f'num_unique_train: {num_unique_train}')\nprint(f'num_unique_test: {num_unique_test}')\nprint(f'num_unique_all: {num_unique_all}')\nhypothesis = (num_unique_all == 6)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.015277Z","iopub.execute_input":"2022-12-23T06:34:59.015667Z","iopub.status.idle":"2022-12-23T06:34:59.038719Z","shell.execute_reply.started":"2022-12-23T06:34:59.015616Z","shell.execute_reply":"2022-12-23T06:34:59.037667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# No of images/patient are all >= 4.","metadata":{}},{"cell_type":"code","source":"temp_df = add_count(test_df, 'patient_id').drop_duplicates('patient_id')\ndisplay(temp_df.head())\nhypothesis = (temp_df.patient_id_count.min() >= 4)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.040217Z","iopub.execute_input":"2022-12-23T06:34:59.04058Z","iopub.status.idle":"2022-12-23T06:34:59.066886Z","shell.execute_reply.started":"2022-12-23T06:34:59.040546Z","shell.execute_reply":"2022-12-23T06:34:59.065819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Site ID is always the same for each patient.","metadata":{}},{"cell_type":"code","source":"len1 = len(test_df.drop_duplicates(['patient_id']))\nlen2 = len(test_df.drop_duplicates(['patient_id','site_id']))\nprint(f'len1: {len1}')\nprint(f'len2: {len2}')\nhypothesis = (len1  == len2)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.068372Z","iopub.execute_input":"2022-12-23T06:34:59.068968Z","iopub.status.idle":"2022-12-23T06:34:59.077885Z","shell.execute_reply.started":"2022-12-23T06:34:59.068934Z","shell.execute_reply":"2022-12-23T06:34:59.076927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Age is always the same for each patient.","metadata":{}},{"cell_type":"code","source":"len1 = len(test_df.drop_duplicates(['patient_id']))\nlen2 = len(test_df.drop_duplicates(['patient_id','age']))\nprint(f'len1: {len1}')\nprint(f'len2: {len2}')\nhypothesis = (len1  == len2)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.079392Z","iopub.execute_input":"2022-12-23T06:34:59.079739Z","iopub.status.idle":"2022-12-23T06:34:59.09186Z","shell.execute_reply.started":"2022-12-23T06:34:59.079703Z","shell.execute_reply":"2022-12-23T06:34:59.090926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There are no overlap of machine IDs between two sites in test dataset.","metadata":{}},{"cell_type":"code","source":"len1 = len(test_df.drop_duplicates(['machine_id']))\nlen2 = len(test_df.drop_duplicates(['site_id','machine_id']))\nprint(f'len1: {len1}')\nprint(f'len2: {len2}')\nhypothesis = (len1  == len2)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.094267Z","iopub.execute_input":"2022-12-23T06:34:59.094705Z","iopub.status.idle":"2022-12-23T06:34:59.107713Z","shell.execute_reply.started":"2022-12-23T06:34:59.094671Z","shell.execute_reply":"2022-12-23T06:34:59.106601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Some patients underwent mammography with multiple machines.","metadata":{}},{"cell_type":"code","source":"len1 = len(test_df.drop_duplicates(['patient_id']))\nlen2 = len(test_df.drop_duplicates(['patient_id','machine_id']))\nprint(f'len1: {len1}')\nprint(f'len2: {len2}')\nhypothesis = (len1  != len2)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.109416Z","iopub.execute_input":"2022-12-23T06:34:59.11027Z","iopub.status.idle":"2022-12-23T06:34:59.125355Z","shell.execute_reply.started":"2022-12-23T06:34:59.110233Z","shell.execute_reply":"2022-12-23T06:34:59.124396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This hypothesis is true for train set","metadata":{}},{"cell_type":"code","source":"len1 = len(train_df.drop_duplicates(['patient_id']))\nlen2 = len(train_df.drop_duplicates(['patient_id','machine_id']))\nprint(f'len1: {len1}')\nprint(f'len2: {len2}')\nhypothesis = (len1  != len2)\nprint(f'hyposthesis: {hypothesis}')\ndisplay(train_df[train_df.patient_id == 22637])","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.130985Z","iopub.execute_input":"2022-12-23T06:34:59.131612Z","iopub.status.idle":"2022-12-23T06:34:59.175703Z","shell.execute_reply.started":"2022-12-23T06:34:59.131576Z","shell.execute_reply":"2022-12-23T06:34:59.174775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# No. of images in site ID 1 >  No. of images in site ID 2\nSuggested by @yujiariyasu","metadata":{}},{"cell_type":"code","source":"mean_site_id_train = train_df.site_id.mean()\nmean_site_id_test = test_df.site_id.mean()\nprint(f'mean site ID train: {mean_site_id_train}')\nprint(f'mean site ID test: {mean_site_id_test}')\nhypothesis = (mean_site_id_test < 1.5)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.179813Z","iopub.execute_input":"2022-12-23T06:34:59.18216Z","iopub.status.idle":"2022-12-23T06:34:59.194426Z","shell.execute_reply.started":"2022-12-23T06:34:59.182121Z","shell.execute_reply":"2022-12-23T06:34:59.193193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# All patients have CC and MLO images for both sides","metadata":{}},{"cell_type":"code","source":"len1 = len(test_df.patient_id.unique())\nlen2 = len(test_df[(test_df.laterality == 'L')&(test_df.view == 'CC')].patient_id.unique())\nlen3 = len(test_df[(test_df.laterality == 'L')&(test_df.view == 'MLO')].patient_id.unique())\nlen4 = len(test_df[(test_df.laterality == 'R')&(test_df.view == 'CC')].patient_id.unique())\nlen5 = len(test_df[(test_df.laterality == 'R')&(test_df.view == 'MLO')].patient_id.unique())\nprint(f'len1: {len1}')\nprint(f'len2: {len2}')\nprint(f'len3: {len3}')\nprint(f'len4: {len4}')\nprint(f'len5: {len5}')\nhypothesis = len1 == len2 == len3 == len4 == len5\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.199672Z","iopub.execute_input":"2022-12-23T06:34:59.202224Z","iopub.status.idle":"2022-12-23T06:34:59.221471Z","shell.execute_reply.started":"2022-12-23T06:34:59.202183Z","shell.execute_reply":"2022-12-23T06:34:59.220161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# More than 40% of images are from machine ID 49 (43% for train dataset)\nSuggested by @kaggleqrdl","metadata":{}},{"cell_type":"code","source":"test_machine_49_count = len(test_df.query(\"machine_id == 49\"))\ntest_len = len(test_df)\ntest_machine_49_ratio = test_machine_49_count/test_len\nprint(f'test_machine_49_count: {test_machine_49_count}')\nprint(f'test_len: {test_len}')\nprint(f'test_machine_49_ratio: {test_machine_49_ratio}')\nhypothesis = test_machine_49_ratio > 0.40\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.225531Z","iopub.execute_input":"2022-12-23T06:34:59.228212Z","iopub.status.idle":"2022-12-23T06:34:59.242819Z","shell.execute_reply.started":"2022-12-23T06:34:59.228176Z","shell.execute_reply":"2022-12-23T06:34:59.241777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_machine_49_count = len(train_df.query(\"machine_id == 49\"))\ntrain_len = len(train_df)\ntrain_machine_49_ratio = train_machine_49_count/train_len\nprint(f'train_machine_49_count: {train_machine_49_count}')\nprint(f'train_len: {train_len}')\nprint(f'train_machine_49_ratio: {train_machine_49_ratio}')","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.247679Z","iopub.execute_input":"2022-12-23T06:34:59.249532Z","iopub.status.idle":"2022-12-23T06:34:59.260307Z","shell.execute_reply.started":"2022-12-23T06:34:59.249494Z","shell.execute_reply":"2022-12-23T06:34:59.259218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mean age of patients is between 56-61 (58.6 for train set), and patients in site 1 are younger than those in site 2.\nSuggested by @kaggleqrdl","metadata":{}},{"cell_type":"code","source":"mean_age_train = train_df.drop_duplicates('patient_id').age.mean()\nmean_age_site1_train = train_df[train_df.site_id == 1].drop_duplicates('patient_id').age.mean()\nmean_age_site2_train = train_df[train_df.site_id == 2].drop_duplicates('patient_id').age.mean()\nmean_age_test = test_df.drop_duplicates('patient_id').age.mean()\nmean_age_site1_test = test_df[test_df.site_id == 1].drop_duplicates('patient_id').age.mean()\nmean_age_site2_test = test_df[test_df.site_id == 2].drop_duplicates('patient_id').age.mean()\nprint(f'mean_age_train: {mean_age_train}')\nprint(f'mean_age_site1_train: {mean_age_site1_train}')\nprint(f'mean_age_site2_train: {mean_age_site2_train}')\nprint(f'mean_age_test: {mean_age_test}')\nprint(f'mean_age_site1_test: {mean_age_site1_test}')\nprint(f'mean_age_site2_test: {mean_age_site2_test}')\nhypothesis = (mean_age_test > 56)&(61 > mean_age_test)&(mean_age_site1_test < mean_age_site2_test)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.262081Z","iopub.execute_input":"2022-12-23T06:34:59.262765Z","iopub.status.idle":"2022-12-23T06:34:59.292097Z","shell.execute_reply.started":"2022-12-23T06:34:59.262731Z","shell.execute_reply":"2022-12-23T06:34:59.291078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1-2% of patients use implants (1.4% for train set).\nSuggested by @kaggleqrdl","metadata":{}},{"cell_type":"code","source":"mean_implant_train = train_df.drop_duplicates(['patient_id','laterality']).implant.mean()\nmean_implant_test = test_df.drop_duplicates(['patient_id','laterality']).implant.mean()\nprint(f'mean_implant_train: {mean_implant_train}')\nprint(f'mean_implant_test: {mean_implant_test}')\nhypothesis = (mean_implant_test > 0.01)&(0.02 > mean_implant_test)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.293478Z","iopub.execute_input":"2022-12-23T06:34:59.293896Z","iopub.status.idle":"2022-12-23T06:34:59.310947Z","shell.execute_reply.started":"2022-12-23T06:34:59.293861Z","shell.execute_reply":"2022-12-23T06:34:59.309926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Age column contains nan, while others do not.","metadata":{}},{"cell_type":"code","source":"isna_age_train = train_df.age.isnull().sum()\nisna_age_test = test_df.age.isnull().sum()\nother_cols = ['site_id', 'patient_id', 'image_id', 'laterality', 'view', \n       'implant', 'machine_id']\nisna_other_train = train_df[other_cols].isnull().sum().sum()\nisna_other_test = test_df[other_cols].isnull().sum().sum()\nprint(train_df.isnull().sum(axis=0))\nprint(test_df.isnull().sum(axis=0))\nprint(f'isna_age_train: {isna_age_train}')\nprint(f'isna_age_test: {isna_age_test}')\nprint(f'isna_other_train: {isna_other_train}')\nprint(f'isna_other_test: {isna_other_test}')\nhypothesis = (isna_age_test > 0)&(isna_other_test==0)\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.312717Z","iopub.execute_input":"2022-12-23T06:34:59.313077Z","iopub.status.idle":"2022-12-23T06:34:59.339749Z","shell.execute_reply.started":"2022-12-23T06:34:59.313042Z","shell.execute_reply":"2022-12-23T06:34:59.338756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# No images are corrupted like train_images/822/1942326353.dcm","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/code/snaker/easy-load-the-image-with-nvjpeg2000/notebook\nj2k_decoder = nvjpeg2k.Decoder()\ndef read_xray_fast(path, j2k_decoder=j2k_decoder, force_slow=False, use_dicomsdl = True, voi_lut = True, fix_monochrome = True):\n    dcm = pydicom.dcmread(path)\n    if not force_slow and dcm.file_meta.TransferSyntaxUID == '1.2.840.10008.1.2.4.90':\n        offset = dcm.PixelData.find(b'\\x00\\x00\\x00\\x0C')\n        jpeg_stream = bytearray(dcm.PixelData[offset:]) \n        data = j2k_decoder.decode(jpeg_stream) \n        PhotometricInterpretation = dcm.PhotometricInterpretation        \n    else:\n        if use_dicomsdl:\n            dicom = dicomsdl.open_file(path)\n            data = dicom.pixelData()\n            PhotometricInterpretation = dicom.getPixelDataInfo()['PhotometricInterpretation']\n        else:\n            data = dcm.pixel_array\n            PhotometricInterpretation = dcm.PhotometricInterpretation\n    if voi_lut:\n        data = pydicom.pixel_data_handlers.util.apply_voi_lut(data, dcm)\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data\n\ndef joblib_parallel(func, args, n_processes=2):\n    res = Parallel(n_jobs=n_processes, prefer=\"threads\")(delayed(func)(x) for x in tqdm(args))\n    return list(res)\n\ndef check_dcm(dcm_path, threshold = 20):\n    img = read_xray_fast(dcm_path, force_slow=False, use_dicomsdl = True, voi_lut = False, fix_monochrome = True)\n    if len(np.unique(img)) < threshold:\n        return dcm_path\n    else:\n        return None","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.341426Z","iopub.execute_input":"2022-12-23T06:34:59.341783Z","iopub.status.idle":"2022-12-23T06:34:59.552104Z","shell.execute_reply.started":"2022-12-23T06:34:59.341748Z","shell.execute_reply":"2022-12-23T06:34:59.550785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_id = 822\ndisplay(train_df[train_df.patient_id == patient_id])\nimage_id = 1942326353\ndcm_dir = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\ndcm_path = f\"{dcm_dir}{patient_id}/{image_id}.dcm\"\nimg = read_xray_fast(dcm_path, force_slow=False, use_dicomsdl = True, voi_lut = False, fix_monochrome = True)\nplt.imshow(img)\nprint(len(np.unique(img)))","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:35:42.921831Z","iopub.execute_input":"2022-12-23T06:35:42.922203Z","iopub.status.idle":"2022-12-23T06:35:45.000739Z","shell.execute_reply.started":"2022-12-23T06:35:42.92217Z","shell.execute_reply":"2022-12-23T06:35:44.999629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndcm_dir = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\ndcm_paths = [f\"{dcm_dir}{p}/{i}.dcm\" for p, i in zip(train_df[\"patient_id\"], train_df[\"image_id\"])]\nng_dcms_train = joblib_parallel(check_dcm, dcm_paths[53000:53100], n_processes=-1)\nng_dcms_train = np.array(ng_dcms_train)\nng_dcms_train = ng_dcms_train[ng_dcms_train!= None]","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:37:09.182108Z","iopub.execute_input":"2022-12-23T06:37:09.182511Z","iopub.status.idle":"2022-12-23T06:37:50.734552Z","shell.execute_reply.started":"2022-12-23T06:37:09.182477Z","shell.execute_reply":"2022-12-23T06:37:50.733554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dcm_dir = \"/kaggle/input/rsna-breast-cancer-detection/test_images/\"\ndcm_paths = [f\"{dcm_dir}{p}/{i}.dcm\" for p, i in zip(test_df[\"patient_id\"], test_df[\"image_id\"])]\nng_dcms_test = joblib_parallel(check_dcm, dcm_paths, n_processes=-1)\nng_dcms_test = np.array(ng_dcms_test)\nng_dcms_test = ng_dcms_test[ng_dcms_test!= None]","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:37:51.031877Z","iopub.execute_input":"2022-12-23T06:37:51.032231Z","iopub.status.idle":"2022-12-23T06:37:51.903986Z","shell.execute_reply.started":"2022-12-23T06:37:51.032201Z","shell.execute_reply":"2022-12-23T06:37:51.903027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'ng_dcms_train: {ng_dcms_train}')\nprint(f'ng_dcms_test: {ng_dcms_test}')\nhypothesis = len(ng_dcms_test) == 0\nprint(f'hyposthesis: {hypothesis}')\nhypotheses.append(hypothesis)","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:37:53.795111Z","iopub.execute_input":"2022-12-23T06:37:53.795499Z","iopub.status.idle":"2022-12-23T06:37:53.801986Z","shell.execute_reply.started":"2022-12-23T06:37:53.795468Z","shell.execute_reply":"2022-12-23T06:37:53.801004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# If the submission of this notebook is successful, all hypotheses are TRUE","metadata":{}},{"cell_type":"code","source":"print(hypotheses)\nprint(all(hypotheses))","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.905241Z","iopub.status.idle":"2022-12-23T06:34:59.906019Z","shell.execute_reply.started":"2022-12-23T06:34:59.905747Z","shell.execute_reply":"2022-12-23T06:34:59.905776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if all(hypotheses):\n    submission = pd.DataFrame(data={'prediction_id': test_df['prediction_id'].unique(), 'cancer': np.random.random(len(test_df['prediction_id'].unique()))})\nelse:\n    submission = pd.DataFrame(data={'prediction_id': test_df['prediction_id'], 'cancer': np.random.random(len(test_df))})\nsubmission.to_csv('submission.csv', index=False)\ndisplay(submission.head())","metadata":{"execution":{"iopub.status.busy":"2022-12-23T06:34:59.907372Z","iopub.status.idle":"2022-12-23T06:34:59.908063Z","shell.execute_reply.started":"2022-12-23T06:34:59.907808Z","shell.execute_reply":"2022-12-23T06:34:59.907832Z"},"trusted":true},"execution_count":null,"outputs":[]}]}