{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install --pre pycaret","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:02.627564Z","iopub.execute_input":"2022-12-21T03:50:02.628147Z","iopub.status.idle":"2022-12-21T03:50:02.633592Z","shell.execute_reply.started":"2022-12-21T03:50:02.62807Z","shell.execute_reply":"2022-12-21T03:50:02.632476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we do the downloading of required packages","metadata":{}},{"cell_type":"code","source":"# !pip download timm fastai -d ./pip_packages/","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:02.636176Z","iopub.execute_input":"2022-12-21T03:50:02.638867Z","iopub.status.idle":"2022-12-21T03:50:02.646412Z","shell.execute_reply.started":"2022-12-21T03:50:02.638721Z","shell.execute_reply":"2022-12-21T03:50:02.645064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# from zipfile import ZipFile\n\n# dirName = \"./pip_packages\"\n# zipName = \"packages.zip\"\n\n# # Create a ZipFile Object\n# with ZipFile(zipName, 'w') as zipObj:\n#     # Iterate over all the files in directory\n#     for folderName, subfolders, filenames in os.walk(dirName):\n#         for filename in filenames:\n#             if (filename != zipName):\n#                 # create complete filepath of file in directory\n#                 filePath = os.path.join(folderName, filename)\n#                 # Add file to zip\n#                 zipObj.write(filePath)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:02.64827Z","iopub.execute_input":"2022-12-21T03:50:02.648955Z","iopub.status.idle":"2022-12-21T03:50:02.656232Z","shell.execute_reply.started":"2022-12-21T03:50:02.648916Z","shell.execute_reply":"2022-12-21T03:50:02.65503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install timm --no-index --find-links=file:///kaggle/input/digitaldeusrsna/pip_packages/","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:02.657905Z","iopub.execute_input":"2022-12-21T03:50:02.65867Z","iopub.status.idle":"2022-12-21T03:50:17.984227Z","shell.execute_reply.started":"2022-12-21T03:50:02.658632Z","shell.execute_reply":"2022-12-21T03:50:17.98292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Constants\nTRAIN_KAGGLE_CSV = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\nTRAIN_IMAGES = '/kaggle/input/rsna-breast-cancer-detection/train_images'\nTEST_IMAGES = '/kaggle/input/rsna-breast-cancer-detection/test_images'\nIN_ROOT_DIR=  '/kaggle/input/rsna-breast-cancer-detection'\nTRAIN_PREPROCESSED_ROOT= '/kaggle/input/rsna-mammography-images-as-pngs'\nOUT_ROOT_DIR=\"/kaggle/working\"\nMETAS_FILE_PATH=f'{OUT_ROOT_DIR}/metas_pydicom'\nTRAIN_KAGGLE_CSV = f'{IN_ROOT_DIR}/train.csv'\nTRAIN_SAVE_DIR = f'{OUT_ROOT_DIR}/train.pickle'\nTARGET_IMG_SIZE = 256\nNUMBER_OF_CANCER_AUG_PER = 40 # How many variations to use of generated cancer images, max 40\nCANCER_GEN_DIR=f'/kaggle/input/digitaldeusrsna/cancer/cancer/images_as_pngs_cv2_dicomsdl_{TARGET_IMG_SIZE}/train_images_processed_cv2_dicomsdl_{TARGET_IMG_SIZE}'\nTRAIN_CSV = TRAIN_KAGGLE_CSV\nTEST_PNG_DIR=f'/kaggle/working/train_images_processed_cv2_dicomsdl_{TARGET_IMG_SIZE}'\nTRAIN_PREPROCESSED_PATH = f\"/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_dicomsdl_{TARGET_IMG_SIZE}/train_images_processed_cv2_dicomsdl_{TARGET_IMG_SIZE}\"\nEPOCHS = 100","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:17.991538Z","iopub.execute_input":"2022-12-21T03:50:17.994425Z","iopub.status.idle":"2022-12-21T03:50:18.0046Z","shell.execute_reply.started":"2022-12-21T03:50:17.994378Z","shell.execute_reply":"2022-12-21T03:50:18.003653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Thanks to https://www.kaggle.com/code/jonathangrant/eda-fastai-timm-approach\n# I couldn't figure this part out\n# # Start -------------------------\n# !pip install pylibjpeg gdcm python-gdcm pylibjpeg-libjpeg pylibjpeg-openjpeg pydicom  --no-index --find-links=file:///kaggle/input/digitaldeusrsna/pip_packages/\n# import pkg_resources\n# list(pkg_resources.WorkingSet(None).iter_entry_points(\"pylibjpeg.pixel_data_decoders\"))\n\n# # Monkey patch in pylibjpeg\n# import pylibjpeg.utils\n# pylibjpeg.utils.iter_entry_points = pkg_resources.WorkingSet(None).iter_entry_points\n# # Prove monkey patch worked\n# pylibjpeg.utils.get_pixel_data_decoders()\n\n# # Proof monkey patch worked all the way\n# import pydicom.pixel_data_handlers.pylibjpeg_handler\n# pydicom.pixel_data_handlers.pylibjpeg_handler._DECODERS\n# # End -------------------------\n\n# !pip install fastai accelerate gdcm pylibjpeg[all] --upgrade\nimport os\nimport pickle\nfrom tqdm import tqdm\nimport glob\nimport warnings\nimport pydicom\n\nfrom fastai.distributed import *\nfrom fastai.basics import *\nfrom fastai.callback.all import *\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\n\n# from accelerate import notebook_launcher\n# from accelerate.utils import write_basic_config\n\nimport pydicom\n\nimport pandas as pd\nimport numpy as np\nimport os\n\nimport torch\n\nif torch.cuda.is_available():\n    torch.cuda.set_device(0)\n\n# # Run once per session\n# if not os.path.exists('/root/.cache/huggingface/accelerate/default_config.yaml'):\n#     write_basic_config()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.009514Z","iopub.execute_input":"2022-12-21T03:50:18.012289Z","iopub.status.idle":"2022-12-21T03:50:18.027718Z","shell.execute_reply.started":"2022-12-21T03:50:18.012251Z","shell.execute_reply":"2022-12-21T03:50:18.026665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(f'{IN_ROOT_DIR}/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.03256Z","iopub.execute_input":"2022-12-21T03:50:18.035296Z","iopub.status.idle":"2022-12-21T03:50:18.161902Z","shell.execute_reply.started":"2022-12-21T03:50:18.035258Z","shell.execute_reply":"2022-12-21T03:50:18.16085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Looks like all the DICOM data we would want has been provided\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.167041Z","iopub.execute_input":"2022-12-21T03:50:18.169403Z","iopub.status.idle":"2022-12-21T03:50:18.200638Z","shell.execute_reply.started":"2022-12-21T03:50:18.169362Z","shell.execute_reply":"2022-12-21T03:50:18.199715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analyze the data","metadata":{}},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.204741Z","iopub.execute_input":"2022-12-21T03:50:18.206993Z","iopub.status.idle":"2022-12-21T03:50:18.241223Z","shell.execute_reply.started":"2022-12-21T03:50:18.206956Z","shell.execute_reply":"2022-12-21T03:50:18.240312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA - Exploratory Data Analysis\n\nWe have some missing fields at least in age, BIRADS, implant and density\n\nEverything else looks good for processing numerically/categorically","metadata":{}},{"cell_type":"code","source":"# Get a count of the missing data\nmissing_data = pd.DataFrame({'total_missing': df.isnull().sum(), 'perc_missing': (df.isnull().sum()/len(df))*100})\nmissing_data","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.245637Z","iopub.execute_input":"2022-12-21T03:50:18.24789Z","iopub.status.idle":"2022-12-21T03:50:18.287469Z","shell.execute_reply.started":"2022-12-21T03:50:18.247853Z","shell.execute_reply":"2022-12-21T03:50:18.286541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the stats\ndf.describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.291534Z","iopub.execute_input":"2022-12-21T03:50:18.293762Z","iopub.status.idle":"2022-12-21T03:50:18.362047Z","shell.execute_reply.started":"2022-12-21T03:50:18.293724Z","shell.execute_reply":"2022-12-21T03:50:18.361158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll of course drop the `image_id` and possibly `patient_id`\n\nWe'll check `machine_id` maybe it's useful as longs as there are not too many different machines","metadata":{}},{"cell_type":"markdown","source":"## Check numerical values","metadata":{}},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.366175Z","iopub.execute_input":"2022-12-21T03:50:18.36847Z","iopub.status.idle":"2022-12-21T03:50:18.378559Z","shell.execute_reply.started":"2022-12-21T03:50:18.368431Z","shell.execute_reply":"2022-12-21T03:50:18.377557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nnum_cols = ['age']\nplt.figure(figsize=(18,9))\ndf[num_cols].boxplot()\nplt.title(\"Numerical variables in RSNA dataset\", fontsize=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.383285Z","iopub.execute_input":"2022-12-21T03:50:18.385596Z","iopub.status.idle":"2022-12-21T03:50:18.704887Z","shell.execute_reply.started":"2022-12-21T03:50:18.385558Z","shell.execute_reply":"2022-12-21T03:50:18.703915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have a few age outliers, but we'll leave these in as they are within the range of a living being.","metadata":{}},{"cell_type":"markdown","source":"# Data Cleaning & Pre-processing","metadata":{}},{"cell_type":"markdown","source":"Now lets go through and clean the data, checking the distribution and doing something about our null values","metadata":{}},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.713682Z","iopub.execute_input":"2022-12-21T03:50:18.715989Z","iopub.status.idle":"2022-12-21T03:50:18.726659Z","shell.execute_reply.started":"2022-12-21T03:50:18.715949Z","shell.execute_reply":"2022-12-21T03:50:18.725471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_dist(col, ax):\n    df[col][df[col].notnull()].value_counts().plot(kind='bar', facecolor='y', ax=ax)\n    ax.set_xlabel('{}'.format(col), fontsize=20)\n    ax.set_title(\"{} on rsna\".format(col), fontsize= 18)\n    return ax\n\nf, ax = plt.subplots(4,3, figsize = (22,15))\nf.tight_layout(h_pad=9, w_pad=2)\ncols = ['site_id', 'laterality', 'view', 'age',\n       'cancer', 'biopsy', 'invasive', 'BIRADS', 'implant', 'density',\n       'machine_id', 'difficult_negative_case']\nk = 0\nfor i in range(4):\n    for j in range(3):\n        plot_dist(cols[k], ax[i][j])\n        k += 1\n__ = plt.suptitle(\"Total Distribution of features\", fontsize= 25)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:18.731206Z","iopub.execute_input":"2022-12-21T03:50:18.734022Z","iopub.status.idle":"2022-12-21T03:50:22.062885Z","shell.execute_reply.started":"2022-12-21T03:50:18.733984Z","shell.execute_reply":"2022-12-21T03:50:22.061913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Step-by-step features processing:\n\n['site_id', 'laterality', 'view', 'age',\n       'cancer', 'biopsy', 'invasive', 'BIRADS', 'implant', 'density',\n       'machine_id', 'difficult_negative_case']\n\n* **site_id**: Should be categorical\n* **laterality**: Should be categorical\n* **view**: Should be categorical\n* **cancer**: Should be categorical\n* **biopsy**: Should be categorical\n* **invasive**: Should be categorical\n* **BIRADS**: Should be categorical\n* **implant**: Should be categorical\n* **density**: Should be categorical\n* **machine_id**: Should be categorical\n* **difficult_negative_case**: Should be categorical\n\n**age** is the only numerical data, with a larger concentration of samples around 50\n\nAnother thing to note is that there distribution of cancer to non cancer is way off.  We may want to limit the training set to only include maybe a 1:1 distribution for cancerous vs non cancerous","metadata":{}},{"cell_type":"markdown","source":"Lets also get he distribution of cancer only","metadata":{}},{"cell_type":"code","source":"cancer_df = df[df.cancer == 1]\n\ndef plot_dist(col, ax):\n    cancer_df[col][cancer_df[col].notnull()].value_counts().plot(kind='bar', facecolor='y', ax=ax)\n    ax.set_xlabel('{}'.format(col), fontsize=20)\n    ax.set_title(\"{} on rsna\".format(col), fontsize= 18)\n    return ax\n\nf, ax = plt.subplots(4,3, figsize = (22,15))\nf.tight_layout(h_pad=9, w_pad=2)\ncols = ['site_id', 'laterality', 'view', 'age',\n       'cancer', 'biopsy', 'invasive', 'BIRADS', 'implant', 'density',\n       'machine_id', 'difficult_negative_case']\nk = 0\nfor i in range(4):\n    for j in range(3):\n        plot_dist(cols[k], ax[i][j])\n        k += 1\n__ = plt.suptitle(\"Cancer Distributions of features\", fontsize= 25)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:22.067255Z","iopub.execute_input":"2022-12-21T03:50:22.069562Z","iopub.status.idle":"2022-12-21T03:50:25.008352Z","shell.execute_reply.started":"2022-12-21T03:50:22.069521Z","shell.execute_reply":"2022-12-21T03:50:25.007165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we can see that:\n\n* Most people that had cancer were of density B & D\n* BIRADS was 0\n* There was a pretty even distribution between L & R\n* It was detected mostly in the MLO && CC View\n* Largely was detected with machine 49 interestingly\n\nThere appears to be some correlation with the tabular data.  We can possibly use this against the image data","metadata":{}},{"cell_type":"markdown","source":"# Clean and check the csv data","metadata":{}},{"cell_type":"code","source":"def clean_df(df):\n    cleaned_df = df.copy()\n    cleaned_df.age = cleaned_df.age.fillna(cleaned_df.age.mean())\n    cleaned_df.density = cleaned_df.density.fillna(\"Unknown\").astype('category')\n    cleaned_df.BIRADS = cleaned_df.BIRADS.fillna(\"Unknown\").astype('category')\n\n    cleaned_df.site_id = cleaned_df.site_id.astype('category')\n    cleaned_df.laterality = cleaned_df.laterality.astype('category')\n    cleaned_df.view = cleaned_df.view.astype('category')\n    cleaned_df.cancer = cleaned_df.cancer.astype('category')\n    cleaned_df.biopsy = cleaned_df.biopsy.astype('category')\n    cleaned_df.invasive = cleaned_df.invasive.astype('category')\n\n    cleaned_df.implant = cleaned_df.implant.astype('category')\n    cleaned_df.machine_id = cleaned_df.machine_id.astype('category')\n    cleaned_df.difficult_negative_case = cleaned_df.difficult_negative_case.astype('category')\n    \n    return cleaned_df\n\n\n\ntrain_df = clean_df(df)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:25.012919Z","iopub.execute_input":"2022-12-21T03:50:25.015376Z","iopub.status.idle":"2022-12-21T03:50:25.080861Z","shell.execute_reply.started":"2022-12-21T03:50:25.015335Z","shell.execute_reply":"2022-12-21T03:50:25.079928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:25.085318Z","iopub.execute_input":"2022-12-21T03:50:25.087625Z","iopub.status.idle":"2022-12-21T03:50:25.112516Z","shell.execute_reply.started":"2022-12-21T03:50:25.087585Z","shell.execute_reply":"2022-12-21T03:50:25.111581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.density","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:25.116986Z","iopub.execute_input":"2022-12-21T03:50:25.119538Z","iopub.status.idle":"2022-12-21T03:50:25.132403Z","shell.execute_reply.started":"2022-12-21T03:50:25.1195Z","shell.execute_reply":"2022-12-21T03:50:25.131261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.BIRADS","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:25.136835Z","iopub.execute_input":"2022-12-21T03:50:25.1393Z","iopub.status.idle":"2022-12-21T03:50:25.151253Z","shell.execute_reply.started":"2022-12-21T03:50:25.139261Z","shell.execute_reply":"2022-12-21T03:50:25.150288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Recheck\nmissing_data = pd.DataFrame({'total_missing': train_df.isnull().sum(), 'perc_missing': (train_df.isnull().sum()/len(df))*100})\nmissing_data","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:25.155496Z","iopub.execute_input":"2022-12-21T03:50:25.157755Z","iopub.status.idle":"2022-12-21T03:50:25.188044Z","shell.execute_reply.started":"2022-12-21T03:50:25.157718Z","shell.execute_reply":"2022-12-21T03:50:25.187155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate more cancer images to balance dataset","metadata":{}},{"cell_type":"code","source":"# Test image augmenation/generation with keras\nimport random\nimport numpy as np\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.preprocessing import image\n\ndef add_noise(img):\n    '''Add random noise to an image'''\n    VARIABILITY = 10\n    deviation = VARIABILITY*random.random()\n    noise = np.random.normal(0, deviation, img.shape)\n    img += noise\n    np.clip(img, 0., 255.)\n    return img\n\n# Prepare data-augmenting data generator\n\ndatagen = ImageDataGenerator(\n        rescale=1./255,\n        rotation_range=100,\n        width_shift_range=0.1,\n        height_shift_range=0.1,\n        zoom_range=0.1,\n        preprocessing_function=add_noise,\n    )\n\n# Load a single image as our example\n\nimg_path = '/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_dicomsdl_256/train_images_processed_cv2_dicomsdl_256/10006/1459541791.png'\nimg = image.load_img(img_path, target_size=(TARGET_IMG_SIZE,TARGET_IMG_SIZE))\n\n# Generate distorted images\nimages = [img]\nimg_arr = image.img_to_array(img)\nimg_arr = img_arr.reshape((1,) + img_arr.shape)\nfor batch in datagen.flow(img_arr, batch_size=1):\n    images.append( image.array_to_img(batch[0]) )\n    if len(images) >= 4:\n        break\n\n# Display\nimport matplotlib.pyplot as plt\nf, xyarr = plt.subplots(2,2)\nxyarr[0,0].imshow(images[0])\nxyarr[0,1].imshow(images[1])\nxyarr[1,0].imshow(images[2])\nxyarr[1,1].imshow(images[3])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:25.192165Z","iopub.execute_input":"2022-12-21T03:50:25.19443Z","iopub.status.idle":"2022-12-21T03:50:25.864046Z","shell.execute_reply.started":"2022-12-21T03:50:25.194392Z","shell.execute_reply":"2022-12-21T03:50:25.863082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now gen the cancer augmentations and save","metadata":{}},{"cell_type":"code","source":"work_df = pd.read_csv(TRAIN_CSV)\ncancer_df = work_df[work_df.cancer == 1]\n\nprint(len(cancer_df))\n\ndatagen = ImageDataGenerator(\n        rescale=1./255,\n        rotation_range=100,\n        width_shift_range=0.1,\n        height_shift_range=0.1,\n        zoom_range=0.1,\n        preprocessing_function=add_noise,\n    )\n\ndef gen_more_cancer(disable=False):\n    if(disable):\n        return\n    \n    base_path = TRAIN_PREPROCESSED_PATH.replace(TRAIN_PREPROCESSED_ROOT, OUT_ROOT_DIR + '/cancer')\n    os.makedirs(base_path, exist_ok=True)\n    \n    for i, r in tqdm(cancer_df.iterrows(), total=len(cancer_df)):\n        # Create the out dir if not exists\n        os.makedirs(f'{base_path}/{r.patient_id}', exist_ok=True)\n        \n        # Input image path\n        img_path = f'{TRAIN_PREPROCESSED_PATH}/{r.patient_id}/{r.image_id}.png'\n        \n        # Output image path without extension\n        img_out_path = f'{base_path}/{r.patient_id}/{r.image_id}'\n        \n        # Load the image\n        img = image.load_img(img_path, target_size=(TARGET_IMG_SIZE,TARGET_IMG_SIZE))\n        \n        # Generate distorted images\n        images = [img]\n        img_arr = image.img_to_array(img)\n        img_arr = img_arr.reshape((1,) + img_arr.shape)\n        \n        test_count=0\n        for batch in datagen.flow(img_arr, batch_size=1):\n            test_count+=1\n            images.append( image.array_to_img(batch[0]) )\n            \n            if len(images) >= NUMBER_OF_CANCER_AUG_PER:\n                break\n                \n        # save the distorted images\n        count = 0\n        for img in images:\n            dest = f'{img_out_path}-gen-{count}.png'\n            img.save(dest, 'png')\n            count += 1\n    \n    \ngen_more_cancer(disable=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:25.868425Z","iopub.execute_input":"2022-12-21T03:50:25.870681Z","iopub.status.idle":"2022-12-21T03:50:25.977635Z","shell.execute_reply.started":"2022-12-21T03:50:25.87064Z","shell.execute_reply":"2022-12-21T03:50:25.976622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets give fastai a go for image classification","metadata":{}},{"cell_type":"code","source":"# We'll use a small sample of the df for testing\ntrain_df = pd.read_csv(TRAIN_CSV)\n\n# Preprocess our features\ntrain_df = clean_df(train_df)\n\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:25.979407Z","iopub.execute_input":"2022-12-21T03:50:25.980097Z","iopub.status.idle":"2022-12-21T03:50:26.150941Z","shell.execute_reply.started":"2022-12-21T03:50:25.980058Z","shell.execute_reply":"2022-12-21T03:50:26.149819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now for each of the cancer images, let add the new image data\nfinal_train_df = train_df.copy()\n\ncancer_free_df = final_train_df[final_train_df.cancer == 0]\ncancer_df = final_train_df[final_train_df.cancer == 1]\n\n# calc the split for train and val\ncancer_mid = int(len(cancer_df) * 0.8)\n\ncancer_idx = cancer_mid\ncancer_free_idx = len(cancer_free_df) - cancer_mid\n\ntrain = pd.concat([cancer_free_df[:cancer_free_idx], cancer_df[:cancer_idx]])\nval = pd.concat([cancer_free_df[cancer_free_idx:], cancer_df[cancer_idx:]])\n\ncancer_rows = []\nval_cancer_aug_rows = []\n\n# Add the train cancer augmentations\nfor i, r in tqdm(cancer_df.iterrows(), total=len(cancer_df)):\n    image_id = r.image_id\n    for n in range(NUMBER_OF_CANCER_AUG_PER):\n        newr = r.copy(deep=True)\n        newr.image_id = f'{image_id}-gen-{n}'\n        cancer_rows.append(newr)\n        \n# Upsample val cancer to around 50/50\nfor i, r in tqdm(val[val.cancer == 1].iterrows(), total=len(val[val.cancer == 1])):\n    image_id = r.image_id\n    for n in range(3):\n        newr = r.copy(deep=True)\n        newr.image_id = f'{image_id}-gen-{n}'\n        val_cancer_aug_rows.append(newr)\n        \ntrain = pd.concat([train, pd.DataFrame(cancer_rows)]).sample(frac=1).reset_index()\nval = pd.concat([val, pd.DataFrame(val_cancer_aug_rows)]).sample(frac=1).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:26.152686Z","iopub.execute_input":"2022-12-21T03:50:26.153431Z","iopub.status.idle":"2022-12-21T03:50:36.757725Z","shell.execute_reply.started":"2022-12-21T03:50:26.153392Z","shell.execute_reply":"2022-12-21T03:50:36.756671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['isval'] = False","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:36.76245Z","iopub.execute_input":"2022-12-21T03:50:36.765165Z","iopub.status.idle":"2022-12-21T03:50:36.772723Z","shell.execute_reply.started":"2022-12-21T03:50:36.765122Z","shell.execute_reply":"2022-12-21T03:50:36.77173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val['isval'] = True","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:36.774331Z","iopub.execute_input":"2022-12-21T03:50:36.775386Z","iopub.status.idle":"2022-12-21T03:50:36.784018Z","shell.execute_reply.started":"2022-12-21T03:50:36.775342Z","shell.execute_reply":"2022-12-21T03:50:36.782977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn_df = pd.concat([val, train]).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:36.788612Z","iopub.execute_input":"2022-12-21T03:50:36.791319Z","iopub.status.idle":"2022-12-21T03:50:36.860183Z","shell.execute_reply.started":"2022-12-21T03:50:36.791281Z","shell.execute_reply":"2022-12-21T03:50:36.859125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.cancer.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:36.865095Z","iopub.execute_input":"2022-12-21T03:50:36.867447Z","iopub.status.idle":"2022-12-21T03:50:36.880916Z","shell.execute_reply.started":"2022-12-21T03:50:36.867407Z","shell.execute_reply":"2022-12-21T03:50:36.879683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val.cancer.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:36.885468Z","iopub.execute_input":"2022-12-21T03:50:36.887754Z","iopub.status.idle":"2022-12-21T03:50:36.899689Z","shell.execute_reply.started":"2022-12-21T03:50:36.887716Z","shell.execute_reply":"2022-12-21T03:50:36.898469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_y(x): return x.cancer\ndef get_x(x):\n    if x.iloc[-1] == 'test':\n        return f\"{TEST_PNG_DIR}/{x.patient_id}/{x.image_id}.png\"\n    \n    if type(x.image_id) == str and '-gen-' in x.image_id:\n        return f'{CANCER_GEN_DIR}/{x.patient_id}/{x.image_id}.png'\n        \n    return f\"{TRAIN_PREPROCESSED_PATH}/{x.patient_id}/{x.image_id}.png\"\n","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:36.904203Z","iopub.execute_input":"2022-12-21T03:50:36.906447Z","iopub.status.idle":"2022-12-21T03:50:36.914685Z","shell.execute_reply.started":"2022-12-21T03:50:36.90641Z","shell.execute_reply":"2022-12-21T03:50:36.913401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_learner(bs, accum):\n    bcd = DataBlock(blocks=(ImageBlock, CategoryBlock),\n                       get_x=get_x,\n                       get_y=get_y,\n                       splitter=ColSplitter(col='isval'),\n                       item_tfms = Resize(224, method=ResizeMethod.Squish),\n                       batch_tfms=[\n                           FlipItem(),\n    #                        DihedralItem(),\n    #                        Brightness(max_lighting=0.1),\n    #                        Contrast(),\n    #                        Hue(),\n                           Normalize.from_stats(*imagenet_stats)\n                       ]\n                   )\n\n    dls = bcd.dataloaders(learn_df, num_workers=8, bs=bs//accum)\n    \n    learn = vision_learner(dls, 'convnext_small_in22k', metrics=[F1Score(), accuracy], cbs=GradientAccumulation(bs))\n    \n    return learn","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:36.920181Z","iopub.execute_input":"2022-12-21T03:50:36.922789Z","iopub.status.idle":"2022-12-21T03:50:36.932857Z","shell.execute_reply.started":"2022-12-21T03:50:36.922752Z","shell.execute_reply":"2022-12-21T03:50:36.93182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load in the cached models locally since we are forced to be offline in the rules\nif not os.path.exists('/root/.cache/torch/hub/checkpoints/'):\n        os.makedirs('/root/.cache/torch/hub/checkpoints/')\n!cp '/kaggle/input/digitaldeusrsna/torchhub/hub/checkpoints/resnet34-b627a593.pth' '/root/.cache/torch/hub/checkpoints/resnet34-b627a593.pth'\n!cp '/kaggle/input/digitaldeusrsna/torchhub/hub/checkpoints/xrn50_940.pth' '/root/.cache/torch/hub/checkpoints/xrn50_940.pth'\n!cp '/kaggle/input/digitaldeusrsna/torchhub/hub/checkpoints/convnext_small_22k_224.pth' '/root/.cache/torch/hub/checkpoints/convnext_small_22k_224.pth'","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:36.937212Z","iopub.execute_input":"2022-12-21T03:50:36.939796Z","iopub.status.idle":"2022-12-21T03:50:43.737821Z","shell.execute_reply.started":"2022-12-21T03:50:36.939759Z","shell.execute_reply":"2022-12-21T03:50:43.736255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# learn = get_learner(bs=128, accum=2)\n# learn.fine_tune(EPOCHS, 1e-2, cbs=[EarlyStoppingCallback(monitor='f1_score', min_delta=0.001, patience=5),  SaveModelCallback(monitor='f1_score', min_delta=0.001)])\n# learn.export()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:43.743525Z","iopub.execute_input":"2022-12-21T03:50:43.745903Z","iopub.status.idle":"2022-12-21T03:50:43.759926Z","shell.execute_reply.started":"2022-12-21T03:50:43.745858Z","shell.execute_reply":"2022-12-21T03:50:43.751653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = load_learner('/kaggle/input/digitaldeusrsna/export.pkl')","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:43.76352Z","iopub.execute_input":"2022-12-21T03:50:43.772535Z","iopub.status.idle":"2022-12-21T03:50:45.852204Z","shell.execute_reply.started":"2022-12-21T03:50:43.772497Z","shell.execute_reply":"2022-12-21T03:50:45.851141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets check out the predictions vs the tta predictions","metadata":{}},{"cell_type":"code","source":"# valid = learn.dls.valid\n# preds,targs = learn.get_preds(dl=valid)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:45.857595Z","iopub.execute_input":"2022-12-21T03:50:45.859989Z","iopub.status.idle":"2022-12-21T03:50:45.865746Z","shell.execute_reply.started":"2022-12-21T03:50:45.859949Z","shell.execute_reply":"2022-12-21T03:50:45.864821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Learn score","metadata":{}},{"cell_type":"code","source":"# error_rate(preds, targs)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:45.870268Z","iopub.execute_input":"2022-12-21T03:50:45.872885Z","iopub.status.idle":"2022-12-21T03:50:45.879122Z","shell.execute_reply.started":"2022-12-21T03:50:45.872796Z","shell.execute_reply":"2022-12-21T03:50:45.878124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# learn.dls.train.show_batch(max_n=6, unique=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:45.882394Z","iopub.execute_input":"2022-12-21T03:50:45.883868Z","iopub.status.idle":"2022-12-21T03:50:45.890295Z","shell.execute_reply.started":"2022-12-21T03:50:45.883815Z","shell.execute_reply":"2022-12-21T03:50:45.889321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And here's the tta score","metadata":{}},{"cell_type":"code","source":"# tta_preds,y = learn.tta(dl=valid)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:45.891877Z","iopub.execute_input":"2022-12-21T03:50:45.892802Z","iopub.status.idle":"2022-12-21T03:50:45.899806Z","shell.execute_reply.started":"2022-12-21T03:50:45.892765Z","shell.execute_reply":"2022-12-21T03:50:45.898812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# error_rate(tta_preds, targs)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:45.9014Z","iopub.execute_input":"2022-12-21T03:50:45.903166Z","iopub.status.idle":"2022-12-21T03:50:45.909383Z","shell.execute_reply.started":"2022-12-21T03:50:45.903128Z","shell.execute_reply":"2022-12-21T03:50:45.908406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now predict on the test data","metadata":{}},{"cell_type":"markdown","source":"* Create augmentation of cancerous images via https://www.tensorflow.org/api_docs/python/tf/keras/preprocessing/image/ImageDataGenerator\n* Convert the test images to png\n* Predict on the test images\n* Run regression on the predictions\n* Save the result","metadata":{}},{"cell_type":"code","source":"!pip install dicomsdl --no-index --find-links=file:///kaggle/input/digitaldeusrsna/pip_packages/pip_packages/\n\n# Thanks!!!\n# https://www.kaggle.com/code/radek1/how-to-process-dicom-images-to-pngs?scriptVersionId=113227375\n\nimport pydicom\nimport numpy as np\nimport cv2\nimport os\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport dicomsdl\n\nRESIZE_TO = (TARGET_IMG_SIZE, TARGET_IMG_SIZE)\n\n!rm -rf train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}\n!mkdir train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}\n\n# https://www.kaggle.com/code/tanlikesmath/brain-tumor-radiogenomic-classification-eda/notebook\ndef dicom_file_to_ary(path):\n    dcm_file = dicomsdl.open(str(path))\n    data = dcm_file.pixelData()\n\n    data = (data - data.min()) / (data.max() - data.min())\n\n    if dcm_file.getPixelDataInfo()['PhotometricInterpretation'] == \"MONOCHROME1\":\n        data = 1 - data\n\n    data = cv2.resize(data, RESIZE_TO)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndirectories = list(Path('/kaggle/input/rsna-breast-cancer-detection/test_images').iterdir())\n\ndef process_directory(directory_path):\n    parent_directory = str(directory_path).split('/')[-1]\n    !mkdir -p train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/{parent_directory}\n    for image_path in directory_path.iterdir():\n        processed_ary = dicom_file_to_ary(image_path)\n        \n        cv2.imwrite(\n            f'train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/{parent_directory}/{image_path.stem}.png',\n            processed_ary\n        )\n    print(\"Done!!!\")\n        \nimport multiprocessing as mp\n\nwith mp.Pool(64) as p:\n    p.map(process_directory, directories)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:50:45.929664Z","iopub.execute_input":"2022-12-21T03:50:45.930324Z","iopub.status.idle":"2022-12-21T03:51:10.837001Z","shell.execute_reply.started":"2022-12-21T03:50:45.930287Z","shell.execute_reply":"2022-12-21T03:51:10.835419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_df['test'] = 'test'\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:51:10.839009Z","iopub.execute_input":"2022-12-21T03:51:10.839629Z","iopub.status.idle":"2022-12-21T03:51:10.883229Z","shell.execute_reply.started":"2022-12-21T03:51:10.83958Z","shell.execute_reply":"2022-12-21T03:51:10.882325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tst_dl = learn.dls.test_dl(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:51:10.884636Z","iopub.execute_input":"2022-12-21T03:51:10.8852Z","iopub.status.idle":"2022-12-21T03:51:10.9424Z","shell.execute_reply.started":"2022-12-21T03:51:10.885162Z","shell.execute_reply":"2022-12-21T03:51:10.941566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds, _ = learn.tta(dl=tst_dl)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:51:10.943742Z","iopub.execute_input":"2022-12-21T03:51:10.944288Z","iopub.status.idle":"2022-12-21T03:51:19.991614Z","shell.execute_reply.started":"2022-12-21T03:51:10.944251Z","shell.execute_reply":"2022-12-21T03:51:19.990428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predsdf = pd.DataFrame(preds, columns=[\"pneg\", \"ppos\"])\nfinal_test_df = pd.concat([test_df, test_predsdf], axis = 1)\nfinal_test_df.drop(['test', 'prediction_id'], inplace=True, axis=1)\nfinal_test_df","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:51:19.996428Z","iopub.execute_input":"2022-12-21T03:51:19.998744Z","iopub.status.idle":"2022-12-21T03:51:20.026648Z","shell.execute_reply.started":"2022-12-21T03:51:19.998696Z","shell.execute_reply":"2022-12-21T03:51:20.025741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(data={'prediction_id': test_df.prediction_id, 'cancer': final_test_df['ppos']}).reset_index(drop=True)\nsubmission = submission.sort_values('cancer', ascending=False).drop_duplicates(['prediction_id']).reset_index(drop=True)\n\nsubmission.cancer = (submission.cancer >= 0.4).astype(int)\n\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:51:20.03076Z","iopub.execute_input":"2022-12-21T03:51:20.033456Z","iopub.status.idle":"2022-12-21T03:51:20.053946Z","shell.execute_reply.started":"2022-12-21T03:51:20.033418Z","shell.execute_reply":"2022-12-21T03:51:20.053005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T03:51:20.058044Z","iopub.execute_input":"2022-12-21T03:51:20.060348Z","iopub.status.idle":"2022-12-21T03:51:20.069729Z","shell.execute_reply.started":"2022-12-21T03:51:20.060311Z","shell.execute_reply":"2022-12-21T03:51:20.068725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}