{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nsys.path.append('/path/to/dicomsdl')","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:44:54.592544Z","iopub.execute_input":"2023-06-30T14:44:54.5933Z","iopub.status.idle":"2023-06-30T14:44:54.653878Z","shell.execute_reply.started":"2023-06-30T14:44:54.593191Z","shell.execute_reply":"2023-06-30T14:44:54.652916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install dicomsdl","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:44:54.655339Z","iopub.execute_input":"2023-06-30T14:44:54.65646Z","iopub.status.idle":"2023-06-30T14:45:12.448243Z","shell.execute_reply.started":"2023-06-30T14:44:54.656419Z","shell.execute_reply":"2023-06-30T14:45:12.446858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport dicomsdl\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport cv2\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:12.451939Z","iopub.execute_input":"2023-06-30T14:45:12.453218Z","iopub.status.idle":"2023-06-30T14:45:17.195658Z","shell.execute_reply.started":"2023-06-30T14:45:12.453166Z","shell.execute_reply":"2023-06-30T14:45:17.19448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntrain.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:17.202331Z","iopub.execute_input":"2023-06-30T14:45:17.202957Z","iopub.status.idle":"2023-06-30T14:45:17.351097Z","shell.execute_reply.started":"2023-06-30T14:45:17.202922Z","shell.execute_reply":"2023-06-30T14:45:17.349897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:17.353571Z","iopub.execute_input":"2023-06-30T14:45:17.354759Z","iopub.status.idle":"2023-06-30T14:45:17.403713Z","shell.execute_reply.started":"2023-06-30T14:45:17.354714Z","shell.execute_reply":"2023-06-30T14:45:17.402378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train.cancer)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:17.405466Z","iopub.execute_input":"2023-06-30T14:45:17.405878Z","iopub.status.idle":"2023-06-30T14:45:17.632547Z","shell.execute_reply.started":"2023-06-30T14:45:17.405838Z","shell.execute_reply":"2023-06-30T14:45:17.631519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train.view)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:17.634063Z","iopub.execute_input":"2023-06-30T14:45:17.635086Z","iopub.status.idle":"2023-06-30T14:45:17.902432Z","shell.execute_reply.started":"2023-06-30T14:45:17.635047Z","shell.execute_reply":"2023-06-30T14:45:17.901317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(train.invasive)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:17.90386Z","iopub.execute_input":"2023-06-30T14:45:17.904279Z","iopub.status.idle":"2023-06-30T14:45:18.088506Z","shell.execute_reply.started":"2023-06-30T14:45:17.904235Z","shell.execute_reply":"2023-06-30T14:45:18.087431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig=px.line_polar(train,r='cancer',theta='view',color='invasive',line_group='biopsy',line_close=True,markers=True, direction='clockwise')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:18.089799Z","iopub.execute_input":"2023-06-30T14:45:18.090419Z","iopub.status.idle":"2023-06-30T14:45:20.168609Z","shell.execute_reply.started":"2023-06-30T14:45:18.090389Z","shell.execute_reply":"2023-06-30T14:45:20.167678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(train,x='age',kde=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:20.169671Z","iopub.execute_input":"2023-06-30T14:45:20.169992Z","iopub.status.idle":"2023-06-30T14:45:20.87931Z","shell.execute_reply.started":"2023-06-30T14:45:20.169961Z","shell.execute_reply":"2023-06-30T14:45:20.878102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['laterality','cancer']].groupby(['laterality']).sum()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:20.883315Z","iopub.execute_input":"2023-06-30T14:45:20.883715Z","iopub.status.idle":"2023-06-30T14:45:20.907293Z","shell.execute_reply.started":"2023-06-30T14:45:20.883672Z","shell.execute_reply":"2023-06-30T14:45:20.905772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.pie(train,names='laterality',values='cancer')","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:20.909099Z","iopub.execute_input":"2023-06-30T14:45:20.909594Z","iopub.status.idle":"2023-06-30T14:45:21.291037Z","shell.execute_reply.started":"2023-06-30T14:45:20.909554Z","shell.execute_reply":"2023-06-30T14:45:21.290139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:21.297754Z","iopub.execute_input":"2023-06-30T14:45:21.298379Z","iopub.status.idle":"2023-06-30T14:45:21.326446Z","shell.execute_reply.started":"2023-06-30T14:45:21.298338Z","shell.execute_reply":"2023-06-30T14:45:21.325564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape\n","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:21.327929Z","iopub.execute_input":"2023-06-30T14:45:21.328547Z","iopub.status.idle":"2023-06-30T14:45:21.335684Z","shell.execute_reply.started":"2023-06-30T14:45:21.328493Z","shell.execute_reply":"2023-06-30T14:45:21.334668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:21.337176Z","iopub.execute_input":"2023-06-30T14:45:21.338246Z","iopub.status.idle":"2023-06-30T14:45:21.352998Z","shell.execute_reply.started":"2023-06-30T14:45:21.338209Z","shell.execute_reply":"2023-06-30T14:45:21.351668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:21.354936Z","iopub.execute_input":"2023-06-30T14:45:21.355647Z","iopub.status.idle":"2023-06-30T14:45:21.387924Z","shell.execute_reply.started":"2023-06-30T14:45:21.35559Z","shell.execute_reply":"2023-06-30T14:45:21.386962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:21.389516Z","iopub.execute_input":"2023-06-30T14:45:21.390208Z","iopub.status.idle":"2023-06-30T14:45:21.397052Z","shell.execute_reply.started":"2023-06-30T14:45:21.39017Z","shell.execute_reply":"2023-06-30T14:45:21.396031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=train.drop(['BIRADS','density'],axis=1)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:21.398676Z","iopub.execute_input":"2023-06-30T14:45:21.399418Z","iopub.status.idle":"2023-06-30T14:45:21.41192Z","shell.execute_reply.started":"2023-06-30T14:45:21.39937Z","shell.execute_reply":"2023-06-30T14:45:21.410848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nmem=SimpleImputer(strategy='mean')\ntrain['age']=mem.fit_transform(train[['age']])\n","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:21.413656Z","iopub.execute_input":"2023-06-30T14:45:21.414354Z","iopub.status.idle":"2023-06-30T14:45:21.624971Z","shell.execute_reply.started":"2023-06-30T14:45:21.414316Z","shell.execute_reply":"2023-06-30T14:45:21.623666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf=train.select_dtypes(include=(['int64','float64']))\nplt.figure(figsize=(9,9))\nsns.heatmap(df.corr(method='spearman'),cmap='Dark2',annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:21.626943Z","iopub.execute_input":"2023-06-30T14:45:21.627362Z","iopub.status.idle":"2023-06-30T14:45:22.388828Z","shell.execute_reply.started":"2023-06-30T14:45:21.627319Z","shell.execute_reply":"2023-06-30T14:45:22.38606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:22.390704Z","iopub.execute_input":"2023-06-30T14:45:22.391368Z","iopub.status.idle":"2023-06-30T14:45:22.398413Z","shell.execute_reply.started":"2023-06-30T14:45:22.391328Z","shell.execute_reply":"2023-06-30T14:45:22.39723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_data='/kaggle/input/rsna-breast-cancer-detection/{}_images/{}/{}.dcm'\ntrain_image='train'\ntest_image='test'\n","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:22.400247Z","iopub.execute_input":"2023-06-30T14:45:22.400657Z","iopub.status.idle":"2023-06-30T14:45:22.410125Z","shell.execute_reply.started":"2023-06-30T14:45:22.400616Z","shell.execute_reply":"2023-06-30T14:45:22.409012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img=dicomsdl.open('/kaggle/input/rsna-breast-cancer-detection/train_images/10006/1874946579.dcm').pixelData()\nplt.imshow(img,cmap='ocean_r')","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:22.412001Z","iopub.execute_input":"2023-06-30T14:45:22.412423Z","iopub.status.idle":"2023-06-30T14:45:24.417535Z","shell.execute_reply.started":"2023-06-30T14:45:22.412384Z","shell.execute_reply":"2023-06-30T14:45:24.416571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_id=train[train.cancer==1].iloc[0].patient_id\npatient_df=train[train.patient_id==patient_id]\npatient_df","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:45:24.420854Z","iopub.execute_input":"2023-06-30T14:45:24.421179Z","iopub.status.idle":"2023-06-30T14:45:24.445931Z","shell.execute_reply.started":"2023-06-30T14:45:24.421149Z","shell.execute_reply":"2023-06-30T14:45:24.444773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install keras_cv_attention_models","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:50:05.007158Z","iopub.execute_input":"2023-06-30T14:50:05.007647Z","iopub.status.idle":"2023-06-30T14:50:50.563362Z","shell.execute_reply.started":"2023-06-30T14:50:05.00758Z","shell.execute_reply":"2023-06-30T14:50:50.562085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nfrom multiprocessing import cpu_count\nfrom keras_cv_attention_models import convnext\n\nimport cv2\nimport glob\nimport importlib\nimport os\nimport joblib\nimport time\nimport dicomsdl\nimport gc","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:50:50.566542Z","iopub.execute_input":"2023-06-30T14:50:50.56789Z","iopub.status.idle":"2023-06-30T14:50:51.727386Z","shell.execute_reply.started":"2023-06-30T14:50:50.567842Z","shell.execute_reply":"2023-06-30T14:50:51.726233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n\nTARGET_HEIGHT = 1344\nTARGET_WIDTH = 768\nN_CHANNELS = 1\nINPUT_SHAPE = (TARGET_HEIGHT, TARGET_WIDTH, N_CHANNELS)\nTARGET_HEIGHT_WIDTH_RATIO = TARGET_HEIGHT / TARGET_WIDTH\nTHRESHOLD_BEST = 0.50\n\nCLAHE = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(32, 32))\n\nCROP_IMAGE = True\nAPPLY_CLAHE = False\nAPPLY_EQ_HIST = False\n\nIMAGE_FORMAT = 'jpg'","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:50:53.037635Z","iopub.execute_input":"2023-06-30T14:50:53.038043Z","iopub.status.idle":"2023-06-30T14:50:53.054115Z","shell.execute_reply.started":"2023-06-30T14:50:53.038002Z","shell.execute_reply":"2023-06-30T14:50:53.052978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def voi_lut(image, dicom):\n    if 'WindowWidth' not in dicom.getPixelDataInfo() or 'WindowWidth' not in dicom.getPixelDataInfo():\n        return image\n    \n   \n    center = dicom['WindowCenter']\n    width = dicom['WindowWidth']\n    bits_stored = dicom['BitsStored']\n    voi_lut_function = dicom['VOILUTFunction']\n    if isinstance(center, list):\n        center = center[0]\n    if isinstance(width, list):\n        width = width[0]\n\n    y_min = 0\n    y_max = float(2**bits_stored - 1)\n    y_range = y_max\n\n   \n    if voi_lut_function == 'SIGMOID':\n        image = y_range / (1 + np.exp(-4 * (image - center) / width)) + y_min\n    else:\n       \n        center -= 0.5\n        width -= 1\n\n        below = image <= (center - width / 2)\n        above = image > (center + width / 2)\n        between = np.logical_and(~below, ~above)\n\n        image[below] = y_min\n        image[above] = y_max\n        if between.any():\n            image[between] = (\n                ((image[between] - center) / width + 0.5) * y_range + y_min\n            )\n\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:50:53.470923Z","iopub.execute_input":"2023-06-30T14:50:53.471314Z","iopub.status.idle":"2023-06-30T14:50:53.484479Z","shell.execute_reply.started":"2023-06-30T14:50:53.471281Z","shell.execute_reply":"2023-06-30T14:50:53.483026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def smooth(l):\n    # kernel size is 1% of vector\n    kernel_size = int(len(l) * 0.01)\n    kernel = np.ones(kernel_size) / kernel_size\n    return np.convolve(l, kernel, mode='same')\n\n# X Crop offset based on first column with sum below 5% of maximum column sums*std\ndef get_x_offset(image, max_col_sum_ratio_threshold=0.05, debug=None):\n    # Image Dimensions\n    H, W = image.shape\n    # Percentual margin added to offset\n    margin = int(image.shape[1] * 0.00)\n    # Threshold values based on smoothed sum x std to capture varying intensity columns\n    vv = smooth(image.sum(axis=0).squeeze()) * smooth(image.std(axis=0).squeeze())\n    # Find maximum sum in first 75% of columns\n    vv_argmax = vv[:int(image.shape[1] * 0.75)].argmax()\n    # Threshold value\n    vv_threshold = vv.max() * max_col_sum_ratio_threshold\n    \n    # Find first column after maximum column below threshold value\n    for offset, v in enumerate(vv):\n        # Start searching from vv_argmax\n        if offset < vv_argmax:\n            continue\n        \n        # Column below threshold value found\n        if v < vv_threshold:\n            offset = min(W, offset + margin)\n            break\n            \n    if isinstance(debug, np.ndarray):\n        debug[1].imshow(image)\n        debug[1].set_title('X Offset')\n        vv_scale = H / vv.max() * 0.90\n        # Values\n        debug[1].plot(H - vv * vv_scale , c='red', label='vv')\n        # Threshold\n        debug[1].hlines(H - vv_threshold * vv_scale, 0, W -1, colors='orange', label='threshold')\n        # Max Value\n        debug[1].scatter(vv_argmax, H - vv[vv_argmax] * vv_scale, c='blue', s=100, label='Max', zorder=np.PINF)\n        # First Column Below Threshold\n        debug[1].scatter(offset, H - vv[offset] * vv_scale, c='purple', s=100, label='Offset', zorder=np.PINF)\n        debug[1].set_ylim(H, 0)\n        debug[1].legend()\n        debug[1].axis('off')\n        \n    return offset\n\n# Y Crop offset based on first bottom and top rows with sum below 10% of maximum row sum*std\ndef get_y_offsets(image, max_row_sum_ratio_threshold=0.10, debug=None):\n    # Image Dimensions\n    H, W = image.shape\n    # Margin to add to offsets\n    margin = 0\n    # Threshold values based on smoothed sum x std to capture varying intensity columns\n    vv = smooth(image.sum(axis=1).squeeze()) * smooth(image.std(axis=1).squeeze())\n    # Find maximum sum * std row in inter quartile rows\n    vv_argmax = int(image.shape[0] * 0.25) + vv[int(image.shape[0] * 0.25):int(image.shape[0] * 0.75)].argmax()\n    # Threshold value\n    vv_threshold = vv.max() * max_row_sum_ratio_threshold\n    # Default crop offsets\n    offset_bottom = 0\n    offset_top = H\n\n    # Bottom offset, search from argmax to bottom\n    for offset in reversed(range(0, vv_argmax)):\n        v = vv[offset]\n        if v < vv_threshold:\n            offset_bottom = offset\n            break\n    \n    if isinstance(debug, np.ndarray):\n        debug[2].imshow(image)\n        debug[2].set_title('Y Bottom Offset')\n        vv_scale = W / vv.max() * 0.90\n        # Values\n        debug[2].plot(vv * vv_scale, np.arange(H), c='red', label='vv')\n        # Threshold\n        debug[2].vlines(vv_threshold * vv_scale, 0, H -1, colors='orange', label='threshold')\n        # Max Value\n        debug[2].scatter(vv[vv_argmax] * vv_scale, vv_argmax, c='blue', s=100, label='Max', zorder=np.PINF)\n        # First Column Below Threshold\n        debug[2].scatter(vv[offset_bottom] * vv_scale, offset_bottom, c='purple', s=100, label='Offset', zorder=np.PINF)\n        debug[2].set_ylim(H, 0)\n        debug[2].legend()\n        debug[2].axis('off')\n            \n    # Top offset, search from argmax to top\n    for offset in range(vv_argmax, H):\n        v = vv[offset]\n        if v < vv_threshold:\n            offset_top = offset\n            break\n            \n    if isinstance(debug, np.ndarray):\n        debug[3].imshow(image)\n        debug[3].set_title('Y Top Offset')\n        vv_scale = W / vv.max() * 0.90\n        # Values\n        debug[3].plot(vv * vv_scale, np.arange(H) , c='red', label='vv')\n        # Threshold\n        debug[3].vlines(vv_threshold * vv_scale, 0, H -1, colors='orange', label='threshold')\n        # Max Value\n        debug[3].scatter(vv[vv_argmax] * vv_scale, vv_argmax, c='blue', s=100, label='Max', zorder=np.PINF)\n        # First Column Below Threshold\n        debug[3].scatter(vv[offset_top] * vv_scale, offset_top, c='purple', s=100, label='Offset', zorder=np.PINF)\n        debug[2].set_ylim(H, 0)\n        debug[3].legend()\n        debug[3].axis('off')\n            \n    return max(0, offset_bottom - margin), min(image.shape[0], offset_top + margin)\n\n# Crop image and pad offsets to target image height/width ratio to preserve information\ndef crop(image, size=None, debug=False):\n    # Image dimensions\n    H, W = image.shape\n    # Compute x/bottom/top offsets\n    x_offset = get_x_offset(image, debug=debug)\n    offset_bottom, offset_top = get_y_offsets(image[:,:x_offset], debug=debug)\n    # Crop Height and Width\n    h_crop = offset_top - offset_bottom\n    w_crop = x_offset\n    \n    # Pad crop offsets to target aspect ratio\n    if size is not None:\n        # Height too large, pad x offset\n        if (h_crop / w_crop) > TARGET_HEIGHT_WIDTH_RATIO:\n            x_offset += int(h_crop / TARGET_HEIGHT_WIDTH_RATIO - w_crop)\n        else:\n            # Height too small, pad bottom/top offsets\n            offset_bottom -= int(0.50 * (w_crop * TARGET_HEIGHT_WIDTH_RATIO - h_crop))\n            offset_bottom_correction = max(0, -offset_bottom)\n            offset_bottom += offset_bottom_correction\n\n            offset_top += int(0.50 * (w_crop * TARGET_HEIGHT_WIDTH_RATIO - h_crop))\n            offset_top += offset_bottom_correction\n        \n    # Crop Image\n    image = image[offset_bottom:offset_top:,:x_offset]\n        \n    return image","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:50:53.850402Z","iopub.execute_input":"2023-06-30T14:50:53.850871Z","iopub.status.idle":"2023-06-30T14:50:53.888649Z","shell.execute_reply.started":"2023-06-30T14:50:53.850834Z","shell.execute_reply":"2023-06-30T14:50:53.887294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(file_path, size=(TARGET_WIDTH, TARGET_HEIGHT), crop_image=CROP_IMAGE, apply_clahe=APPLY_CLAHE, apply_eq_hist=APPLY_EQ_HIST, debug=False, save=True):\n    dicom = dicomsdl.open(file_path)\n    image = dicom.pixelData()\n    \n    if debug:\n        fig, axes = plt.subplots(1, 5, figsize=(20,10))\n        image0 = np.copy(image)\n        axes[0].imshow(image0)\n        axes[0].set_title('Original Image')\n        axes[0].axis('off')\n    else:\n        axes = False\n    \n    try:\n        image = voi_lut(image, dicom)\n    except:\n        pass\n    \n   \n    if dicom.getPixelDataInfo()['PhotometricInterpretation'] == 'MONOCHROME1':\n        image = np.max(image) - image\n    image = (image - image.min()) / (image.max() - image.min())\n\n    image = (image * 255).astype(np.uint8)\n    \n    h0, w0 = image.shape\n    if image[:,int(-w0 * 0.10):].sum() > image[:,:int(w0 * 0.10)].sum():\n        image = np.flip(image, axis=1)\n    \n    if crop_image:\n        image = crop(image, debug=axes)\n        \n    if size is not None:\n        h, w = image.shape\n        if (h / w) > TARGET_HEIGHT_WIDTH_RATIO:\n            pad = int(h / TARGET_HEIGHT_WIDTH_RATIO - w)\n            image = np.pad(image, [[0,0], [0, pad]])\n            h, w = image.shape\n        else:\n            pad = int(0.50 * (w * TARGET_HEIGHT_WIDTH_RATIO - h))\n            image = np.pad(image, [[pad, pad], [0,0]])\n            h, w = image.shape\n        image = cv2.resize(image, size, interpolation=cv2.INTER_AREA)\n        \n    if apply_clahe:\n        image = CLAHE.apply(image)\n    if apply_eq_hist:\n        image = cv2.equalizeHist(image)\n           \n    if debug:\n        axes[4].imshow(image)\n        axes[4].set_title('Processed Image')\n        axes[4].axis('off')\n        plt.show()\n    if save:\n        image_id = file_path.split('/')[-1].split('.')[0]\n        if IMAGE_FORMAT == 'png':\n            cv2.imwrite(f'{image_id}.png', image)\n        else:\n            cv2.imwrite(f'{image_id}.jpg', image, [cv2.IMWRITE_JPEG_QUALITY, 95])","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:50:54.21926Z","iopub.execute_input":"2023-06-30T14:50:54.220567Z","iopub.status.idle":"2023-06-30T14:50:54.250305Z","shell.execute_reply.started":"2023-06-30T14:50:54.220504Z","shell.execute_reply":"2023-06-30T14:50:54.248111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n    \ndef get_file_path(args):\n    patient_id, image_id = args\n    return f'/kaggle/input/rsna-breast-cancer-detection/train_images/{patient_id}/{image_id}.dcm'\n    \ntrain['file_path'] = train[['patient_id', 'image_id']].apply(get_file_path, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:50:54.575203Z","iopub.execute_input":"2023-06-30T14:50:54.57618Z","iopub.status.idle":"2023-06-30T14:50:55.364678Z","shell.execute_reply.started":"2023-06-30T14:50:54.576132Z","shell.execute_reply":"2023-06-30T14:50:55.363567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5*len(patient_df),5))\nfor i in range(len(patient_df)):\n    n=patient_df.iloc[i]\n    plt.subplot(1,len(patient_df),i+1)\n    img=dicomsdl.open(image_data.format(train_image,n.patient_id,n.image_id)).pixelData()\n    plt.imshow(img,cmap=plt.cm.bone)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:50:55.366579Z","iopub.execute_input":"2023-06-30T14:50:55.366955Z","iopub.status.idle":"2023-06-30T14:51:04.156427Z","shell.execute_reply.started":"2023-06-30T14:50:55.366926Z","shell.execute_reply":"2023-06-30T14:51:04.155267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nN = 8\n\nfor fp in tqdm(train['file_path'].head(N)):\n    process(fp, crop_image=True, size=(TARGET_WIDTH, TARGET_HEIGHT), debug=True, save=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:51:04.158461Z","iopub.execute_input":"2023-06-30T14:51:04.158902Z","iopub.status.idle":"2023-06-30T14:51:32.087376Z","shell.execute_reply.started":"2023-06-30T14:51:04.158859Z","shell.execute_reply":"2023-06-30T14:51:32.086459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize(image):\n    image = tf.repeat(image, repeats=3, axis=3)\n    image = tf.cast(image, tf.float32)\n    image = tf.keras.applications.imagenet_utils.preprocess_input(image, mode='torch')\n\n    return image\n","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:51:32.089925Z","iopub.execute_input":"2023-06-30T14:51:32.090852Z","iopub.status.idle":"2023-06-30T14:51:32.097099Z","shell.execute_reply.started":"2023-06-30T14:51:32.090794Z","shell.execute_reply":"2023-06-30T14:51:32.096189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    image = tf.keras.layers.Input(INPUT_SHAPE, name='image', dtype=tf.uint8)\n    image_norm = normalize(image)\n    x = convnext.ConvNeXtV2Tiny(\n        input_shape=(TARGET_HEIGHT, TARGET_WIDTH, 3),\n        pretrained=None,\n        num_classes=0,\n    )(image_norm)\n\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.30)(x)\n    outputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n    model = tf.keras.models.Model(inputs=image, outputs=outputs)\n    model.load_weights('/kaggle/input/rsna-efficientnetv2-training-tensorflow-tpu-ds/model.h5')\n    model.trainable = False\n    model.compile()\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:51:32.098703Z","iopub.execute_input":"2023-06-30T14:51:32.099463Z","iopub.status.idle":"2023-06-30T14:51:32.112818Z","shell.execute_reply.started":"2023-06-30T14:51:32.099425Z","shell.execute_reply":"2023-06-30T14:51:32.111413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\ntf.config.optimizer.set_jit(True)\nmodel = get_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:51:32.114826Z","iopub.execute_input":"2023-06-30T14:51:32.115511Z","iopub.status.idle":"2023-06-30T14:51:42.046755Z","shell.execute_reply.started":"2023-06-30T14:51:32.115473Z","shell.execute_reply":"2023-06-30T14:51:42.045652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\ndef get_file_path(args):\n    patient_id, image_id = args\n    return f'/kaggle/input/rsna-breast-cancer-detection/test_images/{patient_id}/{image_id}.dcm'\n    \ntest['file_path'] = test[['patient_id', 'image_id']].apply(get_file_path, axis=1)\n\ndisplay(test.info())\ndisplay(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:51:42.048228Z","iopub.execute_input":"2023-06-30T14:51:42.049884Z","iopub.status.idle":"2023-06-30T14:51:42.086367Z","shell.execute_reply.started":"2023-06-30T14:51:42.049833Z","shell.execute_reply":"2023-06-30T14:51:42.085273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess a single image and saves it\ndef preprocess_and_save_image(args):\n    (patient_id, laterality), g = args\n    cancer = 0.0\n    for row_idx, row in g.iterrows():\n        process(row['file_path'], size=(TARGET_WIDTH, TARGET_HEIGHT), crop_image=True, save=True)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:51:42.088915Z","iopub.execute_input":"2023-06-30T14:51:42.089766Z","iopub.status.idle":"2023-06-30T14:51:42.09692Z","shell.execute_reply.started":"2023-06-30T14:51:42.089723Z","shell.execute_reply":"2023-06-30T14:51:42.095705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess all images in parallel using Joblib\njobs = [joblib.delayed(preprocess_and_save_image)(args) for args in test.groupby(['patient_id', 'laterality'])]\nSUBMISSION_ROWS = joblib.Parallel(\n    n_jobs=cpu_count(),\n    verbose=1,\n    backend='multiprocessing',\n    prefer='threads',\n)(jobs)","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:51:42.100958Z","iopub.execute_input":"2023-06-30T14:51:42.101466Z","iopub.status.idle":"2023-06-30T14:51:45.516747Z","shell.execute_reply.started":"2023-06-30T14:51:42.101418Z","shell.execute_reply":"2023-06-30T14:51:45.51536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')\n\ndisplay(sample_submission.info())\ndisplay(sample_submission.head())\n","metadata":{"execution":{"iopub.status.busy":"2023-06-30T14:51:45.518772Z","iopub.execute_input":"2023-06-30T14:51:45.519952Z","iopub.status.idle":"2023-06-30T14:51:45.557302Z","shell.execute_reply.started":"2023-06-30T14:51:45.519904Z","shell.execute_reply":"2023-06-30T14:51:45.555327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}