{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n# warnings.simplefilter('ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:24:42.82852Z","iopub.execute_input":"2023-02-05T14:24:42.829193Z","iopub.status.idle":"2023-02-05T14:24:42.850424Z","shell.execute_reply.started":"2023-02-05T14:24:42.829085Z","shell.execute_reply":"2023-02-05T14:24:42.849558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install -qU python-gdcm pydicom pylibjpeg\n# !pip install --quiet vit-keras\n\n# !tar xvf /kaggle/input/simpletransformers-package/simpletransformers/validators-0.20.0.tar.gz.tmp -C /kaggle/\n# !pip install /kaggle/validators-0.20.0/\n!pip install /kaggle/input/rsna-2022-whl/pydicom-2.3.0-py3-none-any.whl\n!pip install /kaggle/input/rsna-2022-whl/pylibjpeg-1.4.0-py3-none-any.whl\n!pip install /kaggle/input/rsna-2022-whl/python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install /kaggle/input/dicomsdl-offline-installer/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!pip install /kaggle/input/nvidia-dali-wheel/nvidia_dali_nightly_cuda110-1.22.0.dev20221213-6757685-py3-none-manylinux2014_x86_64.whl\nimport os, sys\nsys.path.append('/kaggle/input/vit-keras/vit-keras-master')","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:24:42.852306Z","iopub.execute_input":"2023-02-05T14:24:42.852663Z","iopub.status.idle":"2023-02-05T14:27:26.139479Z","shell.execute_reply.started":"2023-02-05T14:24:42.852627Z","shell.execute_reply":"2023-02-05T14:27:26.13818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n#import warnings\n#warnings.filterwarnings('ignore')\n#warnings.simplefilter('ignore')\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport tqdm\nimport time\nimport glob\nimport cv2\nimport os\nimport shutil\n\nfrom pydicom.filebase import DicomBytesIO\nimport dicomsdl\nimport dicomsdl as dicoml\nimport pylibjpeg\nimport pydicom\n\n\nimport joblib\nfrom joblib import Parallel, delayed\nfrom multiprocessing import Process, Pool\nimport multiprocessing\n\nimport tensorflow as tf\nprint(tf.__version__)\nimport tensorflow_io as tfio\n# from vit_keras import vit\n\n# Nvidia GPU image preprocessing progress\nfrom nvidia.dali import pipeline_def\nimport nvidia.dali.fn as fn\nimport nvidia.dali.types as types\nfrom nvidia.dali.types import DALIDataType\n\nfrom IPython.display import clear_output\nimport gc\nimport random","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":false,"execution":{"iopub.status.busy":"2023-02-05T14:27:26.14237Z","iopub.execute_input":"2023-02-05T14:27:26.143139Z","iopub.status.idle":"2023-02-05T14:27:32.153335Z","shell.execute_reply.started":"2023-02-05T14:27:26.143096Z","shell.execute_reply":"2023-02-05T14:27:32.152375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = 224\nCROP_IMAGE_SIZE = 184\nFETCH = 5000\n\nEPOCH = 25\nTOGGLE = False\nDEBUG = False\nSUBMIT = True\n\nJ2K_FOLDER = \"/kaggle/working/j2k/\"\nSAVE_FOLDER = \"/kaggle/working/save_testing/\"","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:27:32.15564Z","iopub.execute_input":"2023-02-05T14:27:32.15613Z","iopub.status.idle":"2023-02-05T14:27:32.168126Z","shell.execute_reply.started":"2023-02-05T14:27:32.156101Z","shell.execute_reply":"2023-02-05T14:27:32.165946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport glob\ndf_train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf_test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndf_submit = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')\nimgs_train = glob.glob('/kaggle/input/rsna-breast-cancer-detection/train_images/*')\nimgs_test = glob.glob('/kaggle/input/rsna-breast-cancer-detection/test_images/*')\nimgs_train_png = glob.glob('/kaggle/input//kaggle/input/rsna-breast-cancer-512-pngs/*')\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:27:32.170264Z","iopub.execute_input":"2023-02-05T14:27:32.170952Z","iopub.status.idle":"2023-02-05T14:27:32.575612Z","shell.execute_reply.started":"2023-02-05T14:27:32.170916Z","shell.execute_reply":"2023-02-05T14:27:32.574636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = Image.open('/kaggle/input/rsna-breast-cancer-512-pngs/9989_398038886.png')\nplt.imshow(image, cmap='bone')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:27:32.576992Z","iopub.execute_input":"2023-02-05T14:27:32.57757Z","iopub.status.idle":"2023-02-05T14:27:32.819713Z","shell.execute_reply.started":"2023-02-05T14:27:32.577532Z","shell.execute_reply":"2023-02-05T14:27:32.818653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_shuffle = df_train.iloc[np.random.permutation(len(df_train))].reset_index(drop=True)\ndf_shuffle","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:27:32.821047Z","iopub.execute_input":"2023-02-05T14:27:32.824447Z","iopub.status.idle":"2023-02-05T14:27:32.870424Z","shell.execute_reply.started":"2023-02-05T14:27:32.824408Z","shell.execute_reply":"2023-02-05T14:27:32.869309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training images:\nif DEBUG:\n    training_path = []\n    testing_path = []\n    for idx in range(df_train.shape[0]):\n        patient_id = df_train.iloc[idx].patient_id\n        image_id = df_train.iloc[idx].image_id\n        training_path.append('/kaggle/input/rsna-breast-cancer-512-pngs/{}_{}.png'.format(patient_id, image_id))\n        testing_path.append('/kaggle/input/rsna-breast-cancer-detection/train_images/{}/{}.dcm'.format(patient_id, image_id))\nelse:\n    \n    training_path_negative = []\n    y_train_negative = []\n    training_path_positive = []\n    y_train_positive = []\n    \n    training_path = []\n    y_train = []\n\n    choice_laterality = 'L'\n   \n    count = 0\n\n    for idx in range(df_shuffle.shape[0]):\n        patient_id = df_shuffle.iloc[idx].patient_id\n        image_id = df_shuffle.iloc[idx].image_id\n        cancer_val = df_shuffle.iloc[idx].cancer\n        laterality = df_shuffle.iloc[idx].laterality\n        view = df_shuffle.iloc[idx]['view']\n        if TOGGLE == True:\n            if cancer_val == 1: \n                training_path.append('/kaggle/input/rsna-breast-cancer-512-pngs/{}_{}.png'.format(patient_id, image_id))\n                y_train.append(df_shuffle.iloc[idx].cancer)\n            elif laterality == choice_laterality:\n                training_path.append('/kaggle/input/rsna-breast-cancer-512-pngs/{}_{}.png'.format(patient_id, image_id))\n                y_train.append(df_shuffle.iloc[idx].cancer)\n                if choice_laterality == 'L':\n                    choice_laterality = 'R'   \n                else:\n                    choice_laterality = 'L'\n        else:\n            if cancer_val == 1: \n                training_path_positive.append('/kaggle/input/rsna-breast-cancer-512-pngs/{}_{}.png'.format(patient_id, image_id))\n                y_train_positive.append(df_shuffle.iloc[idx].cancer)\n            elif count < FETCH:\n                count += 1\n                training_path_negative.append('/kaggle/input/rsna-breast-cancer-512-pngs/{}_{}.png'.format(patient_id, image_id))\n                y_train_negative.append(df_shuffle.iloc[idx].cancer)\n        '''\n        else:\n            if cancer_val == 1: \n                training_path.append('/kaggle/input/rsna-breast-cancer-512-pngs/{}_{}.png'.format(patient_id, image_id))\n                y_train.append(df_shuffle.iloc[idx].cancer)\n            elif count < FETCH:\n                count += 1\n                training_path.append('/kaggle/input/rsna-breast-cancer-512-pngs/{}_{}.png'.format(patient_id, image_id))\n                y_train.append(df_shuffle.iloc[idx].cancer) \n        '''\n                \n    # combined = list(zip(training_path, y_train))\n    # random.shuffle(combined)\n    # training_path, y_train = zip(*combined)\n    # training_path = list(training_path)\n    # y_train = list(y_train)\n    \n    # Copy positive case for 5 time (Kind of Data Augumentation)\n    training_path_positive = training_path_positive * 5\n    y_train_positive = y_train_positive * 5\n    \n    # Testing images:\n    testing_path = []\n\n    for idx in range(df_test.shape[0]):\n        patient_id = df_test.iloc[idx].patient_id\n        image_id = df_test.iloc[idx].image_id\n        testing_path.append('/kaggle/input/rsna-breast-cancer-detection/test_images/{}/{}.dcm'.format(patient_id, image_id))\n    \n    # print('Training: {}, Testing: {}'.format(len(training_path), len(testing_path)))\n    print('Training: {}, Testing: {}'.format(len(training_path_positive + training_path_negative), len(testing_path)))","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:27:32.872091Z","iopub.execute_input":"2023-02-05T14:27:32.872678Z","iopub.status.idle":"2023-02-05T14:28:01.512054Z","shell.execute_reply.started":"2023-02-05T14:27:32.872641Z","shell.execute_reply":"2023-02-05T14:28:01.510895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_dicom_to_j2k(filepath, save_folder=J2K_FOLDER):\n    patient_id = filepath.split('/')[-2]\n    image_id = filepath.split('/')[-1].split('.')[0]\n    dcmfile = pydicom.dcmread(filepath)\n    \n    if dcmfile.file_meta.TransferSyntaxUID == '1.2.840.10008.1.2.4.90':\n        with open(filepath, 'rb') as fp:\n            raw = DicomBytesIO(fp.read())\n            ds = pydicom.dcmread(raw)\n        offset = ds.PixelData.find(b\"\\x00\\x00\\x00\\x0C\")  #<---- the jpeg2000 header info we're looking for\n        hackedbitstream = bytearray()\n        hackedbitstream.extend(ds.PixelData[offset:])\n        with open(save_folder + f\"{patient_id}_{image_id}.jp2\", \"wb\") as binary_file:\n            binary_file.write(hackedbitstream)\n            \n@pipeline_def\ndef j2k_decode_pipeline(j2kfiles):\n    jpegs, _ = fn.readers.file(files=j2kfiles)\n    images = fn.experimental.decoders.image(jpegs, device='mixed', output_type=types.ANY_DATA, dtype=DALIDataType.UINT16)\n    return images","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:01.516195Z","iopub.execute_input":"2023-02-05T14:28:01.516487Z","iopub.status.idle":"2023-02-05T14:28:01.524024Z","shell.execute_reply.started":"2023-02-05T14:28:01.516461Z","shell.execute_reply":"2023-02-05T14:28:01.522946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(testing_path) > 100:\n    N_CHUNKS = 10\nelse:\n    N_CHUNKS = 1\n    \nif DEBUG:\n    CHUNKS = np.array_split(testing_path[:10], N_CHUNKS)\n    # CHUNKS = np.array_split(testing_path[:5000], N_CHUNKS)\nelse:\n    CHUNKS = np.array_split(testing_path, N_CHUNKS)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:01.525468Z","iopub.execute_input":"2023-02-05T14:28:01.526045Z","iopub.status.idle":"2023-02-05T14:28:01.560418Z","shell.execute_reply.started":"2023-02-05T14:28:01.52601Z","shell.execute_reply":"2023-02-05T14:28:01.559311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tracemalloc","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:01.564241Z","iopub.execute_input":"2023-02-05T14:28:01.564563Z","iopub.status.idle":"2023-02-05T14:28:01.57417Z","shell.execute_reply.started":"2023-02-05T14:28:01.564536Z","shell.execute_reply":"2023-02-05T14:28:01.573044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tracemalloc.start()\n\nfor chunk in tqdm.tqdm(CHUNKS):\n    \n    os.makedirs(J2K_FOLDER, exist_ok=True)\n    os.makedirs(SAVE_FOLDER, exist_ok=True)\n\n    \n    pool = Pool(2, maxtasksperchild=2)\n    pool.map(convert_dicom_to_j2k, chunk)\n    pool.close()\n    pool.join()\n    \n    j2kfiles = glob.glob(J2K_FOLDER + \"*.jp2\")\n\n    if not len(j2kfiles):\n        print('None j2kfiles')\n    \n\n    pipe = j2k_decode_pipeline(j2kfiles, batch_size=1, num_threads=2, device_id=0, debug=True)\n    pipe.build()\n\n    for i, f in enumerate(j2kfiles):\n        try:\n            patient_id, image_id = f.split('/')[-1][:-4].split('_')\n            if DEBUG:\n                dicom = pydicom.dcmread('/kaggle/input/rsna-breast-cancer-detection/train_images/' + f\"{patient_id}/{image_id}.dcm\")\n            else:\n                dicom = pydicom.dcmread('/kaggle/input/rsna-breast-cancer-detection/test_images/' + f\"{patient_id}/{image_id}.dcm\")\n\n            out = pipe.run()\n\n            # Dali -> Torch\n            img = out[0][0]\n            imgCpu = np.array(img.as_cpu())\n            imgCpu = (imgCpu - imgCpu.min()) / (imgCpu.max() - imgCpu.min())\n        \n            if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n                imgCpu = 1 - imgCpu\n            \n            # imgCpu = np.resize(imgCpu, (IMAGE_SIZE, IMAGE_SIZE)) # Wrong method\n            imgCpu = cv2.resize(imgCpu, dsize=(512, 512))\n            imgCpu = (imgCpu * 255).astype(np.uint8)\n\n            cv2.imwrite(SAVE_FOLDER + f\"{patient_id}_{image_id}.png\", imgCpu)\n        except:\n            continue\n    # import time\n    # time.sleep(5)\n    shutil.rmtree(J2K_FOLDER)\n    gc.collect()\n\n#del pipe\n","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:01.575997Z","iopub.execute_input":"2023-02-05T14:28:01.576464Z","iopub.status.idle":"2023-02-05T14:28:02.746428Z","shell.execute_reply.started":"2023-02-05T14:28:01.576426Z","shell.execute_reply":"2023-02-05T14:28:02.745213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicomsdl_to_numpy_image(dicom, index=0):\n    info = dicom.getPixelDataInfo()\n    dtype = info['dtype']\n    if info['SamplesPerPixel'] != 1:\n        raise RuntimeError('SamplesPerPixel != 1')\n    else:\n        shape = [info['Rows'], info['Cols']]\n    outarr = np.empty(shape, dtype=dtype)\n    dicom.copyFrameData(index, outarr)\n    return outarr\n\ndef load_img_dicomsdl(f):\n    return dicomsdl_to_numpy_image(dicomsdl.open(f))","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:02.749551Z","iopub.execute_input":"2023-02-05T14:28:02.7499Z","iopub.status.idle":"2023-02-05T14:28:02.759823Z","shell.execute_reply.started":"2023-02-05T14:28:02.749867Z","shell.execute_reply":"2023-02-05T14:28:02.757943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(filepath):\n    os.makedirs(SAVE_FOLDER, exist_ok=True)\n    patient_id = filepath.split('/')[-2]\n    image_id = filepath.split('/')[-1].split('.')[0]\n    dicom = pydicom.dcmread(filepath)\n\n    if dicom.file_meta.TransferSyntaxUID == '1.2.840.10008.1.2.4.90':  # ALREADY PROCESSED\n        return\n    \n    try:\n        img = load_img_dicomsdl(filepath)\n    except:\n        img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (IMAGE_SIZE, IMAGE_SIZE))\n\n    cv2.imwrite(SAVE_FOLDER + f\"{patient_id}_{image_id}.png\", (img * 255).astype(np.uint8))","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:02.762469Z","iopub.execute_input":"2023-02-05T14:28:02.763301Z","iopub.status.idle":"2023-02-05T14:28:02.776819Z","shell.execute_reply.started":"2023-02-05T14:28:02.763237Z","shell.execute_reply":"2023-02-05T14:28:02.775656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pool = Pool(2, maxtasksperchild=2)\npool.map(process, testing_path[:10])\npool.close()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:02.778522Z","iopub.execute_input":"2023-02-05T14:28:02.779368Z","iopub.status.idle":"2023-02-05T14:28:02.839879Z","shell.execute_reply.started":"2023-02-05T14:28:02.779284Z","shell.execute_reply":"2023-02-05T14:28:02.837266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing_path_png = glob.glob(SAVE_FOLDER+'*.png')\ntesting_path_png","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:02.842967Z","iopub.execute_input":"2023-02-05T14:28:02.843502Z","iopub.status.idle":"2023-02-05T14:28:02.864046Z","shell.execute_reply.started":"2023-02-05T14:28:02.843449Z","shell.execute_reply":"2023-02-05T14:28:02.862645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @tf.function\ndef preprocessing(filepath, label):\n    raw = tf.io.read_file(filepath)\n    img = tf.image.resize(tf.image.decode_png(raw, channels=1), [IMAGE_SIZE, IMAGE_SIZE])\n    img = tf.cast(img, tf.float32)\n    img = tf.image.resize(img, [128, 128])\n    # img = tf.stack((img, )*3, axis=-1)\n    img = tf.image.grayscale_to_rgb(img)\n    img = tf.image.random_flip_left_right(img)\n    img = tf.image.random_flip_up_down(img)\n    # img = tf.image.random_crop(img, size=(CROP_IMAGE_SIZE, CROP_IMAGE_SIZE, 3))\n    # img = tf.image.resize(img, [IMAGE_SIZE, IMAGE_SIZE])\n    # img = tf.cond(label == 1, lambda: tf.image.random_flip_left_right(img), lambda: img)\n    # img = tf.cond(label == 1, lambda: tf.image.random_flip_up_down(img), lambda: img)\n    # img = tf.cond(label == 1, lambda: tf.image.random_crop(img, size=(CROP_IMAGE_SIZE, CROP_IMAGE_SIZE, 3)), lambda: img)\n    # img = tf.cond(label == 1, lambda: tf.image.resize(img, [IMAGE_SIZE, IMAGE_SIZE]), lambda: img)\n    img /= 255\n    print(img.shape)\n    \n    return img, label\n\nif DEBUG:\n    labels = np.array([0]*(len(testing_path_png)-1) + [1]*1) # debug\n    dataset_train = tf.data.Dataset.from_tensor_slices((testing_path_png, labels)) # debug\n    dataset_train = dataset_train.map(preprocessing).batch(64).prefetch(64)\nelse:\n    dataset_train_positive = tf.data.Dataset.from_tensor_slices((training_path_positive, y_train_positive))\n    dataset_train_positive = dataset_train_positive.map(preprocessing)\n    dataset_train_negative = tf.data.Dataset.from_tensor_slices((training_path_negative, y_train_negative))\n    dataset_train_negative = dataset_train_negative.map(preprocessing)\n    dataset_train = tf.data.experimental.sample_from_datasets([dataset_train_positive, dataset_train_negative], weights=[0.5, 0.5])\n    \n    # dataset_train = tf.data.Dataset.from_tensor_slices((training_path, y_train))\n    # dataset_train = dataset_train.map(preprocessing)\n    \n    dataset_train = dataset_train.shuffle(buffer_size=512, reshuffle_each_iteration=True).batch(256).prefetch(256)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:02.866211Z","iopub.execute_input":"2023-02-05T14:28:02.866948Z","iopub.status.idle":"2023-02-05T14:28:05.547064Z","shell.execute_reply.started":"2023-02-05T14:28:02.866897Z","shell.execute_reply":"2023-02-05T14:28:05.544923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for a, b in dataset_train.take(1):\n#     print(b)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:05.550877Z","iopub.execute_input":"2023-02-05T14:28:05.551263Z","iopub.status.idle":"2023-02-05T14:28:05.56291Z","shell.execute_reply.started":"2023-02-05T14:28:05.551228Z","shell.execute_reply":"2023-02-05T14:28:05.559813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resnet50 = tf.keras.applications.resnet50.ResNet50(\n    include_top=False,\n    weights='/kaggle/input/tf-keras-pretrained-model-weights/No Top/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5',\n    input_shape=(128, 128, 3)\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:05.565093Z","iopub.execute_input":"2023-02-05T14:28:05.566785Z","iopub.status.idle":"2023-02-05T14:28:08.221848Z","shell.execute_reply.started":"2023-02-05T14:28:05.56669Z","shell.execute_reply":"2023-02-05T14:28:08.220413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resnet50_Model = tf.keras.Sequential([\n        resnet50,\n        tf.keras.layers.GlobalAveragePooling2D(),\n        # tf.keras.layers.Dense(256, activation=\"relu\", name=\"layer1\"),\n        # tf.keras.layers.Dense(1024, activation='relu'),\n        # tf.keras.layers.Dropout(0.3),\n        tf.keras.layers.Dense(2, 'softmax')\n    ],\n    name = 'ResNet50')\nresnet50_Model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:08.227016Z","iopub.execute_input":"2023-02-05T14:28:08.227878Z","iopub.status.idle":"2023-02-05T14:28:08.85627Z","shell.execute_reply.started":"2023-02-05T14:28:08.227825Z","shell.execute_reply":"2023-02-05T14:28:08.854886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ckpt = tf.train.Checkpoint(epoch=tf.Variable(0), net=resnet50_Model)\n\nmanager = tf.train.CheckpointManager(ckpt, '/ckpts/ResNet50/', max_to_keep=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:08.858328Z","iopub.execute_input":"2023-02-05T14:28:08.859116Z","iopub.status.idle":"2023-02-05T14:28:08.869374Z","shell.execute_reply.started":"2023-02-05T14:28:08.85907Z","shell.execute_reply":"2023-02-05T14:28:08.867829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_addons as tfa\n\n# @tf.function\ndef train_step(model, opt, x, y, m):\n    # focalLoss = tf.keras.losses.BinaryFocalCrossentropy(from_logits=False)\n    # focalLoss = tfa.losses.SigmoidFocalCrossEntropy(from_logits=False, alpha=0.25, gamma=2.0, reduction=tf.keras.losses.Reduction.AUTO)\n    with tf.GradientTape() as tape:\n        prediction = model(x, training=True)\n        y_label = tf.one_hot(y, 2)\n        # y_label = tf.expand_dims(y, axis=-1)\n        class_weight = tf.where(y==1, 0.7, 0.3)\n        loss = tf.keras.losses.BinaryCrossentropy(from_logits=False)(y_label, prediction, sample_weight=class_weight)\n       \n        # loss = focalLoss(y_label, prediction)\n        gradient = tape.gradient(loss, model.trainable_variables)\n        opt.apply_gradients(zip(gradient, model.trainable_variables))\n        # y_pred = tf.where(prediction > 0.5, 1, 0)\n        y_pred = tf.math.argmax(prediction, axis=-1)\n        \n    m.update_state(y_pred, y)\n\n    return loss","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:08.871656Z","iopub.execute_input":"2023-02-05T14:28:08.873037Z","iopub.status.idle":"2023-02-05T14:28:09.009965Z","shell.execute_reply.started":"2023-02-05T14:28:08.873003Z","shell.execute_reply":"2023-02-05T14:28:09.008713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not SUBMIT:\n    with open('lossRecord.txt', 'w') as f:\n        f.write('')","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:09.012125Z","iopub.execute_input":"2023-02-05T14:28:09.012627Z","iopub.status.idle":"2023-02-05T14:28:09.020319Z","shell.execute_reply.started":"2023-02-05T14:28:09.01258Z","shell.execute_reply":"2023-02-05T14:28:09.017692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:09.028802Z","iopub.execute_input":"2023-02-05T14:28:09.029193Z","iopub.status.idle":"2023-02-05T14:28:09.253101Z","shell.execute_reply.started":"2023-02-05T14:28:09.029136Z","shell.execute_reply":"2023-02-05T14:28:09.251694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(learning_rate=0.0001)\nBACC = tf.keras.metrics.BinaryAccuracy()\n\nfor epoch in range(EPOCH):\n    ckpt.epoch.assign_add(1)\n    BACC.reset_state()\n    print('---------- Epoch {} ----------'.format(epoch))\n    lossList = []\n    for x, y in tqdm.tqdm(dataset_train):\n        loss = train_step(resnet50_Model, opt, x, y, BACC)\n        lossList.append(np.mean(loss.numpy()))\n        clear_output()\n    print('Loss: {}, Acc: {}'.format(np.mean(np.array(lossList)), BACC.result().numpy()))\n    if not SUBMIT:\n        with open('lossRecord.txt', 'a') as f:\n            f.write('Epoch: {} Loss: {}, Acc: {}'.format(epoch, np.mean(np.array(lossList)), BACC.result().numpy()))\n            f.write('\\n')\n    \n    # if (epoch) % 5 == 0:\n    #     save_path = manager.save()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:28:09.254962Z","iopub.execute_input":"2023-02-05T14:28:09.25606Z","iopub.status.idle":"2023-02-05T14:29:18.156251Z","shell.execute_reply.started":"2023-02-05T14:28:09.256003Z","shell.execute_reply":"2023-02-05T14:29:18.154019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing_test(filepath):\n    # patient_id, image_id = filepath.split('/')[-1][:-4].split('_')\n    raw = tf.io.read_file(filepath)\n    img = tf.image.resize(tf.image.decode_png(raw, channels=1), [IMAGE_SIZE, IMAGE_SIZE])\n    img = tf.cast(img, tf.float32)\n    # img = tf.reshape(img, [IMAGE_SIZE, IMAGE_SIZE, 1])\n    img = tf.image.resize(img, [128, 128])\n    # img = tf.reshape(img, [IMAGE_SIZE, IMAGE_SIZE])\n    # img = tf.stack((img, )*3, axis=-1)\n    img = tf.image.grayscale_to_rgb(img)\n    img /= 255\n    return img, filepath","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:29:18.158564Z","iopub.execute_input":"2023-02-05T14:29:18.159378Z","iopub.status.idle":"2023-02-05T14:29:18.168846Z","shell.execute_reply.started":"2023-02-05T14:29:18.159326Z","shell.execute_reply":"2023-02-05T14:29:18.167519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_test = tf.data.Dataset.from_tensor_slices(testing_path_png) # Need Fix\ndataset_test = dataset_test.map(preprocessing_test).batch(32).prefetch(32) # Need Fix","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:29:18.17101Z","iopub.execute_input":"2023-02-05T14:29:18.171539Z","iopub.status.idle":"2023-02-05T14:29:18.292404Z","shell.execute_reply.started":"2023-02-05T14:29:18.171483Z","shell.execute_reply":"2023-02-05T14:29:18.291191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if DEBUG:\n    df_train['cancer_new'] = 0.0\n    df_train.head()\nelse:\n    df_test['cancer'] = 0.0\n    df_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:29:18.294269Z","iopub.execute_input":"2023-02-05T14:29:18.294648Z","iopub.status.idle":"2023-02-05T14:29:18.305579Z","shell.execute_reply.started":"2023-02-05T14:29:18.294616Z","shell.execute_reply":"2023-02-05T14:29:18.303459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_test_preds = []\npatient_id_tests = []\nimage_id_tests = []\n\nif DEBUG:\n    for imgs, filepath in dataset_test:\n        predictions = resnet50_Model(imgs, training=False)\n        for img_path, preds in zip(filepath, predictions):\n            patient_id, image_id = img_path.numpy().decode('ascii').split('/')[-1][:-4].split('_')\n            print(patient_id, image_id)\n            idx = df_train[(df_train['patient_id'] == int(patient_id)) & (df_train['image_id'] == int(image_id))].index\n            df_train['cancer_new'].iloc[idx] = preds.numpy()[1]\nelse:\n    for imgs, filepath in dataset_test:\n        predictions = resnet50_Model(imgs, training=False)\n        for img_path, preds in zip(filepath, predictions):\n            patient_id, image_id = img_path.numpy().decode('ascii').split('/')[-1][:-4].split('_')\n            print(patient_id, image_id)\n            idx = df_test[(df_test['patient_id'] == int(patient_id)) & (df_test['image_id'] == int(image_id))].index\n            df_test['cancer'].iloc[idx] = preds.numpy()[1]\n","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:29:18.307583Z","iopub.execute_input":"2023-02-05T14:29:18.30826Z","iopub.status.idle":"2023-02-05T14:29:18.483571Z","shell.execute_reply.started":"2023-02-05T14:29:18.308212Z","shell.execute_reply":"2023-02-05T14:29:18.481477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if DEBUG:\n    df_train.head()\nelse:\n    df_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:29:18.48563Z","iopub.execute_input":"2023-02-05T14:29:18.485964Z","iopub.status.idle":"2023-02-05T14:29:18.494029Z","shell.execute_reply.started":"2023-02-05T14:29:18.485932Z","shell.execute_reply":"2023-02-05T14:29:18.49241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if DEBUG:\n    df_train['prediction_id'] = df_train['patient_id'].astype(str) + \"_\" + df_train['laterality']\n    sub = df_train[['prediction_id', 'cancer_new']].groupby(\"prediction_id\").mean().reset_index()\n    sub.to_csv('/kaggle/working/submission.csv', index=False)\nelse:\n    sub = df_test[['prediction_id', 'cancer']].groupby(\"prediction_id\").mean().reset_index()\n    sub.to_csv('/kaggle/working/submission.csv', index=False)\n    \nif not SUBMIT:\n    print(sub.head())","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:29:18.496283Z","iopub.execute_input":"2023-02-05T14:29:18.497285Z","iopub.status.idle":"2023-02-05T14:29:18.521057Z","shell.execute_reply.started":"2023-02-05T14:29:18.497239Z","shell.execute_reply":"2023-02-05T14:29:18.519645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not SUBMIT:\n    for x, y in tqdm.tqdm(dataset_train):\n        predictions = resnet50_Model(x, training=False)\n        for a, b in zip(y, predictions):\n            print(a, b)\n        break","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:29:18.522689Z","iopub.execute_input":"2023-02-05T14:29:18.524447Z","iopub.status.idle":"2023-02-05T14:29:18.532421Z","shell.execute_reply.started":"2023-02-05T14:29:18.524398Z","shell.execute_reply":"2023-02-05T14:29:18.530938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r save_testing","metadata":{"execution":{"iopub.status.busy":"2023-02-05T14:29:18.534395Z","iopub.execute_input":"2023-02-05T14:29:18.534795Z","iopub.status.idle":"2023-02-05T14:29:19.533413Z","shell.execute_reply.started":"2023-02-05T14:29:18.534761Z","shell.execute_reply":"2023-02-05T14:29:19.531719Z"},"trusted":true},"execution_count":null,"outputs":[]}]}