{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Hi, everyone! This is my first baseline in this competition. In this notebook I used **ResNet50** as the backbone. It got a 0.05 score, which means there are still many works to do.<br/><br/>\nI found little notebook that uses Tensorflow in this competition, so I decided to make my notebook public. I hope it can help someone who has decided to make their own baseline in Tensorflow.<br/><br/>\nGoodluck guys!","metadata":{}},{"cell_type":"code","source":"try:\n    import pylibjpeg\nexcept:\n    !rm -rf /root/.cache/torch/hub/checkpoints/\n    !mkdir -p /root/.cache/torch/hub/checkpoints/\n    !pip install /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}\n    !pip install /kaggle/input/rsna-2022-whl/{torch-1.12.1-cp37-cp37m-manylinux1_x86_64.whl,torchvision-0.13.1-cp37-cp37m-manylinux1_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:43:18.791463Z","iopub.execute_input":"2023-01-30T15:43:18.792382Z","iopub.status.idle":"2023-01-30T15:45:11.081737Z","shell.execute_reply.started":"2023-01-30T15:43:18.792274Z","shell.execute_reply":"2023-01-30T15:45:11.080505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport cv2\nimport os\nimport gdcm\nimport glob\nimport seaborn as sns\nimport pydicom\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.layers import Layer","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:11.084539Z","iopub.execute_input":"2023-01-30T15:45:11.084963Z","iopub.status.idle":"2023-01-30T15:45:17.4347Z","shell.execute_reply.started":"2023-01-30T15:45:11.08492Z","shell.execute_reply":"2023-01-30T15:45:17.433593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hypers","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = [1024, 512]\nBATCH_SIZE = 4\nAUTOTUNE = tf.data.experimental.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.436159Z","iopub.execute_input":"2023-01-30T15:45:17.436879Z","iopub.status.idle":"2023-01-30T15:45:17.442389Z","shell.execute_reply.started":"2023-01-30T15:45:17.436839Z","shell.execute_reply":"2023-01-30T15:45:17.441336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set up GPU","metadata":{}},{"cell_type":"code","source":"os.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0\" #0,1,2,3 for four gpu\n\n# VERSION FOR SAVING/LOADING MODEL WEIGHTS\n# THIS SHOULD MATCH THE MODEL IN LOAD_MODEL_FROM\nVER=14 \nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\nif os.environ[\"CUDA_VISIBLE_DEVICES\"].count(',') == 0:\n    strategy = tf.distribute.get_strategy()\n    print('single strategy')\nelse:\n    strategy = tf.distribute.MirroredStrategy()\n    print('multiple strategy')\ntf.config.optimizer.set_experimental_options({\"auto_mixed_precision\": True})\nprint('Mixed precision enabled')\n\nAUTO = tf.data.experimental.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.45203Z","iopub.execute_input":"2023-01-30T15:45:17.452411Z","iopub.status.idle":"2023-01-30T15:45:17.475375Z","shell.execute_reply.started":"2023-01-30T15:45:17.452383Z","shell.execute_reply":"2023-01-30T15:45:17.474112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build model","metadata":{}},{"cell_type":"code","source":"#with strategy.scope():\n    #encoder = keras.models.load_model('/kaggle/input/rsna-efficientnetb7-weights/512_effnetb4')","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.476677Z","iopub.execute_input":"2023-01-30T15:45:17.477207Z","iopub.status.idle":"2023-01-30T15:45:17.482189Z","shell.execute_reply.started":"2023-01-30T15:45:17.47717Z","shell.execute_reply":"2023-01-30T15:45:17.481044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    x = encoder.output\n    x = keras.layers.GlobalAveragePooling2D()(x)\n    #x = keras.layers.BatchNormalization()(x)\n    x = keras.layers.Dense(32, activation = 'silu')(x)\n    #x = keras.layers.Dropout(0.3)(x)\n    outputs = keras.layers.Dense(1, activation = 'sigmoid')(x)\n    model = tf.keras.models.Model(inputs = encoder.inputs, outputs = outputs)\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.48377Z","iopub.execute_input":"2023-01-30T15:45:17.484534Z","iopub.status.idle":"2023-01-30T15:45:17.492574Z","shell.execute_reply.started":"2023-01-30T15:45:17.484494Z","shell.execute_reply":"2023-01-30T15:45:17.491718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#with strategy.scope():\n    #model = build_model()\n    #model.compile(#optimizer = tfa.optimizers.AdamW(learning_rate=3e-4, weight_decay = 1e-4),\n                   #optimizer=keras.optimizers.Adam(learning_rate = 3e-4),\n                   #loss=tf.keras.losses.BinaryCrossentropy(from_logits = False),\n                   #metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.493724Z","iopub.execute_input":"2023-01-30T15:45:17.494948Z","iopub.status.idle":"2023-01-30T15:45:17.507434Z","shell.execute_reply.started":"2023-01-30T15:45:17.494909Z","shell.execute_reply":"2023-01-30T15:45:17.506298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.load_weights('/kaggle/input/rsna-efficientnetb7-weights/512_effnetb4_weights_8.h5')","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.510834Z","iopub.execute_input":"2023-01-30T15:45:17.511107Z","iopub.status.idle":"2023-01-30T15:45:17.518263Z","shell.execute_reply.started":"2023-01-30T15:45:17.511066Z","shell.execute_reply":"2023-01-30T15:45:17.51727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.save('/kaggle/working/eff_base_model')","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.521017Z","iopub.execute_input":"2023-01-30T15:45:17.521797Z","iopub.status.idle":"2023-01-30T15:45:17.529045Z","shell.execute_reply.started":"2023-01-30T15:45:17.521762Z","shell.execute_reply":"2023-01-30T15:45:17.527924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class pFBeta(tf.keras.metrics.Metric):\n    \"\"\"Compute overall probabilistic F-beta score.\"\"\"\n    def __init__(self, beta=1, epsilon=1e-5, name='pF1', **kwargs):\n        super().__init__(name=name, **kwargs)\n        self.beta = beta\n        self.epsilon = epsilon\n        self.pos = self.add_weight(name='pos', initializer='zeros')\n        self.ctp = self.add_weight(name='ctp', initializer='zeros')\n        self.cfp = self.add_weight(name='cfp', initializer='zeros')\n\n    def update_state(self, y_true, y_pred, sample_weight=None):\n        y_true = tf.cast(y_true, tf.float32)\n        y_pred = tf.clip_by_value(y_pred, 0, 1)\n        pos = tf.reduce_sum(y_true)\n        ctp = tf.reduce_sum(y_pred[y_true==1])\n        cfp = tf.reduce_sum(y_pred[y_true==0])\n        self.pos.assign_add(pos)\n        self.ctp.assign_add(ctp)\n        self.cfp.assign_add(cfp)\n\n    def result(self):\n        beta_squared = self.beta * self.beta\n        c_precision = self.ctp / (self.ctp + self.cfp + self.epsilon)\n        c_recall = self.ctp / (self.pos + self.epsilon)\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return tf.cond(c_precision > 0 and c_recall > 0, lambda: result, lambda: 0.0)\n\npFBeta.__name__='pF1'\n    \ndef pfbeta_tf(labels, preds, beta=1):\n    eps = 1e-5\n    preds = tf.clip_by_value(preds, 0, 1)\n    y_true_count = tf.reduce_sum(labels)\n    ctp = tf.reduce_sum(preds[labels==1])\n    cfp = tf.reduce_sum(preds[labels==0])\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp + eps)\n    c_recall = ctp / (y_true_count + eps)\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall + eps)\n        return result\n    else:\n        return tf.constant(0.0, dtype=tf.float32)\n    \ndef pfbeta_thr(labels, preds):\n    thrs = tf.range(0, 1, 0.05)\n    best_score = tf.constant(0, dtype=tf.float32)\n    for thr in thrs:\n        score = pfbeta_tf(labels, tf.cast(preds>thr, tf.float32))\n        best_score = tf.cond(score > best_score, lambda: score, lambda: best_score)\n    return best_score\n\npfbeta_thr.__name__='pF1_thr'\n\nclass ContrastiveLoss(Layer):\n    def __init__(self, **kwargs):\n        super(ContrastiveLoss, self).__init__(**kwargs)\n        \n    def NTXentLoss(self, pos_sims_1, pos_sims_2, neg_sims_1, neg_sims_2):\n        l1 = -tf.math.log(Softmax(pos_sims_1, neg_sims_1))\n        l2 = -tf.math.log(Softmax(pos_sims_2, neg_sims_2))\n        loss = (1/2*NUM_POS_IN_BATCH)*(l1+l2)\n        return loss\n    \n    def call(self, inputs):\n        features_1, features_2 = inputs\n        pos_sims_1 = []\n        pos_sims_2 = []\n        neg_sims_1 = []\n        neg_sims_2 = []\n        for i in range(NUM_POS_IN_BATCH):\n            pos_pos = []\n            pos_neg = []\n            for j in range(BATCH_SIZE):\n                if j < NUM_POS_IN_BATCH:\n                    pos_pos.append(Similarity(features_1[i], features_2[j]))\n                else:\n                    pos_neg.append(Similarity(features_1[i], features_2[j]))\n            aver_pos_sim = tf.reduce_sum(pos_pos)/NUM_POS_IN_BATCH\n            pos_sims_1.append(aver_pos_sim)\n            neg_sims_1.append(pos_neg)\n    \n        for i in range(NUM_POS_IN_BATCH):\n            pos_pos = []\n            pos_neg = []\n            for j in range(BATCH_SIZE):\n                if j < NUM_POS_IN_BATCH:\n                    pos_pos.append(Similarity(features_1[j], features_2[i]))\n                else:\n                    pos_neg.append(Similarity(features_1[j], features_2[i]))\n            aver_pos_sim = tf.reduce_sum(pos_pos)/NUM_POS_IN_BATCH\n            pos_sims_2.append(aver_pos_sim)\n            neg_sims_2.append(pos_neg)\n    \n        neg_sims_1 = tf.convert_to_tensor(neg_sims_1)\n        pos_sims_1 = tf.reshape(tf.convert_to_tensor(pos_sims_1), [NUM_POS_IN_BATCH, 1])\n        neg_sims_2 = tf.convert_to_tensor(neg_sims_2)\n        pos_sims_2 = tf.reshape(tf.convert_to_tensor(pos_sims_2), [NUM_POS_IN_BATCH, 1])\n        loss = self.NTXentLoss(pos_sims_1, pos_sims_2, neg_sims_1, neg_sims_2)\n        self.add_loss(loss , inputs=True)\n        self.add_metric(loss , aggregation=\"mean\", name=\"contrastive loss\")\n        return loss\n    \nContrastiveLoss.__name__='loss'\n\nclass Augmentation(Layer):\n    def __init__(self, **kwargs):\n        super(Augmentation, self).__init__(**kwargs)\n        \n    def call(self, image):\n        image = tf.image.random_crop(image, [BATCH_SIZE, 512, 512, 3])\n        image = tf.image.random_hue(image, 0.2)\n        image = tf.image.random_saturation(image, 0.70, 1.30)\n        image = tf.image.random_contrast(image, 0.80, 1.20)\n        image = tf.image.random_brightness(image, 0.10)\n        image = tf.image.resize(image,[IMG_SIZE[0],IMG_SIZE[1]])\n        return image\n    \nAugmentation.__name__='augment'\n    \nclass ProjectionHead(Layer):\n    def __init__(self, **kwargs):\n        super(ProjectionHead, self).__init__(**kwargs)\n        self.dense_1 = layers.Dense(512, activation = 'relu')\n        self.dense_2 = layers.Dense(64)\n        \n    def call(self, inputs):\n        x = self.dense_1(inputs)\n        x = self.dense_2(x)\n        return x\n    \nProjectionHead.__name__='head'","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.532354Z","iopub.execute_input":"2023-01-30T15:45:17.532908Z","iopub.status.idle":"2023-01-30T15:45:17.56194Z","shell.execute_reply.started":"2023-01-30T15:45:17.532875Z","shell.execute_reply":"2023-01-30T15:45:17.56095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_objects = {'pF1': pFBeta, 'pF1_thr': pfbeta_thr, 'augment': Augmentation, 'head': ProjectionHead, 'loss': ContrastiveLoss}","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.566913Z","iopub.execute_input":"2023-01-30T15:45:17.567859Z","iopub.status.idle":"2023-01-30T15:45:17.577791Z","shell.execute_reply.started":"2023-01-30T15:45:17.567831Z","shell.execute_reply":"2023-01-30T15:45:17.576602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_flip_model(model_path):\n    base = tf.keras.models.load_model(model_path, compile=False)\n    inp1 = base.input\n    inp2 = tf.keras.layers.Lambda(lambda x: tf.image.flip_left_right(x), name='flip')(inp1)\n    out1 = base(inp1)\n    out2 = base(inp2)\n    out = tf.keras.layers.Average(name='ensemble')([out1, out2])\n    model = tf.keras.models.Model(inp1, out)\n    return model\n\nwith strategy.scope():\n    model = keras.models.load_model('/kaggle/input/rsna-efficientnetb7-weights/fine_tune_model_1.h5', custom_objects=custom_objects)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:17.579463Z","iopub.execute_input":"2023-01-30T15:45:17.58026Z","iopub.status.idle":"2023-01-30T15:45:24.842031Z","shell.execute_reply.started":"2023-01-30T15:45:17.580209Z","shell.execute_reply":"2023-01-30T15:45:24.840958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Process image & generate TF dataset","metadata":{}},{"cell_type":"code","source":"test_images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\")\n\nimage_dir = '/kaggle/tmp/test_images'\nos.makedirs(image_dir, exist_ok=True)\n\nimage_size = IMG_SIZE\ndcm_dir  = '/kaggle/input/rsna-breast-cancer-detection/test_images'\ntest_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\ndef dicom_to_png(dcm_file, image_size=image_size, image_dir=''):\n    patient_id = dcm_file.split('/')[-2]\n    image_id   = dcm_file.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(dcm_file)\n    img = dicom.pixel_array\n    img = (img - img.min()) / (img.max() - img.min())\n    if dicom.PhotometricInterpretation == 'MONOCHROME1':\n        img = 1 - img\n\n    img = cv2.resize(img, (image_size[0], image_size[1]), interpolation=cv2.INTER_LINEAR)\n    img = (img * 255).astype(np.uint8)\n    cv2.imwrite(image_dir +'/'+ f'{patient_id}_{image_id}.png', img)\n\n\ndcm_file = dcm_dir + '/' + test_df.patient_id.astype(str) + '/'  + test_df.image_id.astype(str) + '.dcm'\nParallel(n_jobs=2)(\n        delayed(dicom_to_png)(f, image_size=image_size, image_dir=image_dir)\n        for f in tqdm(dcm_file))\n\ndef load_image(image_path):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels = 3)\n    img = tf.image.resize(img, IMG_SIZE)\n    img = tf.cast(img, dtype = tf.float32)\n    img = img/255.0\n    return img\n\ntest_image_paths = []\ntest_dir = os.listdir('/kaggle/tmp/test_images')\nfor i in range(len(test_dir)):\n    img_path = '/kaggle/tmp/test_images' + '/' + test_dir[i]\n    test_image_paths.append(img_path)\ntest_img_path_ds = tf.data.Dataset.from_tensor_slices(test_image_paths)\ntest_ds = test_img_path_ds.map(load_image, num_parallel_calls = AUTOTUNE)\ntest_ds = test_ds.batch(BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:24.843942Z","iopub.execute_input":"2023-01-30T15:45:24.844313Z","iopub.status.idle":"2023-01-30T15:45:29.432958Z","shell.execute_reply.started":"2023-01-30T15:45:24.844277Z","shell.execute_reply":"2023-01-30T15:45:29.431922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## prediction","metadata":{}},{"cell_type":"code","source":"preds = model.predict(test_ds)\npreds[0:10]","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:29.434969Z","iopub.execute_input":"2023-01-30T15:45:29.435405Z","iopub.status.idle":"2023-01-30T15:45:38.598021Z","shell.execute_reply.started":"2023-01-30T15:45:29.435364Z","shell.execute_reply":"2023-01-30T15:45:38.596923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make submission","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ndf['cancer'] = 0\n\nTHRESHOLD = 0.23\n\n#preds = np.mean([prediction], 0)\npreds = (preds > THRESHOLD).astype(int)\ndf[\"cancer\"] = preds","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:38.603303Z","iopub.execute_input":"2023-01-30T15:45:38.606076Z","iopub.status.idle":"2023-01-30T15:45:38.627273Z","shell.execute_reply.started":"2023-01-30T15:45:38.60603Z","shell.execute_reply":"2023-01-30T15:45:38.625968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['prediction_id'] = df['patient_id'].astype(str) + \"_\" + df['laterality']\n\nsub = df[['prediction_id', 'cancer']].groupby(\"prediction_id\").mean().reset_index()\n\nsub.to_csv('/kaggle/working/submission.csv', index=False)\n\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-30T15:45:38.63269Z","iopub.execute_input":"2023-01-30T15:45:38.635391Z","iopub.status.idle":"2023-01-30T15:45:38.708494Z","shell.execute_reply.started":"2023-01-30T15:45:38.635349Z","shell.execute_reply":"2023-01-30T15:45:38.707476Z"},"trusted":true},"execution_count":null,"outputs":[]}]}