{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Please, upvote this notebook if You like it! :)","metadata":{}},{"cell_type":"markdown","source":"________________________________________\nThanks to @ipythonx for his great notebook! A lot of ideas were taken from it. So, upvote his notebook too please!\n\nhttps://www.kaggle.com/code/ipythonx/keras-rsna-breast-cancer-detection-training\n\n(link for upvoting)\n\n_______________________________________","metadata":{}},{"cell_type":"markdown","source":"# Importing libraries\n\nThere is an option to train the model on the GPU and on the TPU. But the competition conditions do not allow submit notebook with TPU\n\nTo use TPU specify device = 'TPU'","metadata":{}},{"cell_type":"code","source":"# !pip install -qU scikit-learn\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport random\nimport os\n\nimport cv2\nimport warnings\nfrom packaging.version import parse\nfrom matplotlib import pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import GroupKFold\n\nDEVICE = 'GPU' # 'GPU', 'TPU'\nwarnings.simplefilter(action=\"ignore\")\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = \"3\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-07T13:43:34.565711Z","iopub.execute_input":"2023-01-07T13:43:34.566031Z","iopub.status.idle":"2023-01-07T13:43:35.622812Z","shell.execute_reply.started":"2023-01-07T13:43:34.565939Z","shell.execute_reply":"2023-01-07T13:43:35.622017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To use TPU, uncomment the line strategy = set_device(accelerator='TPU')","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nimport tensorflow_hub as hub\nimport tensorflow_addons as tfa\n\ndef set_device(accelerator='GPU'):\n    if accelerator.upper() == 'TPU':\n        tpu  = tf.distribute.cluster_resolver.TPUClusterResolver.connect() \n        strategy = tf.distribute.TPUStrategy(tpu)\n        tf.config.optimizer.set_jit(True)\n        keras.mixed_precision.set_global_policy(\"mixed_bfloat16\")\n        tf.config.set_soft_device_placement(True)\n        return strategy\n    \n    if accelerator.upper() == 'GPU':\n        strategy = tf.distribute.MirroredStrategy()\n        physical_devices = tf.config.list_physical_devices(accelerator.upper())\n        tf.config.optimizer.set_jit(True)\n        keras.mixed_precision.set_global_policy(\"mixed_float16\")\n        print('GPUs: ', physical_devices)\n        return strategy\n    \ndef set_tpu(mixed_precision=True):\n    try: \n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect() \n        if mixed_precision:\n            keras.mixed_precision.set_global_policy(\"mixed_bfloat16\") \n        tf.config.set_soft_device_placement(True)\n        strategy = tf.distribute.TPUStrategy(tpu)\n        physical_devices = tf.config.list_logical_devices('TPU')\n        return (strategy, physical_devices)\n    except:\n        False\n        \ndef set_cpu_gpus(mixed_precision=True):\n    try: \n        # printed out the detected devices\n        list_ld = device_lib.list_local_devices()\n        for dev in list_ld: print(dev.name,dev.memory_limit)\n        physical_devices = tf.config.list_physical_devices(\n            'GPU' if len(list_ld) - 1 else 'CPU'\n        )\n        # For GPU devices, set growth memory constraint\n        if 'GPU' in physical_devices[-1]:\n            tf.config.optimizer.set_jit(True)\n            keras.mixed_precision.set_global_policy(\"mixed_float16\")\n            for pd in physical_devices:\n                tf.config.experimental.set_memory_growth(pd, True)\n        strategy = tf.distribute.MirroredStrategy()\n        return (strategy, physical_devices)\n    except: \n        raise ValueError('No Device Detected!')   \n\n\nif DEVICE == 'TPU':\n    strategy, physical_devices = set_tpu(mixed_precision=False)\n#     strategy = set_device(accelerator='TPU')  \n# print('TensorFlow  : ', tf.__version__)\n# print(\"Number of accelerators: \", strategy.num_replicas_in_sync)\n# print(tf.sysconfig.get_build_info()[\"cuda_version\"])\n# print(tf.sysconfig.get_build_info()[\"cudnn_version\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:35.625724Z","iopub.execute_input":"2023-01-07T13:43:35.626338Z","iopub.status.idle":"2023-01-07T13:43:41.177882Z","shell.execute_reply.started":"2023-01-07T13:43:35.626308Z","shell.execute_reply":"2023-01-07T13:43:41.177127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random.seed(101)\nnp.random.seed(101)\ntf.random.set_seed(101)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:41.18059Z","iopub.execute_input":"2023-01-07T13:43:41.181047Z","iopub.status.idle":"2023-01-07T13:43:41.185189Z","shell.execute_reply.started":"2023-01-07T13:43:41.181008Z","shell.execute_reply":"2023-01-07T13:43:41.1845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_plot(tfdata, take_batch=1, figsize=(20, 20)):\n    for images, labels in tfdata.take(take_batch):\n        plt.figure(figsize=figsize)\n        xy = int(np.ceil(images.shape[0] * 0.5))\n\n        for i in range(images.shape[0]):\n            plt.subplot(xy, xy, i + 1)\n            plt.imshow(tf.cast(images[i], dtype=tf.uint8))\n            plt.axis(\"off\")\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:41.187333Z","iopub.execute_input":"2023-01-07T13:43:41.187823Z","iopub.status.idle":"2023-01-07T13:43:41.203526Z","shell.execute_reply.started":"2023-01-07T13:43:41.187774Z","shell.execute_reply":"2023-01-07T13:43:41.202887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Set\n\nTo use TPU, uncomment the lines if DEVICE = 'TPU' ...\n\nYou can try to use different image resolutions by correcting the number in the link","metadata":{}},{"cell_type":"code","source":"if DEVICE == 'TPU':\n    DF_PATH = KaggleDatasets().get_gcs_path('rsna-breast-cancer-detection')\n    IMG_PATH = KaggleDatasets().get_gcs_path('rsna-breast-cancer-512-pngs')\nelse:\n    DF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\n    IMG_PATH = '/kaggle/input/rsna-breast-cancer-1024-pngs/output'\n\n# DF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\n# IMG_PATH = '/kaggle/input/rsna-breast-cancer-512-pngs'\n    \n    \ndf = pd.read_csv(f\"{DF_PATH}/train.csv\")\ndf['img_path'] = df.apply(\n    lambda i: os.path.join(\n        f\"{IMG_PATH}\", str(i['patient_id']) + \"_\" + str(i['image_id']) + '.png'\n    ), axis=1\n)\n\ndisplay(df.head())\nprint(df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:41.204855Z","iopub.execute_input":"2023-01-07T13:43:41.205304Z","iopub.status.idle":"2023-01-07T13:43:42.122092Z","shell.execute_reply.started":"2023-01-07T13:43:41.20527Z","shell.execute_reply":"2023-01-07T13:43:42.120796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.age = df.age.fillna(df.age.median())\ndf.BIRADS = df.BIRADS.fillna(df.BIRADS.mode())\ndf.density = df.density.fillna(df.density.mode())\ndf.density = df.density.fillna('E')\ndf.BIRADS = df.BIRADS.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.123216Z","iopub.execute_input":"2023-01-07T13:43:42.123526Z","iopub.status.idle":"2023-01-07T13:43:42.149232Z","shell.execute_reply.started":"2023-01-07T13:43:42.123486Z","shell.execute_reply":"2023-01-07T13:43:42.148553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = ['laterality', 'view', 'density', 'difficult_negative_case']\ndf = pd.get_dummies(df, columns = cat_cols)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.152037Z","iopub.execute_input":"2023-01-07T13:43:42.152229Z","iopub.status.idle":"2023-01-07T13:43:42.187576Z","shell.execute_reply.started":"2023-01-07T13:43:42.152205Z","shell.execute_reply":"2023-01-07T13:43:42.186818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pr_cols = ['site_id', 'age', 'implant', 'machine_id',\n           'laterality_L', 'laterality_R', \n           'view_AT', 'view_CC', 'view_LM',\n           'view_LMO', 'view_ML', 'view_MLO']","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.188923Z","iopub.execute_input":"2023-01-07T13:43:42.189264Z","iopub.status.idle":"2023-01-07T13:43:42.193868Z","shell.execute_reply.started":"2023-01-07T13:43:42.189227Z","shell.execute_reply":"2023-01-07T13:43:42.193088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.cancer.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.195137Z","iopub.execute_input":"2023-01-07T13:43:42.195886Z","iopub.status.idle":"2023-01-07T13:43:42.208426Z","shell.execute_reply.started":"2023-01-07T13:43:42.195843Z","shell.execute_reply":"2023-01-07T13:43:42.207591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkfold = GroupKFold(n_splits=5)\ndf['fold'] = -1\n\nfor fold, (_, valid_idx) in enumerate(\n    gkfold.split(df, groups=df.patient_id)\n):\n    df.loc[valid_idx, 'fold'] = fold","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.211837Z","iopub.execute_input":"2023-01-07T13:43:42.212041Z","iopub.status.idle":"2023-01-07T13:43:42.269405Z","shell.execute_reply.started":"2023-01-07T13:43:42.212017Z","shell.execute_reply":"2023-01-07T13:43:42.26865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df.query('fold == 1 | fold == 2 | fold == 3 | fold == 4')\nvalid_df = df.query('fold == 0')\n\nprint(train_df.shape, valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.270573Z","iopub.execute_input":"2023-01-07T13:43:42.271238Z","iopub.status.idle":"2023-01-07T13:43:42.300498Z","shell.execute_reply.started":"2023-01-07T13:43:42.271197Z","shell.execute_reply":"2023-01-07T13:43:42.299668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.patient_id.nunique())\nprint(train_df.cancer.value_counts(normalize=True))\n\nprint(valid_df.patient_id.nunique())\nprint(valid_df.cancer.value_counts(normalize=True))","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.301964Z","iopub.execute_input":"2023-01-07T13:43:42.302218Z","iopub.status.idle":"2023-01-07T13:43:42.313595Z","shell.execute_reply.started":"2023-01-07T13:43:42.302184Z","shell.execute_reply":"2023-01-07T13:43:42.312729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Loader","metadata":{}},{"cell_type":"code","source":"# SetAutoTune\n# BATCH_SIZE = 32 * strategy.num_replicas_in_sync\nBATCH_SIZE = 4\nINP_SIZE = (1024, 1024)\n# INP_SIZE = (256, 256)\nAUTOTUNE = tf.data.AUTOTUNE ","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.315127Z","iopub.execute_input":"2023-01-07T13:43:42.315421Z","iopub.status.idle":"2023-01-07T13:43:42.319486Z","shell.execute_reply.started":"2023-01-07T13:43:42.315379Z","shell.execute_reply":"2023-01-07T13:43:42.318786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def build_augmenter(is_labelled):\n#     def augment(inp):\n#         img = inp['input1']\n#         img = tfa.image.rotate(\n#             img, \n#             tf.random.uniform(\n#                 shape=[1], \n#                 minval=0 * (np.pi / 180.0), \n#                 maxval=90 * (np.pi / 180.0) \n#             )\n#         )\n#         inp['input1'] = img\n#         return inp\n    \n#     def augment_with_labels(inp, label):\n#         return augment(inp), label\n    \n#     return augment_with_labels if is_labelled else augment\n\ndef build_decoder(is_labelled):\n    def decode(inp):\n        file_bytes = tf.io.read_file(inp['input1'])\n        img = tf.image.decode_jpeg(file_bytes, channels = 3)\n#         img = tf.cast(img, tf.float16)\n#         img = tf.image.resize(img, (INP_SIZE))\n        img = tf.reshape(img, [*INP_SIZE, 3])\n#         img = tf.image.convert_image_dtype(img, dtype=tf.float16)\n        inp['input1'] = img\n        return inp\n    \n    def decode_with_labels(inp, label):\n        return decode(inp), label\n    \n    return decode_with_labels if is_labelled else decode\n\n\ndef create_dataset(\n    df, \n    batch_size  = 32, \n    is_labelled = False,\n    shuffle     = False\n):\n    decode_fn    = build_decoder(is_labelled)\n    \n    # Create Dataset\n    if is_labelled:\n        dataset = tf.data.Dataset.from_tensor_slices(({'input1': df['img_path'].values, 'input2': df[pr_cols].values}, df['cancer'].values))\n    else:\n        dataset = tf.data.Dataset.from_tensor_slices(({'input1': df['img_path'].values, 'input2': df[pr_cols].values}))\n        \n    dataset = dataset.map(decode_fn, num_parallel_calls = AUTOTUNE)\n    dataset = dataset.shuffle(8 * BATCH_SIZE, reshuffle_each_iteration = True) if shuffle else dataset\n    dataset = dataset.batch(batch_size, drop_remainder=shuffle)\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.320834Z","iopub.execute_input":"2023-01-07T13:43:42.321296Z","iopub.status.idle":"2023-01-07T13:43:42.331766Z","shell.execute_reply.started":"2023-01-07T13:43:42.321259Z","shell.execute_reply":"2023-01-07T13:43:42.331033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_dataset = create_dataset(\n    train_df,\n    batch_size  = BATCH_SIZE, \n    is_labelled = True,\n    shuffle = True\n)\n\nvalid_dataset = create_dataset(\n    valid_df,\n    batch_size  = BATCH_SIZE, \n    is_labelled = True,\n    shuffle = False\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:42.332937Z","iopub.execute_input":"2023-01-07T13:43:42.333193Z","iopub.status.idle":"2023-01-07T13:43:45.15301Z","shell.execute_reply.started":"2023-01-07T13:43:42.333161Z","shell.execute_reply":"2023-01-07T13:43:45.15226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make_plot(training_dataset.unbatch().batch(16), take_batch=1) \n# make_plot(valid_dataset.unbatch().batch(16), take_batch=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:45.154418Z","iopub.execute_input":"2023-01-07T13:43:45.154681Z","iopub.status.idle":"2023-01-07T13:43:45.159256Z","shell.execute_reply.started":"2023-01-07T13:43:45.154646Z","shell.execute_reply":"2023-01-07T13:43:45.158512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling","metadata":{}},{"cell_type":"code","source":"from tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import losses\nfrom tensorflow.keras import metrics\nfrom tensorflow.keras import optimizers\nfrom tensorflow.keras import callbacks","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:45.160513Z","iopub.execute_input":"2023-01-07T13:43:45.161065Z","iopub.status.idle":"2023-01-07T13:43:45.169243Z","shell.execute_reply.started":"2023-01-07T13:43:45.161027Z","shell.execute_reply":"2023-01-07T13:43:45.168472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Learning rate with warm up\n\nBut you can try to run training with a constant value of the learning rate","metadata":{}},{"cell_type":"code","source":"class WarmupLearningRateSchedule(optimizers.schedules.LearningRateSchedule):\n    \"\"\"WarmupLearningRateSchedule a variety of learning rate\n    decay schedules with warm up.\n    \n    Ref. https://gist.github.com/innat/69e8f3500c2418c69b150a0a651f31dc\n    \"\"\"\n\n    def __init__(\n        self,\n        initial_lr,\n        steps_per_epoch=None,\n        lr_decay_type=\"exponential\",\n        decay_factor=0.97,\n        decay_epochs=2.4,\n        total_steps=None,\n        warmup_epochs=5,\n        minimal_lr=0, \n        **kwargs\n    ):\n        super().__init__(**kwargs)\n        self.initial_lr = initial_lr\n        self.steps_per_epoch = steps_per_epoch\n        self.lr_decay_type = lr_decay_type\n        self.decay_factor = decay_factor\n        self.decay_epochs = decay_epochs\n        self.total_steps = total_steps\n        self.warmup_epochs = warmup_epochs\n        self.minimal_lr = minimal_lr\n\n    def __call__(self, step):\n        if self.lr_decay_type == \"exponential\":\n            assert self.steps_per_epoch is not None\n            decay_steps = self.steps_per_epoch * self.decay_epochs\n            lr = optimizers.schedules.ExponentialDecay(\n                self.initial_lr, decay_steps, self.decay_factor, staircase=True\n            )(step)\n            \n        elif self.lr_decay_type == \"cosine\":\n            assert self.total_steps is not None\n            lr = (\n                0.5\n                * self.initial_lr\n                * (1 + tf.cos(np.pi * tf.cast(step, tf.float32) / self.total_steps))\n            )\n\n        elif self.lr_decay_type == \"linear\":\n            assert self.total_steps is not None\n            lr = (1.0 - tf.cast(step, tf.float32) / self.total_steps) * self.initial_lr\n\n        elif self.lr_decay_type == \"constant\":\n            lr = self.initial_lr\n\n        elif self.lr_decay_type == \"cosine_restart\":\n            decay_steps = self.steps_per_epoch * self.decay_epochs\n            lr = tf.keras.experimental.CosineDecayRestarts(\n                self.initial_lr, decay_steps\n            )(step)\n        else:\n            assert False, \"Unknown lr_decay_type : %s\" % self.lr_decay_type\n\n        if self.minimal_lr:\n            lr = tf.math.maximum(lr, self.minimal_lr)\n\n        if self.warmup_epochs:\n            warmup_steps = int(self.warmup_epochs * self.steps_per_epoch)\n            warmup_lr = (\n                self.initial_lr\n                * tf.cast(step, tf.float32)\n                / tf.cast(warmup_steps, tf.float32)\n            )\n            lr = tf.cond(step < warmup_steps, lambda: warmup_lr, lambda: lr)\n\n        return lr\n\n    def get_config(self):\n        return {\n            \"initial_lr\": self.initial_lr,\n            \"steps_per_epoch\": self.steps_per_epoch,\n            \"lr_decay_type\": self.lr_decay_type,\n            \"decay_factor\": self.decay_factor,\n            \"decay_epochs\": self.decay_epochs,\n            \"total_steps\": self.total_steps,\n            \"warmup_epochs\": self.warmup_epochs,\n            \"minimal_lr\": self.minimal_lr,\n        }","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:45.170393Z","iopub.execute_input":"2023-01-07T13:43:45.170852Z","iopub.status.idle":"2023-01-07T13:43:45.185427Z","shell.execute_reply.started":"2023-01-07T13:43:45.170817Z","shell.execute_reply":"2023-01-07T13:43:45.18451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nepochs = 30\nlr_sched = 'cosine' # [or, exponential, cosine, linear, constant]\nlr_base  = 0.016\nlr_min   = 0\nlr_decay_epoch  = 2.4\nlr_warmup_epoch = 2\nlr_decay_factor = 0.97\n\nscaled_lr = lr_base * (BATCH_SIZE / 256.0)\nscaled_lr_min = lr_min * (BATCH_SIZE / 256.0)\n\nnum_training_sample = train_df.shape[0]\ntrain_step = int(np.ceil(num_training_sample / float(BATCH_SIZE)))\ntotal_steps = train_step * epochs\n\nlearning_rate = WarmupLearningRateSchedule(\n    scaled_lr,\n    steps_per_epoch=train_step,\n    decay_epochs=lr_decay_epoch,\n    warmup_epochs=lr_warmup_epoch,\n    decay_factor=lr_decay_factor,\n    lr_decay_type=lr_sched,\n    total_steps=total_steps,\n    minimal_lr=scaled_lr_min,\n)\n\nrng = [i for i in range(total_steps)]\nlr_y = [learning_rate(x) for x in rng]\nplt.figure(figsize=(10, 4))\nplt.plot(rng, lr_y)\nplt.xlabel(\"Iteration\", size=14)\nplt.ylabel(\"Learning Rate\", size=14)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:43:45.186491Z","iopub.execute_input":"2023-01-07T13:43:45.186796Z","iopub.status.idle":"2023-01-07T13:47:51.637655Z","shell.execute_reply.started":"2023-01-07T13:43:45.186763Z","shell.execute_reply":"2023-01-07T13:47:51.636973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_schedule = keras.optimizers.schedules.ExponentialDecay(\n                    initial_learning_rate=0.001,\n                    decay_steps=1000,\n                    decay_rate=0.9)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:51.638987Z","iopub.execute_input":"2023-01-07T13:47:51.639398Z","iopub.status.idle":"2023-01-07T13:47:51.645201Z","shell.execute_reply.started":"2023-01-07T13:47:51.63936Z","shell.execute_reply":"2023-01-07T13:47:51.644469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[**Competition Metrics**](https://www.kaggle.com/code/sohier/probabilistic-f-score)","metadata":{}},{"cell_type":"code","source":"\ndef pfbeta_tf(labels, preds, beta=1):\n    preds = tf.clip_by_value(preds, 0, 1)\n    y_true_count = tf.reduce_sum(labels)\n    ctp = tf.reduce_sum(preds[labels==1])\n    cfp = tf.reduce_sum(preds[labels==0])\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0.0\n    \ndef tf_pfbeta(from_logits=True, beta=1.0, epsilon=1e-07):\n    \n    def pfbeta(y_true, y_pred):\n        y_pred = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(y_pred),\n            lambda: y_pred,\n        )\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n\n        ctp = tf.reduce_sum(y_true * y_pred, axis=-1)\n        cfp = tf.reduce_sum(y_pred, axis=-1) - ctp\n\n        c_precision = ctp / (ctp + cfp)\n        c_recall = ctp / tf.reduce_sum(y_true)\n        \n        def compute_fractions():\n            numerator = c_precision * c_recall\n            denominator = beta**2 * c_precision + c_recall\n            return (1 + beta**2) * tf.math.divide_no_nan(numerator, denominator)\n        \n        return tf.cond(\n            tf.logical_and(\n                tf.greater(c_precision, 0.), tf.greater(c_recall, 0.)\n            ),\n            compute_fractions,\n            lambda: tf.constant(0, dtype=tf.float32)\n        )\n    \n    return pfbeta","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:51.646396Z","iopub.execute_input":"2023-01-07T13:47:51.647718Z","iopub.status.idle":"2023-01-07T13:47:51.659319Z","shell.execute_reply.started":"2023-01-07T13:47:51.647688Z","shell.execute_reply":"2023-01-07T13:47:51.658499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auc_logit():\n    auc_fn = metrics.AUC()\n    \n    def auc(y_true, y_pred):\n        y_pred = tf.nn.sigmoid(y_pred)\n        return auc_fn(y_true, y_pred)\n    \n    return auc","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:51.661725Z","iopub.execute_input":"2023-01-07T13:47:51.661956Z","iopub.status.idle":"2023-01-07T13:47:51.672832Z","shell.execute_reply.started":"2023-01-07T13:47:51.661926Z","shell.execute_reply":"2023-01-07T13:47:51.672115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Weighted Binary Loss**\n\n> A value `pos_weight > 1` decreases the false negative count, hence increasing the recall. Conversely setting `pos_weight < 1` decreases the false positive count and increases the precision. This can be seen from the fact that `pos_weight` is introduced as a multiplicative coefficient for the positive labels term in the loss expression:","metadata":{}},{"cell_type":"code","source":"def weighted_binary_loss(weight):\n    def weighted_loss(labels, logits):\n        loss = tf.nn.weighted_cross_entropy_with_logits(\n            tf.cast(labels, dtype=tf.float32), logits, weight\n        )\n        return loss\n    return weighted_loss","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:51.675202Z","iopub.execute_input":"2023-01-07T13:47:51.675803Z","iopub.status.idle":"2023-01-07T13:47:51.683849Z","shell.execute_reply.started":"2023-01-07T13:47:51.675776Z","shell.execute_reply":"2023-01-07T13:47:51.68321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if parse(tf.__version__) < parse('2.6.2'):\n    from tensorflow.keras.layers.experimental.preprocessing import Rescaling\n    from tensorflow.keras.layers.experimental.preprocessing import Resizing\n    from tensorflow.keras.layers.experimental.preprocessing import RandomCrop\n    from tensorflow.keras.layers.experimental.preprocessing import RandomFlip\n    from tensorflow.keras.layers.experimental.preprocessing import RandomZoom\n    from tensorflow.keras.layers.experimental.preprocessing import RandomRotation\nelse:\n    from tensorflow.keras.layers import Resizing\n    from tensorflow.keras.layers import Rescaling\n    from tensorflow.keras.layers import RandomCrop\n    from tensorflow.keras.layers import RandomFlip\n    from tensorflow.keras.layers import RandomZoom\n    from tensorflow.keras.layers import RandomRotation","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:51.685289Z","iopub.execute_input":"2023-01-07T13:47:51.685905Z","iopub.status.idle":"2023-01-07T13:47:51.694267Z","shell.execute_reply.started":"2023-01-07T13:47:51.68587Z","shell.execute_reply":"2023-01-07T13:47:51.693442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocessing\ndata_preprocessing = keras.Sequential(\n    [\n        Resizing(\n            *INP_SIZE, \n            interpolation=\"bilinear\"\n        ),\n        \n    ], \n    name='PreprocessingLayers'\n)\n\n# Augmentation\ndata_augmentations = keras.Sequential(\n    [\n        RandomCrop(*INP_SIZE),\n        RandomFlip(\"horizontal\"),\n        RandomZoom(0.1, fill_mode='nearest'),\n        RandomRotation(0.1, fill_mode='nearest')\n    ],\n    name='AugmentationLayers'\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:51.695352Z","iopub.execute_input":"2023-01-07T13:47:51.695738Z","iopub.status.idle":"2023-01-07T13:47:51.741214Z","shell.execute_reply.started":"2023-01-07T13:47:51.695644Z","shell.execute_reply":"2023-01-07T13:47:51.74056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To use TPU, uncomment the line with strategy.scope():\n\nand put the model declaration inside like this:\n_______________________________________\nwith strategy.scope():\n\n    model = ...\n_______________________________________\n    \n ","metadata":{}},{"cell_type":"markdown","source":"On Kaggle you can find and add \"keras-applications-models\" to your notebook.\n\nThere are different models of neural networks for working with pictures, which you can try to retrain in this notebook","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import Lambda, Concatenate, Input\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras import applications\n\nif DEVICE == 'TPU':\n    with strategy.scope():\n        custom_metric={\"pfbeta_tf\": tf_pfbeta}\n        path_model = '/kaggle/input/keras-applications-models/InceptionV3.h5'\n#             model1 = tf.keras.models.load_model(path_model, custom_objects=custom_metric)\n        model1 = applications.InceptionV3(include_top=False, pooling='avg')\n\n    \n        input1 = Input(shape=INP_SIZE+(3,), name='input1')\n        input2 = Input(shape=12, name='input2')\n        pr1 = data_preprocessing(input1)\n        pr2 =  data_augmentations(pr1)\n        ress = keras.layers.experimental.preprocessing.Resizing(299, 299)(pr2)\n        mod = model1(ress)\n        concat = Concatenate()([mod, input2])\n        d1 = keras.layers.Dense(32, activation='relu', dtype='float32')(concat)\n        out = keras.layers.Dense(1, activation=None, dtype='float32')(d1)\n\n        model = Model([input1, input2], out)\n\n        model.layers[4].trainable = False\n\n        model.compile(\n            optimizer = optimizers.Adam(learning_rate=lr_schedule, amsgrad=False),  \n            loss = weighted_binary_loss(weight=20), \n            metrics = [\n#             pfbeta_tf,\n              tf_pfbeta(beta=1.0, from_logits=True),\n              auc_logit()\n              ]\n           ) \nelse:\n    custom_metric={\"pfbeta_tf\": tf_pfbeta}\n    path_model = '/kaggle/input/keras-applications-models/EfficientNetB3.h5'\n    model1 = tf.keras.models.load_model(path_model, custom_objects=custom_metric)\n#     model1 = applications.InceptionV3(include_top=False, pooling='avg')\n#     model1 = applications.EfficientNetB3(include_top=False, pooling='avg')\n#     model1 = applications.EfficientNetB7(include_top=False, pooling='avg')\n\n    \n    input1 = Input(shape=INP_SIZE+(3,), name='input1')\n    input2 = Input(shape=12, name='input2')\n    pr1 = data_preprocessing(input1)\n    pr2 =  data_augmentations(pr1)\n#     ress = keras.layers.Resizing(299, 299)(pr2)\n    mod = model1(pr2)\n    concat = Concatenate()([mod, input2])\n    d1 = keras.layers.Dense(32, activation='relu', dtype='float32')(concat)\n    out = keras.layers.Dense(1, activation=None, dtype='float32')(d1)\n\n    model = Model([input1, input2], out)\n\n#     model.layers[4].trainable = False\n\n    model.compile(\n        optimizer = optimizers.Adam(learning_rate=lr_schedule, amsgrad=False),  \n        loss = weighted_binary_loss(weight=30), \n        metrics = [\n#            pfbeta_tf,\n            tf_pfbeta(beta=1.0, from_logits=True),\n            auc_logit()\n            ]\n        ) \n    \n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:51.742377Z","iopub.execute_input":"2023-01-07T13:47:51.742659Z","iopub.status.idle":"2023-01-07T13:47:56.959568Z","shell.execute_reply.started":"2023-01-07T13:47:51.742625Z","shell.execute_reply":"2023-01-07T13:47:56.958841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_callbacks = [\n    callbacks.ModelCheckpoint(\n       filepath='model.{epoch:02d}-{val_loss:.2f}.h5',\n       monitor='val_pfbeta',\n       mode='max',\n       save_best_only=True\n   ),\n   callbacks.CSVLogger('history.csv')\n]","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:56.960997Z","iopub.execute_input":"2023-01-07T13:47:56.961261Z","iopub.status.idle":"2023-01-07T13:47:56.966213Z","shell.execute_reply.started":"2023-01-07T13:47:56.961226Z","shell.execute_reply":"2023-01-07T13:47:56.964768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.fit(\n#     training_dataset,  \n#     validation_data = valid_dataset, \n#     epochs=epochs,\n#     callbacks=training_callbacks\n# )","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:56.971108Z","iopub.execute_input":"2023-01-07T13:47:56.9713Z","iopub.status.idle":"2023-01-07T13:47:56.98393Z","shell.execute_reply.started":"2023-01-07T13:47:56.971276Z","shell.execute_reply":"2023-01-07T13:47:56.98323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.save('model_2_exp_1024_30.h5')","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:56.98509Z","iopub.execute_input":"2023-01-07T13:47:56.985345Z","iopub.status.idle":"2023-01-07T13:47:56.994108Z","shell.execute_reply.started":"2023-01-07T13:47:56.985309Z","shell.execute_reply":"2023-01-07T13:47:56.993362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with InceptionV3\n\n# model_10_20:                val_auc=0.71, val_pfbeta=0.041\n# model_10_100:               val_auc=0.69, val_pfbeta=0.036\n# model_10_wu:                val_auc=0.60, val_pfbeta=0.033\n# model_10_non_trainable:     val_auc=0.58, val_pfbeta=0.033\n# model_10_exp:               val_auc=0.74, val_pfbeta=0.047\n# model_10_exp_nr:            val_auc=0.68, val_pfbeta=0.043\n# model_4_exp_1024:           val_auc=0.61, val_pfbeta=0.020\n# model_30_exp_256:           val_auc=0.80, val_pfbeta=0.040\n# model_9_exp_256:            val_auc=0.69, val_pfbeta=0.047\n# model_21_exp_256_30         val_auc=0.66, val_pfbeta=0.022\n# model_18_exp_256_40         val_auc=0.71, val_pfbeta=0.026\n# model_9_exp_512_30          val_auc=0.64, val_pfbeta=0.022\n# model_16_exp_512_30         val_auc=0.67, val_pfbeta=0.021\n# model_23_exp_512_30         val_auc=0.68, val_pfbeta=0.021\n# model_39_exp_256_30         val_auc=0.66, val_pfbeta=0.034\n# model_2_exp_1024_30         val_auc=0.56, val_pfbeta=0.016\n# model_4_exp_1024_30         val_auc=0.60, val_pfbeta=0.016\n# model_8_exp_1024_30         val_auc=0.62, val_pfbeta=0.016\n\n# with efficientnetB7\n\n# model_10_exp_non_trainable:     val_auc=0.68, val_pfbeta=0.034\n# model_5_exp_256:                val_auc=0.68, val_pfbeta=0.024\n\n# with efficientnetB3\n\n# model_10_exp:          val_auc=0.79, val_pfbeta=0.042\n# model_10_exp_best:     val_auc=0.71, val_pfbeta=0.045\n\n# with Xception\n\n#model_3_exp_512:    val_auc=0.58, val_pfbeta=0.021","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:56.995499Z","iopub.execute_input":"2023-01-07T13:47:56.995951Z","iopub.status.idle":"2023-01-07T13:47:57.004791Z","shell.execute_reply.started":"2023-01-07T13:47:56.995894Z","shell.execute_reply":"2023-01-07T13:47:57.003999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_model = '/kaggle/input/model-8-exp-1024-30/model_8_exp_1024_30.h5'\ncustom_obj={\n            \"weighted_loss\": weighted_binary_loss(weight=30), \n            'pfbeta': tf_pfbeta(beta=1.0, from_logits=True), \n            'auc': auc_logit(), \n#             'WarmupLearningRateSchedule': WarmupLearningRateSchedule\n           }\nmodel = keras.models.load_model(path_model, custom_objects=custom_obj)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:47:57.005745Z","iopub.execute_input":"2023-01-07T13:47:57.006367Z","iopub.status.idle":"2023-01-07T13:48:03.736941Z","shell.execute_reply.started":"2023-01-07T13:47:57.006327Z","shell.execute_reply":"2023-01-07T13:48:03.736166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.fit(\n#     training_dataset,  \n#     validation_data = valid_dataset, \n#     epochs=epochs,\n#     callbacks=training_callbacks\n# )","metadata":{"execution":{"iopub.status.busy":"2023-01-07T13:48:03.738135Z","iopub.execute_input":"2023-01-07T13:48:03.738387Z","iopub.status.idle":"2023-01-07T19:06:09.592013Z","shell.execute_reply.started":"2023-01-07T13:48:03.738353Z","shell.execute_reply":"2023-01-07T19:06:09.590538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.save('model_8_exp_1024_30.h5')","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:36.030146Z","iopub.execute_input":"2023-01-07T19:06:36.030423Z","iopub.status.idle":"2023-01-07T19:06:37.015113Z","shell.execute_reply.started":"2023-01-07T19:06:36.030385Z","shell.execute_reply":"2023-01-07T19:06:37.014352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plots for loss functions and metrics","metadata":{}},{"cell_type":"code","source":"# history = pd.read_csv('/kaggle/working/history.csv')\n\n# history","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:10:10.643059Z","iopub.execute_input":"2023-01-07T19:10:10.643343Z","iopub.status.idle":"2023-01-07T19:10:10.667016Z","shell.execute_reply.started":"2023-01-07T19:10:10.643312Z","shell.execute_reply":"2023-01-07T19:10:10.665804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history.pfbeta.plot()\n# history.val_pfbeta.plot()","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:09.598332Z","iopub.status.idle":"2023-01-07T19:06:09.598994Z","shell.execute_reply.started":"2023-01-07T19:06:09.598744Z","shell.execute_reply":"2023-01-07T19:06:09.598768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history.auc.plot()\n# history.val_auc.plot()","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:09.600196Z","iopub.status.idle":"2023-01-07T19:06:09.60085Z","shell.execute_reply.started":"2023-01-07T19:06:09.600591Z","shell.execute_reply":"2023-01-07T19:06:09.600632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history.loss.plot()\n# history.val_loss.plot()","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:09.602091Z","iopub.status.idle":"2023-01-07T19:06:09.602749Z","shell.execute_reply.started":"2023-01-07T19:06:09.602497Z","shell.execute_reply":"2023-01-07T19:06:09.602522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting for test data","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(f\"{DF_PATH}/test.csv\")\ntest_df['img_path'] = df.apply(\n    lambda i: os.path.join(\n        f\"{IMG_PATH}\", str(i['patient_id']) + \"_\" + str(i['image_id']) + '.png'\n    ), axis=1\n)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:09.603965Z","iopub.status.idle":"2023-01-07T19:06:09.604587Z","shell.execute_reply.started":"2023-01-07T19:06:09.604352Z","shell.execute_reply":"2023-01-07T19:06:09.604376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_0 = pd.get_dummies(test_df, columns = ['laterality', 'view'])\n\ntest_df_0 = pd.DataFrame(test_df_0, columns = df.columns).fillna(0)\ntest_df_0","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:09.605874Z","iopub.status.idle":"2023-01-07T19:06:09.606509Z","shell.execute_reply.started":"2023-01-07T19:06:09.606258Z","shell.execute_reply":"2023-01-07T19:06:09.606282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = create_dataset(\n    test_df_0,\n    batch_size  = BATCH_SIZE, \n    is_labelled = False, \n    shuffle = False\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:09.607844Z","iopub.status.idle":"2023-01-07T19:06:09.608455Z","shell.execute_reply.started":"2023-01-07T19:06:09.60822Z","shell.execute_reply":"2023-01-07T19:06:09.608244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_dataset)\n\npred = tf.nn.sigmoid(pred)\n\ntest_df['cancer'] = pred\n\nsub = test_df.groupby('prediction_id').cancer.mean().reset_index()\n# sub = test_df.groupby('prediction_id').cancer.max().reset_index()\nTHRESHOLD = 0.3  #   0.8: 0.02, 0.7: 0.04(max), 0.57: 0.05, 0.25: 0.04\n                 # test_model 0.7: 0, 0.5: 0.05, 0.4: 0.04\n                 # model_10  0.7: 0.03, 0.6: 0.06, 0.575: 0.05, 0.55: 0.07, 0.5: 0.05\n                 # model_10_100  0.6: 0.04, 0.5: 0.04\n                 # model_10_wu   0.55: 0.05, 0.5: 0.05\n                 # model_10_nt   0.6: 0.01, 0.5: 0.06\n                 # model_10_exp  0.6: 0.02, 0.55: 0.04, 0.5: 0.06, 0.45: 0.05   ---:0.04\n                 # model_10_eff_exp_nt   0.6: 0.01, 0.5: 0.03\n                 # model_10_eff3_exp     0.6: 0.03, 0.5: 0.05\n                 # model_10_eff3_exp_best     0.6: 0.00, 0.5: 0.00?\n                 # model_10_exp_nr   0.06: 0.05, 0.55: 0.05, 0.5: 0.06\n                 # model_4_exp_1024   0.06: 0, 0.5: 0.03, 0.4: 0.04, 0.3: 0.04   ---:0.04\n                 # model_3_xcept_exp_512  0.5: 0.05, --: 0.04\n                 # model_30_exp_256  0.6: 0.02, 0.5: 0.05, 0.45: 0.05, 0.4: 0.06, 0.3: 0.04\n                 # model_5_eff7_exp_256:  0.5: 0.00, 0.42: 0.05, 0.4: 0.06, 0.35: 0.04, 0.3: 0.04\n                 # model_21_inceptV3_exp_256_30:  0.7: 0.05, 0.62: 0.08, 0.6: 0.08, 0.58: 0.06, 0.5: 0.05\n                 # model_18_inceptV3_exp_256_40:  0.7: 0.05, 0.65: 0.06, 0.6: 0.06, 0.55: 0.05, 0.5: 0.05\n                 # model_9_inceptV3_exp_512_30:  0.7: 0.04, 0.6: 0.07, 0.58: 0.07, 0.55: 0.05, 0.5: 0.05\n                 # model_16_inceptV3_exp_512_30: 0.6: 0.04, 0.5: 0.07, 0.48: 0.06, 0.45: 0.06, 0.4: 0.05\n                 # model_23_inceptV3_exp_512_30: 0.6: 0.04, 0.52: 0.07, 0.51: 0.07, 0.5: 0.07, 0.48: 0.06\n                 # model_39_inceptV3_exp_256_30: 0.8: 0, 0.7: 0.06, 0.65: 0.06, 0.6: 0.05, 0.5: 0.04\n                 # model_2_inceptV3_exp_1024_30: 0.5: 0, 0.4: 0.04, 0.3: 0.04\n                 # model_4_inceptV3_exp_1024_30: 0.4: 0.04, 0.3: 0.04\n                 # model_8_inceptV3_exp_1024_30: 0.4: 0.04, 0.3: 0.04\n                \nsub['cancer'] = (sub.cancer > THRESHOLD).astype(int)\nsub","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:09.60973Z","iopub.status.idle":"2023-01-07T19:06:09.610356Z","shell.execute_reply.started":"2023-01-07T19:06:09.610122Z","shell.execute_reply":"2023-01-07T19:06:09.610147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:09.611581Z","iopub.status.idle":"2023-01-07T19:06:09.612214Z","shell.execute_reply.started":"2023-01-07T19:06:09.611981Z","shell.execute_reply":"2023-01-07T19:06:09.612006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making submission","metadata":{}},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-07T19:06:09.613444Z","iopub.status.idle":"2023-01-07T19:06:09.614093Z","shell.execute_reply.started":"2023-01-07T19:06:09.613843Z","shell.execute_reply":"2023-01-07T19:06:09.613867Z"},"trusted":true},"execution_count":null,"outputs":[]}]}