{"cells":[{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:26:26.348614Z","iopub.status.busy":"2021-02-08T01:26:26.347825Z","iopub.status.idle":"2021-02-08T01:26:40.844483Z","shell.execute_reply":"2021-02-08T01:26:40.844992Z"},"papermill":{"duration":14.536165,"end_time":"2021-02-08T01:26:40.845381","exception":false,"start_time":"2021-02-08T01:26:26.309216","status":"completed"},"tags":[],"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"# !pip install git+https://github.com/keras-team/keras-applications.git -q\n\n# import keras_applications as ka\n# def set_to_tf(ka):\n#     from tensorflow.keras import backend, layers, models, utils\n#     ka._KERAS_BACKEND = backend\n#     ka._KERAS_LAYERS = layers\n#     ka._KERAS_MODELS = models\n#     ka._KERAS_UTILS = utils\n    \n    \n# set_to_tf(ka)\n\n!pip install /kaggle/input/keras-pretrained-imagenet-weights/image_classifiers-1.0.0-py3-none-any.whl\n\nfrom classification_models.tfkeras import Classifiers\nClassifiers.models_names()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2021-02-08T01:26:40.887675Z","iopub.status.busy":"2021-02-08T01:26:40.886985Z","iopub.status.idle":"2021-02-08T01:26:41.640667Z","shell.execute_reply":"2021-02-08T01:26:41.641162Z"},"papermill":{"duration":0.776,"end_time":"2021-02-08T01:26:41.641342","exception":false,"start_time":"2021-02-08T01:26:40.865342","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# ML tools \nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nfrom keras.models import Sequential\nfrom keras import layers\nfrom keras.optimizers import Adam\nfrom tensorflow.keras import Model\n# import tensorflow.keras.applications.efficientnet as efn\nfrom tensorflow.keras.applications import *\nimport os\nfrom keras import optimizers\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, ModelCheckpoint, EarlyStopping","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2021-02-08T01:26:41.682783Z","iopub.status.busy":"2021-02-08T01:26:41.682157Z","iopub.status.idle":"2021-02-08T01:26:41.897908Z","shell.execute_reply":"2021-02-08T01:26:41.897223Z"},"papermill":{"duration":0.237485,"end_time":"2021-02-08T01:26:41.898073","exception":false,"start_time":"2021-02-08T01:26:41.660588","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/nih-dataframe/NIH_Dataframe.csv')\ndf.img_ind= df.img_ind.apply(lambda x: x.split('.')[0])\ndisplay(df.head(4))\nprint(df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:26:42.265384Z","iopub.status.busy":"2021-02-08T01:26:42.26372Z","iopub.status.idle":"2021-02-08T01:26:42.266064Z","shell.execute_reply":"2021-02-08T01:26:42.266567Z"},"papermill":{"duration":0.027297,"end_time":"2021-02-08T01:26:42.266735","exception":false,"start_time":"2021-02-08T01:26:42.239438","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"target_cols = df.drop(['img_ind'], axis=1).columns.to_list()\nn_classes = len(target_cols)\nimg_size = 600\nn_epochs = 35\nlr= 0.0001\nseed= 11\nval_split= 0.2\nseed= 33\nbatch_size=12\nn_classes","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:26:42.309792Z","iopub.status.busy":"2021-02-08T01:26:42.30912Z","iopub.status.idle":"2021-02-08T01:26:42.328391Z","shell.execute_reply":"2021-02-08T01:26:42.327831Z"},"papermill":{"duration":0.041826,"end_time":"2021-02-08T01:26:42.328543","exception":false,"start_time":"2021-02-08T01:26:42.286717","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy\n\n'''\nReference\nhttps://www.kaggle.com/xhlulu/ranzcr-efficientnet-tpu-training\n\n'''\n\ndef build_decoder(with_labels=True, target_size=(img_size, img_size), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path) # Reads and outputs the entire contents of the input filename.\n\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3) # Decode a PNG-encoded image to a uint8 or uint16 tensor\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3) # Decode a JPEG-encoded image to a uint8 tensor\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.cast(img, tf.float32) / 255.0 # Casts a tensor to the type float32 and divides by 255.\n        img = tf.image.resize(img, target_size) # Resizing to target size\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        img = tf.image.random_saturation(img, 0.8, 1.2)\n        img = tf.image.random_brightness(img, 0.1)\n        img = tf.image.random_contrast(img, 0.8, 1.2)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\ndef build_dataset(paths, labels=None, bsize=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO) # overlaps data preprocessing and model execution while training\n    return dset\n","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:26:42.423657Z","iopub.status.busy":"2021-02-08T01:26:42.422947Z","iopub.status.idle":"2021-02-08T01:26:48.662539Z","shell.execute_reply":"2021-02-08T01:26:48.663005Z"},"papermill":{"duration":6.314494,"end_time":"2021-02-08T01:26:48.663271","exception":false,"start_time":"2021-02-08T01:26:42.348777","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"DATASET_NAME = \"nih-image-600x600-data\"\nstrategy = auto_select_accelerator()\nbatch_size = strategy.num_replicas_in_sync * batch_size\nprint('batch size', batch_size)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path(DATASET_NAME)\nGCS_DS_PATH","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:26:48.707162Z","iopub.status.busy":"2021-02-08T01:26:48.706538Z","iopub.status.idle":"2021-02-08T01:26:48.800673Z","shell.execute_reply":"2021-02-08T01:26:48.80117Z"},"papermill":{"duration":0.11783,"end_time":"2021-02-08T01:26:48.801455","exception":false,"start_time":"2021-02-08T01:26:48.683625","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"paths = GCS_DS_PATH + \"/NIH_Images/\" + df['img_ind'] + '.jpg'\n\n\n#Get the multi-labels\nlabel_cols = df.columns[:-1]\nlabels = df[label_cols].values","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:26:48.846336Z","iopub.status.busy":"2021-02-08T01:26:48.845673Z","iopub.status.idle":"2021-02-08T01:26:48.877978Z","shell.execute_reply":"2021-02-08T01:26:48.878528Z"},"papermill":{"duration":0.056789,"end_time":"2021-02-08T01:26:48.878703","exception":false,"start_time":"2021-02-08T01:26:48.821914","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"# Train test split\n(train_paths, valid_paths, \n  train_labels, valid_labels) = train_test_split(paths, labels, test_size=val_split, random_state=11)\n\nprint(train_paths.shape, valid_paths.shape)\ntrain_labels.sum(axis=0), valid_labels.sum(axis=0)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:26:49.108111Z","iopub.status.busy":"2021-02-08T01:26:49.105941Z","iopub.status.idle":"2021-02-08T01:26:49.431763Z","shell.execute_reply":"2021-02-08T01:26:49.430954Z"},"papermill":{"duration":0.372349,"end_time":"2021-02-08T01:26:49.431926","exception":false,"start_time":"2021-02-08T01:26:49.059577","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"# Build the tensorflow datasets\n\ndecoder = build_decoder(with_labels=True, target_size=(img_size, img_size))\n\n# Build the tensorflow datasets\ndtrain = build_dataset(\n    train_paths, train_labels, bsize=batch_size, decode_fn=decoder\n)\n\ndvalid = build_dataset(\n    valid_paths, valid_labels, bsize=batch_size, \n    repeat=False, shuffle=False, augment=False, decode_fn=decoder\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data, _ = dtrain.take(2)\nimages = data[0].numpy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, axes = plt.subplots(3, 4, figsize=(20,10))\naxes = axes.flatten()\nfor img, ax in zip(images, axes):\n    ax.imshow(img)\n    ax.axis('off')\nplt.tight_layout()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:26:49.48503Z","iopub.status.busy":"2021-02-08T01:26:49.484346Z","iopub.status.idle":"2021-02-08T01:26:49.486946Z","shell.execute_reply":"2021-02-08T01:26:49.487474Z"},"papermill":{"duration":0.033245,"end_time":"2021-02-08T01:26:49.48767","exception":false,"start_time":"2021-02-08T01:26:49.454425","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"def build_model():\n    seresnet152, _ = Classifiers.get('seresnet50')\n    base = seresnet152(input_shape=(img_size, img_size, 3), include_top=False, weights='imagenet')\n    \n    inp = layers.Input(shape = (img_size, img_size, 3))\n    x= base(inp)\n    x= layers.GlobalAveragePooling2D()(layers.Dropout(0.16)(x))\n    x= layers.Dropout(0.3)(x)\n    x= layers.Dense(n_classes, 'sigmoid')(x)\n    return Model(inp, x)\n    ","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:26:49.538798Z","iopub.status.busy":"2021-02-08T01:26:49.53815Z","iopub.status.idle":"2021-02-08T01:28:51.861528Z","shell.execute_reply":"2021-02-08T01:28:51.86234Z"},"papermill":{"duration":122.352746,"end_time":"2021-02-08T01:28:51.862611","exception":false,"start_time":"2021-02-08T01:26:49.509865","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"with strategy.scope():\n    model= build_model()\n    loss= tf.keras.losses.BinaryCrossentropy(label_smoothing=0.0)\n    model.compile(optimizers.Adam(lr=lr),loss=loss,metrics=[tf.keras.metrics.AUC(multi_label=True)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:28:52.21314Z","iopub.status.busy":"2021-02-08T01:28:52.212123Z","iopub.status.idle":"2021-02-08T01:28:52.21587Z","shell.execute_reply":"2021-02-08T01:28:52.216514Z"},"papermill":{"duration":0.060939,"end_time":"2021-02-08T01:28:52.216709","exception":false,"start_time":"2021-02-08T01:28:52.15577","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"name= 'NIH_Seresnet50_model.h5'\n\nrlr = ReduceLROnPlateau(monitor = 'val_loss', factor = 0.1, patience = 2, verbose = 1, \n                                min_delta = 1e-4, min_lr = 1e-6, mode = 'min', cooldown=1)\n        \nckp = ModelCheckpoint(name,monitor = 'val_loss',\n                      verbose = 1, save_best_only = True, mode = 'min')\n        \nes = EarlyStopping(monitor = 'val_loss', min_delta = 1e-4, patience = 5, mode = 'min', \n                    restore_best_weights = True, verbose = 1)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:28:52.305179Z","iopub.status.busy":"2021-02-08T01:28:52.30415Z","iopub.status.idle":"2021-02-08T01:28:52.310571Z","shell.execute_reply":"2021-02-08T01:28:52.312678Z"},"papermill":{"duration":0.054513,"end_time":"2021-02-08T01:28:52.313072","exception":false,"start_time":"2021-02-08T01:28:52.258559","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"steps_per_epoch = (train_paths.shape[0] // batch_size)\nsteps_per_epoch","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T01:28:52.421579Z","iopub.status.busy":"2021-02-08T01:28:52.405835Z","iopub.status.idle":"2021-02-08T02:44:37.790194Z","shell.execute_reply":"2021-02-08T02:44:37.789657Z"},"papermill":{"duration":4545.439448,"end_time":"2021-02-08T02:44:37.790427","exception":false,"start_time":"2021-02-08T01:28:52.350979","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"history = model.fit(dtrain,                      \n                    validation_data=dvalid,                                       \n                    epochs=15,\n                    callbacks=[rlr,es,ckp],\n                    steps_per_epoch=steps_per_epoch,\n                    verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T02:44:38.78401Z","iopub.status.busy":"2021-02-08T02:44:38.783008Z","iopub.status.idle":"2021-02-08T02:44:39.033038Z","shell.execute_reply":"2021-02-08T02:44:39.032487Z"},"papermill":{"duration":0.74803,"end_time":"2021-02-08T02:44:39.033205","exception":false,"start_time":"2021-02-08T02:44:38.285175","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12, 6))\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.plot( history.history[\"loss\"], label = \"Training Loss\", marker='o')\nplt.plot( history.history[\"val_loss\"], label = \"Validation Loss\", marker='+')\nplt.grid(True)\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T02:44:40.105828Z","iopub.status.busy":"2021-02-08T02:44:40.09697Z","iopub.status.idle":"2021-02-08T02:44:40.278958Z","shell.execute_reply":"2021-02-08T02:44:40.278417Z"},"papermill":{"duration":0.693171,"end_time":"2021-02-08T02:44:40.279118","exception":false,"start_time":"2021-02-08T02:44:39.585947","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (12, 6))\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"AUC\")\nplt.plot( history.history[\"auc\"], label = \"Training AUC\" , marker='o')\nplt.plot( history.history[\"val_auc\"], label = \"Validation AUC\", marker='+')\nplt.grid(True)\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-02-08T02:44:41.268981Z","iopub.status.busy":"2021-02-08T02:44:41.267999Z","iopub.status.idle":"2021-02-08T02:53:02.442665Z","shell.execute_reply":"2021-02-08T02:53:02.443217Z"},"papermill":{"duration":501.673242,"end_time":"2021-02-08T02:53:02.443459","exception":false,"start_time":"2021-02-08T02:44:40.770217","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nfrom sklearn.metrics import roc_auc_score\n\nwith strategy.scope():\n    model= tf.keras.models.load_model(name)\npred= model.predict(dvalid, verbose=1)\n\nprint('AUC CKECK-UP per CLASS')\n\nclasses= df.columns[:-1]\nfor i, n in enumerate(classes):\n  print(classes[i])\n  print(i, roc_auc_score(valid_labels[:, i], pred[:, i]))\n  print('---------')","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.524725,"end_time":"2021-02-08T02:53:03.556695","exception":false,"start_time":"2021-02-08T02:53:03.03197","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}