{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pydicom\nimport os\nimport matplotlib.pyplot as plt\nimport collections\nfrom tqdm import tqdm_notebook as tqdm\nfrom datetime import datetime\n\nfrom math import ceil, floor\nimport cv2\n\nimport tensorflow as tf\nimport keras\n\nimport sys\n\n# from keras_applications.resnet import ResNet50\nfrom keras_applications.inception_v3 import InceptionV3\n\nfrom sklearn.model_selection import ShuffleSplit\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom pydicom.data import get_testdata_files\nimport os\nfrom os import listdir\nfrom os.path import isfile, join\nfrom glob import glob\nfrom random import sample\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport cv2\nfrom sklearn.model_selection import KFold\nimport collections\nfrom tqdm import tqdm_notebook as tqdm\nfrom datetime import datetime\n\nfrom math import ceil, floor\n\n\nimport tensorflow as tf\nimport keras\n\nimport sys\n\nfrom keras_applications.resnet import ResNet50\n\nfrom sklearn.model_selection import ShuffleSplit\n\n\nimport seaborn as sns\nsns.set()\n\nfrom os import listdir\n\nfrom skimage.transform import resize\nimport scipy.ndimage\nfrom skimage import morphology\nfrom skimage import measure\nfrom skimage.transform import resize\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import MinMaxScaler\nfrom plotly import __version__\nfrom plotly.offline import download_plotlyjs, init_notebook_mode, plot, iplot\nfrom plotly.tools import FigureFactory as FF\nfrom plotly.graph_objs import *\n\nfrom sklearn.model_selection import train_test_split\n\nfrom keras.applications import ResNet50, VGG16\nfrom keras.applications.resnet50 import preprocess_input as preprocess_resnet_50\nfrom keras.applications.vgg16 import preprocess_input as preprocess_vgg_16\nfrom keras.layers import GlobalAveragePooling2D, Dense, Activation\nfrom keras.models import Model\nfrom keras.utils import Sequence\nfrom keras import metrics\nfrom keras.layers import Dense\nfrom keras.layers import Conv2D\n\nfrom keras.layers import BatchNormalization\n\n#from tensorflow.keras.applications.resnet50 import preprocess_input\n#from tensorflow.keras.applications import ResNet50\n#from tensorflow.keras.preprocessing.image import load_img, img_to_array\n\nfrom keras.models import Sequential\nfrom keras.layers import Convolution2D\n\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.layers import MaxPooling2D\nfrom keras.layers import Flatten\nfrom keras.layers import Dense\n\nfrom keras.layers import *\nfrom keras.models import Sequential\nfrom keras.applications.resnet50 import ResNet50\n\nfrom keras_applications.inception_v3 import InceptionV3\n\nfrom sklearn.model_selection import ShuffleSplit\n\n\nfrom keras.utils import to_categorical\nfrom keras.metrics import categorical_accuracy\n\nfrom keras.utils import np_utils\n\nfrom keras import optimizers\n\nimport tqdm\n\nfrom skimage.filters.rank import median\nfrom skimage.morphology import disk\n\ntest_images_dir = '../input/rsna-intracranial-hemorrhage-detection/stage_2_test_images/'\n#train_images_dir = '../input/rsna-intracranial-hemorrhage-detection/stage_2_train_images/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from skimage import data, color\nfrom skimage.transform import rescale, resize, downscale_local_mean\n\nprct_epid = 0.02\nprct_intrap = 0.25\nprct_intrav = 0.17\nprct_subar = 0.25\nprct_subdu = 0.31\n\n\n\ndef sample_by_distr_subtype(prct, size_of_samples, subtype,df):\n    tr_type_1 = df[df[subtype] == 1]\n    tr_type_0 = df[df[subtype] == 0]\n    tr_type = pd.concat([tr_type_1.sample(int(size_of_samples*prct/2)),tr_type_0.sample(int(size_of_samples*prct/2))])\n    return tr_type\n\n\ndef sample_by_distribution(size_of_samples,df):\n    list_subtypes = list(df.columns)\n\n    tr_epidural = sample_by_distr_subtype( prct_epid, size_of_samples,'epidural' ,df)\n    tr_intrap = sample_by_distr_subtype( prct_intrap , size_of_samples,'intraparenchymal' ,df)\n    tr_intrav = sample_by_distr_subtype( prct_intrav, size_of_samples,'intraventricular' ,df)\n    tr_subar = sample_by_distr_subtype( prct_subar, size_of_samples,'subarachnoid' ,df)\n    tr_subdu = sample_by_distr_subtype( prct_subdu, size_of_samples,'subdural' ,df)\n    tr_any = df[df[\"any\"] == 0].sample(int(size_of_samples * 0.72))\n    \n    return pd.concat([tr_epidural, tr_intrap, tr_intrav, tr_subar, tr_subdu, tr_any])\n\n\ndef get_data_samples(size_of_samples, df):\n    #sample_df = pd.concat([train_df[train_df['Label'] == 1].sample(size_of_samples),train_df[train_df['Label'] == 0].sample(size_of_samples)])\n    sample_df = sample_by_distribution(size_of_samples,df)\n    xtrain, xval = train_test_split(sample_df,\n                            test_size=0.1,\n                            shuffle = True,\n                            stratify=sample_df.values)\n                            #random_state=1)\n    return xtrain,xval\n\nimage_size = 224\ndef read_and_prep_images(img_paths, img_size=(224,224), pixels_out=True, folder=train_images_dir, display_mode=0):\n    \n    img_array = []\n    gauge = tqdm.tqdm(range(len(img_paths))) if len(img_paths) > 1 else range(len(img_paths))\n    \n    for i in gauge:\n        img_array.append(resize(apply_mask(get_pixels_hu(img_paths[i], folder=folder),-150,2000,display_mode=display_mode),(img_size[0],img_size[1])))\n\n    img_array = np.array(img_array)\n\n    mean = np.mean(np.array(img_array))\n    std = np.std(np.array(img_array))\n    std = 1 if std == 0 else std\n    \n    for i in range(0,img_array.shape[0]):\n#         std = np.std(img_array[i])\n#         std = 1 if std == 0 else std\n        img_array[i] = (img_array[i] - mean) / std\n    \n    if pixels_out:\n        img_array = np.repeat(img_array[..., np.newaxis], 3, -1)\n    output = img_array\n    #img_array = np.array([cv2.resize(img.pixel_array,(224,224)) for img in imgs])\n\n    #output = preprocess_input(img_array)\n    \n    return output\n\ndef get_ds(img_name, folder=train_images_dir):\n    pydicom_filedataset = pydicom.read_file(os.path.join(folder, img_name))\n    return pydicom_filedataset\n\ndef get_pixel_array(img_name, folder=train_images_dir):\n    pydicom_filedataset = pydicom.read_file(os.path.join(folder, img_name))\n    return pydicom_filedataset.pixel_array\n\ndef get_pixels_hu(img_name, folder=train_images_dir):\n    try:\n        image = get_pixel_array(img_name, folder=folder)\n        dataset = get_ds(img_name, folder=folder)\n        # Convert to int16 (from sometimes int16), \n        # should be possible as values should always be low enough (<32k)\n        image= image.astype(np.int16)\n\n        # Set outside-of-scan pixels to 1\n        # The intercept is usually -1024, so air is approximately 0\n        image[image <= -2000] = 0\n\n        # Convert to Hounsfield units (HU)\n        intercept = dataset.RescaleIntercept\n        slope = dataset.RescaleSlope\n\n        if slope != 1:\n            image = slope * image.astype(np.float64)\n            image = image.astype(np.int16)\n\n        image += np.int16(intercept)\n        return np.array(image, dtype=np.int16)\n    except:\n        print(\"File {} is corrupted, removing from set\".format(img_name))\n        return False\n\ndef set_manual_window(hu_image, custom_center, custom_width):\n    min_value = custom_center - (custom_width/2)\n    max_value = custom_center + (custom_width/2)\n    hu_image[hu_image < min_value] = min_value\n    hu_image[hu_image > max_value] = max_value\n    return hu_image\n\n\ndef view_image(img):\n    plt.imshow(get_pixel_array(img), cmap='bone')\n\ndef apply_mask(img,th,upper_th,display_mode=0):\n    print('starting apply mask')\n    if img is False:            # In case get_pixels_hu crashed, we need to return an empty image\n        return np.array([[0]], dtype='float32')\n    row_size= img.shape[0]\n    col_size = img.shape[1]\n    \n    #thresh_img = np.where((img - 1000) / 150 < 1,1.0,0.0)  # threshold the image\n    filter_hu_bounds = (img>th) & (img<upper_th)\n    thresh_img = np.where(filter_hu_bounds,1.0,0.0)  # threshold the image\n    \n    \n    #eroded = morphology.erosion(thresh_img,np.ones([3,3]))\n    #dilation = morphology.dilation(eroded,np.ones([8,8]))\n    \n    dilation = thresh_img #median(thresh_img, disk(10))\n\n    labels = measure.label(dilation, background= -1) # Different labels are displayed in different colors\n    regions = measure.regionprops(labels,img)\n    regions.sort(reverse=True, key=lambda x:x.bbox_area)\n    regions.remove(regions[0])\n    regions.sort(reverse=True, key=lambda x:x.area)\n    \n    if (len(regions) <= 0):\n        return np.array([[0]], dtype='float32')\n    skull_region=regions[0]\n    x,y,x1,y1=skull_region.bbox\n    cropped_image = img[x:x1,y:y1]\n    cropped_image_filled = np.where(skull_region.filled_image, cropped_image, -100000 * np.ones(cropped_image.shape))\n    mask = np.where(cropped_image_filled>-100000,1,0)\n    \n    if display_mode == 1:\n        fig, ax = plt.subplots(3, 2, figsize=[12, 12])\n        ax[0, 0].set_title(\"Original\")\n        ax[0, 0].imshow(img, cmap='gray')\n        ax[0, 0].axis('off')\n        ax[0, 1].set_title(\"Threshold\")\n        ax[0, 1].imshow(thresh_img, cmap='gray')\n        ax[0, 1].axis('off')\n        ax[1, 0].set_title(\"After Erosion and Dilation\")\n        ax[1, 0].imshow(dilation, cmap='gray')\n        ax[1, 0].axis('off')\n        ax[1, 1].set_title(\"Color Labels\")\n        ax[1, 1].imshow(labels)\n        ax[1, 1].axis('off')\n        ax[2, 0].set_title(\"Final Mask\")\n        ax[2, 0].imshow(mask, cmap='gray')\n        ax[2, 0].axis('off')\n        ax[2, 1].set_title(\"Apply Mask on Original\")\n        ax[2, 1].imshow(cropped_image * mask, cmap='gray')\n        ax[2, 1].axis('off')\n        plt.show()\n\n    if display_mode == 2:\n        fig, ax = plt.subplots(1, 3, figsize=[12, 12])\n        ax[0].set_title(\"Original\")\n        ax[0].imshow(img, cmap='gray')\n        ax[0].axis('off')\n        ax[1].set_title(\"Apply Mask on Original\")\n        ax[1].imshow(cropped_image * mask, cmap='gray')\n        ax[1].axis('off')\n        ax[2].set_title(\"Original windowed\")\n        ax[2].imshow(set_manual_window(cropped_image * mask, 30, 150), cmap='gray')\n        ax[2].axis('off')\n        plt.show()\n    \n    \n    result = cropped_image * mask\n    del regions\n    del labels\n    del cropped_image\n    del mask\n    return result.astype('float64')\n\n\ndef img_intensity_histo(img_path):\n    imgs_to_process = get_pixels_hu(img_path).astype(np.float64) \n    plt.hist(imgs_to_process.flatten(), bins=50, color='c')\n    plt.xlabel(\"Hounsfield Units (HU)\")\n    plt.ylabel(\"Frequency\")\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class DataGenerator(keras.utils.Sequence):\n\n    def __init__(self, list_IDs, labels=None, batch_size=1, img_size=(512, 512, 3), \n                 img_dir=train_images_dir, *args, **kwargs):\n\n        self.list_IDs = list_IDs\n        self.labels = labels\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.img_dir = img_dir\n        self.shuffle = False\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(ceil(len(self.indices) / self.batch_size))\n\n    def __getitem__(self, index):\n        indices = self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indices]\n        if self.labels is not None:\n            X, Y = self.__data_generation(list_IDs_temp)\n            return X, Y\n        else:\n            X = self.__data_generation(list_IDs_temp)\n            return X\n        \n        \n    def on_epoch_end(self):\n#         if self.labels is not None: # for training phase we undersample and shuffle\n#             # keep probability of any=0 and any=1\n#             keep_prob = self.labels.iloc[:, 0].map({0: 0.35, 1: 0.5})\n#             keep = (keep_prob > np.random.rand(len(keep_prob)))\n#             self.indices = np.arange(len(self.list_IDs))[keep]\n#             np.random.shuffle(self.indices)\n#         else:\n        self.indices = np.arange(len(self.list_IDs))\n        if self.shuffle == True:\n            np.random.shuffle(self.indices)\n\n            \n            \n    def __data_generation(self, list_IDs_temp):\n        X = np.empty((self.batch_size, *self.img_size))\n        \n        if self.labels is not None: # training phase\n            Y = np.empty((self.batch_size, 6), dtype=np.float32)\n            for i, ID in enumerate(list_IDs_temp):\n                X[i,] = read_and_prep_images([ID+\".dcm\"], self.img_size, pixels_out=True)\n                Y[i,] = self.labels.loc[ID].values\n\n            return X, Y\n        else: # test phase\n            print('read and prep on test images')\n            for i, ID in enumerate(list_IDs_temp):\n                X[i,] = read_and_prep_images([ID+\".dcm\"], self.img_size,pixels_out=True,folder=self.img_dir)\n            \n            return X\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras import backend as K\n\ndef weighted_log_loss(y_true, y_pred):\n    \"\"\"\n    Can be used as the loss function in model.compile()\n    ---------------------------------------------------\n    \"\"\"\n    \n    class_weights = np.array([2., 1., 1., 1., 1., 1.])\n    \n    eps = K.epsilon()\n    \n    y_pred = K.clip(y_pred, eps, 1.0-eps)\n\n    out = -(         y_true  * K.log(      y_pred) * class_weights\n            + (1.0 - y_true) * K.log(1.0 - y_pred) * class_weights)\n    \n    return K.mean(out, axis=-1)\n\n\ndef _normalized_weighted_average(arr, weights=None):\n    \"\"\"\n    A simple Keras implementation that mimics that of \n    numpy.average(), specifically for this competition\n    \"\"\"\n    \n    if weights is not None:\n        scl = K.sum(weights)\n        weights = K.expand_dims(weights, axis=1)\n        return K.sum(K.dot(arr, weights), axis=1) / scl\n    return K.mean(arr, axis=1)\n\n\ndef weighted_loss(y_true, y_pred):\n    \"\"\"\n    Will be used as the metric in model.compile()\n    ---------------------------------------------\n    \n    Similar to the custom loss function 'weighted_log_loss()' above\n    but with normalized weights, which should be very similar \n    to the official competition metric:\n        https://www.kaggle.com/kambarakun/lb-probe-weights-n-of-positives-scoring\n    and hence:\n        sklearn.metrics.log_loss with sample weights\n    \"\"\"\n    \n    class_weights = K.variable([2., 1., 1., 1., 1., 1.])\n    \n    eps = K.epsilon()\n    \n    y_pred = K.clip(y_pred, eps, 1.0-eps)\n\n    loss = -(        y_true  * K.log(      y_pred)\n            + (1.0 - y_true) * K.log(1.0 - y_pred))\n    \n    loss_samples = _normalized_weighted_average(loss, class_weights)\n    \n    return K.mean(loss_samples)\n\n\ndef weighted_log_loss_metric(trues, preds):\n    \"\"\"\n    Will be used to calculate the log loss \n    of the validation set in PredictionCheckpoint()\n    ------------------------------------------\n    \"\"\"\n    class_weights = [2., 1., 1., 1., 1., 1.]\n    \n    epsilon = 1e-7\n    \n    preds = np.clip(preds, epsilon, 1-epsilon)\n    loss = trues * np.log(preds) + (1 - trues) * np.log(1 - preds)\n    loss_samples = np.average(loss, axis=1, weights=class_weights)\n\n    return - loss_samples.mean()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class PredictionCheckpoint(keras.callbacks.Callback):\n    \n    def __init__(self, test_df, valid_df, \n                 test_images_dir=test_images_dir, \n                 valid_images_dir=train_images_dir, \n                 batch_size=32, input_size=(224, 224, 3)):\n        \n        self.test_df = test_df\n        self.valid_df = valid_df\n        self.test_images_dir = test_images_dir\n        self.valid_images_dir = valid_images_dir\n        self.batch_size = batch_size\n        self.input_size = input_size\n        \n    def on_train_begin(self, logs={}):\n        self.test_predictions = []\n        self.valid_predictions = []\n        \n    def on_epoch_end(self,batch, logs={}):\n        self.test_predictions.append(\n            self.model.predict_generator(\n                DataGenerator(self.test_df.index, None, self.batch_size, self.input_size, self.test_images_dir), verbose=2)[:len(self.test_df)])\n                # Commented out to save time\n#         self.valid_predictions.append(\n#             self.model.predict_generator(\n#                 DataGenerator(self.valid_df.index, None, self.batch_size, self.input_size, self.valid_images_dir), verbose=2)[:len(self.valid_df)])\n        \n#         print(\"validation loss: %.4f\" %\n#               weighted_log_loss_metric(self.valid_df.values, \n#                                    np.average(self.valid_predictions, axis=0, \n#                                               weights=[2**i for i in range(len(self.valid_predictions))])))\n        \n        # here you could also save the predictions with np.save()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nclass MyDeepModel:\n    \n    def __init__(self, engine, input_dims, batch_size=5, num_epochs=4, learning_rate=1e-3, \n                 decay_rate=1.0, decay_steps=1, weights='imagenet', verbose=1):\n        \n        self.engine = engine\n        self.input_dims = input_dims\n        self.batch_size = batch_size\n        self.num_epochs = num_epochs\n        self.learning_rate = learning_rate\n        self.decay_rate = decay_rate\n        self.decay_steps = decay_steps\n        self.weights = weights\n        self.verbose = verbose\n        self._build()\n\n    def _build(self):\n        \n    \n        engine = self.engine(include_top=False, weights=self.weights, input_shape=(*self.input_dims[:2], 3),\n                             backend = keras.backend, layers = keras.layers,\n                             models = keras.models, utils = keras.utils)\n        \n\n        x = keras.layers.GlobalAveragePooling2D(name='avg_pool')(engine.output)\n        x = keras.layers.Dropout(0.2)(x)\n        x = keras.layers.Dense(keras.backend.int_shape(x)[1], activation=\"relu\", name=\"dense_hidden_1\")(x)\n        x = keras.layers.Dropout(0.1)(x)\n        out = keras.layers.Dense(6, activation=\"sigmoid\", name='dense_output')(x)\n\n        self.model = keras.models.Model(inputs=engine.input, outputs=out)\n\n        self.model.compile(loss=weighted_log_loss, optimizer=keras.optimizers.Adam(), metrics=[weighted_loss])\n    \n    def fit_and_predict_subset(self, train_df, test_df, subsetNumber):\n        print('fit and predict subset')\n        test_df_sub = test_df.sample(subsetNumber)\n        sample_df = train_df.sample(subsetNumber)\n        ss = ShuffleSplit(n_splits=1, test_size=0.1, random_state=42).split(sample_df.index)\n        train_idx, valid_idx = next(ss)\n        model.fit_and_predict(sample_df.iloc[train_idx], sample_df.iloc[valid_idx], test_df_sub)\n       \n\n    def fit_and_predict(self, train_df, valid_df, test_df):\n        \n        # callbacks\n        print('Fit and predict')\n        pred_history = PredictionCheckpoint(test_df, valid_df, input_size=self.input_dims)\n        print('pred_history done')\n        checkpointer = keras.callbacks.ModelCheckpoint(filepath='%s-{epoch:02d}.hdf5' % self.engine.__name__, verbose=1, save_weights_only=True, save_best_only=False)\n        print('checkpointer done')\n        scheduler = keras.callbacks.LearningRateScheduler(lambda epoch: self.learning_rate * pow(self.decay_rate, floor(epoch / self.decay_steps)))\n        print('scheduler done')\n        \n        self.model.fit_generator(\n            DataGenerator(\n                train_df.index, \n                train_df, \n                self.batch_size, \n                self.input_dims, \n                train_images_dir\n            ),\n            epochs=self.num_epochs,\n            verbose=self.verbose,\n            use_multiprocessing=True,\n            workers=4,\n            #callbacks=[scheduler]\n            callbacks=[pred_history, scheduler]\n        )\n        \n        return pred_history\n    \n    def predict(self,test_df):\n        preds = self.model.predict_generator(DataGenerator(\n                test_df.index, \n                test_df, \n                self.batch_size, \n                self.input_dims, \n                test_images_dir\n            ),verbose=self.verbose,steps = 1)\n        return preds\n        \n    def save(self, path):\n        self.model.save_weights(path)\n    \n    def load(self, path):\n        self.model.load_weights(path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def read_testset(filename=\"../input/rsna-intracranial-hemorrhage-detection/stage_2_sample_submission.csv\"):\n    \n    df = pd.read_csv(filename)\n    \n    # Extract Id image form the ID column\n    df['Image'] = df['ID'].apply(lambda st:\"ID_\" + st.split('_')[1])\n    \n    # Extract the subtype of hymorrage from the ID column\n    df[\"Diagnosis\"] =  df['ID'].apply(lambda st: st.split('_')[2])\n\n    # Dropping the ID column\n    df = df[['Image',\"Diagnosis\",'Label']]\n    \n    # Drop the duplicates\n    df = df.drop_duplicates(subset=['Image','Diagnosis'], keep='first')\n    \n    ## Seting a multi indexing and pivot the table on dignosis\n    df = df.set_index(['Image', 'Diagnosis']).unstack(level=-1)\n    \n    return df\n\ndef read_trainset(filename=\"../input/rsna-intracranial-hemorrhage-detection/stage_2_train.csv\"):\n    df = pd.read_csv(filename)\n     \n    # Extract Id image form the ID column\n    df['Image'] = df['ID'].apply(lambda st:\"ID_\" + st.split('_')[1])\n    \n    # Extract the subtype of hymorrage from the ID column\n    df[\"Diagnosis\"] =  df['ID'].apply(lambda st: st.split('_')[2])\n    \n        \n    # Dropping the ID column\n    df = df[['Image',\"Diagnosis\",'Label']]\n    \n    # Drop the duplicates\n    df = df.drop_duplicates(subset=['Image','Diagnosis'], keep='first')\n\n    ## Seting a multi indexing and pivot the table on dignosis\n    df = df.set_index(['Image', 'Diagnosis']).unstack(level=-1)\n    return df\n\n    \ntest_df = read_testset()\ntrain_df = read_trainset()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# obtain model\nmodel = MyDeepModel(engine=InceptionV3, input_dims=(224, 224, 3), batch_size=32, learning_rate=5e-4,\n                    num_epochs=4, decay_rate=0.8, decay_steps=1, weights=\"imagenet\", verbose=1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_test_df = test_df.sample(32)\npreds = model.predict(sample_test_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# obtain test + validation predictions (history.test_predictions, history.valid_predictions)\nsample_test_df = test_df.sample(50)\nimport warnings\n\nwith warnings.catch_warnings():\n    warnings.simplefilter('error')\n    #history = model.fit_and_predict(train_df.iloc[train_idx], train_df.iloc[valid_idx], test_df)\n    history = model.fit_and_predict_subset(train_df, test_df, 10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"# model.save(\"20191102-2315.weights\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"#test_df1 = test_df\npreds = model.model.predict_generator(DataGenerator(test_df.index, None, 32, (224,224,3), test_images_dir), verbose=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"test_df = read_testset()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"print(test_df.head())\nprint(test_df.shape)\n#preds.head()\nprint(len(history.test_predictions))\n#print(history.test_predictions[0])\nprint(len(np.average(history.test_predictions, axis=0, weights=[0, 2, 4, 6]) ))","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"#print(preds)\ntest_df.iloc[:, :] = np.average(history.test_predictions, axis=0, weights=[0, 2, 4, 6]) # let's do a weighted average for epochs (>1)\n\n#test_df.iloc[:, :] = preds[0:78545]\n\ntest_df = test_df.stack().reset_index()\n\ntest_df.insert(loc=0, column='ID', value=test_df['Image'].astype(str) + \"_\" + test_df['Diagnosis'])\n\ntest_df = test_df.drop([\"Image\", \"Diagnosis\"], axis=1)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"test_df.to_csv('submission_with_history.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.7.4"}},"nbformat":4,"nbformat_minor":1}