{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\n\nimport os\nimport pandas as pd\nimport numpy as np\nfrom skimage.io import imread_collection\nimport skimage.io\nimport skimage.color\nimport skimage.transform\nfrom platform import python_version\nimport matplotlib.pyplot as plt\n\nprint(tf.__version__)\nprint(python_version())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# extract filenames from the folder of images\nfilenames = []\nfor root, dirs, files in os.walk('../input/rsna-hemorrhage-jpg/train_jpg/train_jpg'):\n    for file in files:\n        if file.endswith('.jpg'):\n            filenames.append(file)\n            \n# should be the same as the images imported\nlen(filenames)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"col_dir = '../input/rsna-hemorrhage-jpg/train_jpg/train_jpg/*.jpg'\n\n# Create a collection with the available images\nimages = imread_collection(col_dir)\n#we could also try what is below,\n#this should load the images in the order that we expect, \n#but if automatically alphabetical this isn't necessary:\n#images = imread_collection(col_dir, load_pattern = filenames)\n\n#make sure this is equivalent with the number of filenames\nlen(images)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot the first image\nplt.figure()\nplt.imshow(images[0])\nplt.colorbar()\nplt.grid(False)\nplt.show()\n\nprint(images[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check shape\nprint(images[0].shape)\nprint(images[1].shape)\nprint(images[2].shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Select only the first 5000 images\nimages_trn = images[:2000]\nprint(len(images_trn))\nimages_val = images[20000:22000]\nprint(len(images_val))\nimages_tst = images[25000:30000]\nprint(len(images_tst))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_arr_trn = skimage.io.collection.concatenate_images(images_trn)\nimages_arr_val = skimage.io.collection.concatenate_images(images_val)\nimages_arr_tst = skimage.io.collection.concatenate_images(images_tst)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Import labels and selct only first 5000 labels without any additional columns\n#labels = pd.read_feather('../input/rsna-hemorrhage-jpg/meta/meta/labels.fth')\n#labels = labels.iloc[:5000, 1]\n#print(labels)\n#print(type(labels))\n#print(labels.sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = pd.read_feather('../input/rsna-hemorrhage-jpg/meta/meta/labels.fth')\n\n#manipulate the filenames list, stripping the .jpg at the end\nidstosearch = [item.rstrip(\".jpg\") for item in filenames]\n\n#now search the \"ID\" column for ids that correspond to our filenames\n#made the reduced dataframe \"labels2\" for now\nlabels2 = labels[labels['ID'].isin(idstosearch)]\nlabels2.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = labels2.iloc[:, 1]\nprint(labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_trn = labels[:2000]\nprint(len(labels_trn))\nlabels_val = labels[20000:22000]\nprint(len(labels_val))\nlabels_tst = labels[25000:30000]\nprint(len(labels_tst))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(type(labels_trn))\nprint(labels_trn.sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Transform labels into array\nlabels_trn = pd.Series.to_numpy(labels_trn)\nlen(labels_trn)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_val = pd.Series.to_numpy(labels_val)\nlen(labels_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_tst = pd.Series.to_numpy(labels_tst)\nlen(labels_tst)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Build the model\n#model = keras.Sequential([\n#    keras.layers.Flatten(input_shape=(256, 256, 3)),\n#    keras.layers.Dense(128, activation='relu'),\n#    keras.layers.Dense(2, activation='softmax')\n#])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# CNN -> train/test accuracy both at 50%\n#model = keras.Sequential()\n#model.add(keras.layers.Conv2D(20, kernel_size=(6, 6), strides=(1, 1),\n#                 activation='relu',\n#                 input_shape=(256, 256, 3)))\n#model.add(keras.layers.MaxPooling2D(pool_size=(2, 2), strides=(2, 2)))\n#model.add(keras.layers.Flatten())\n#model.add(keras.layers.Dense(50, activation='relu'))\n#model.add(keras.layers.Dense(2, activation='softmax'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# CNN -> train/test accuracy at 60%/50%\n#model = keras.Sequential()\n#model.add(keras.layers.Conv2D(32, kernel_size=(5, 5), strides=(1, 1),\n#                 activation='relu',\n#                 input_shape=(256, 256, 3)))\n#model.add(keras.layers.MaxPooling2D(pool_size=(2, 2), strides=(2, 2)))\n#model.add(keras.layers.Conv2D(64, (5, 5), activation='relu'))\n#model.add(keras.layers.MaxPooling2D(pool_size=(2, 2)))\n#model.add(keras.layers.Flatten())\n#model.add(keras.layers.Dense(1000, activation='relu'))\n#model.add(keras.layers.Dense(2, activation='softmax'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications import resnet50\n\nmodel = resnet50.ResNet50(weights=\"imagenet\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Resize all images \n\nimages_final = []\n\nfor i in range(len(images_arr_trn)):\n  image_rescaled = skimage.transform.resize(images_arr_trn[i], (224, 224, 3))\n  images_final.append(image_rescaled)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(images_final)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_final = skimage.io.collection.concatenate_images(images_final)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer='adam',\n              loss='sparse_categorical_crossentropy',\n              metrics=['accuracy'])\n\n# data = train_images.reshape(2000,75,100,1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Train model\nmodel.fit(images_final, labels_trn, epochs=8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_val = []\n\nfor i in range(len(images_arr_val)):\n  image_rescaled = skimage.transform.resize(images_arr_val[i], (224, 224, 3))\n  images_val.append(image_rescaled)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(images_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_val = skimage.io.collection.concatenate_images(images_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Validate model\ntest_loss, test_acc = model.evaluate(images_val, labels_val, verbose=2)\n\nprint('\\nTest accuracy:', test_acc)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# ToDos:\n# 1. Increase data size\n# 2. Use pretrained model to compare"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.7.4"}},"nbformat":4,"nbformat_minor":1}