{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport re\nfrom glob import glob\nfrom tqdm import tqdm\nimport ast\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image, ImageDraw \nfrom dask import bag\n\n%matplotlib inline\nimport os\nprint(os.listdir(\"../input\"))\n\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom tensorflow.keras.metrics import top_k_categorical_accuracy\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\n\nimport pickle # Read/Write with Serialization\nimport requests # Makes HTTP requests\nfrom io import BytesIO # Use When expecting bytes-like objects\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2f36bf96f650e121fb845e2f4058d6629496103e"},"cell_type":"code","source":"# Classes we will load\ncategories = ['cannon','eye', 'face', 'nail', 'pear','piano','radio','spider','star','sword']\n\n# Dictionary for URL and class labels\nURL_DATA = {}\nfor category in categories:\n    URL_DATA[category] = 'https://storage.googleapis.com/quickdraw_dataset/full/numpy_bitmap/' + category +'.npy'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5b5827969b65ad9efe5cbcc5ced6c0200a5a57d6"},"cell_type":"code","source":"x = URL_DATA['eye']\nresponse = requests.get(URL_DATA['eye'])\ny = np.load(BytesIO(response.content))\ny.shape\nim = np.reshape(y, (125888, 28, 28))\nim1 = im[5]\nim2 = cv2.resize(im1,(128,128))\n#cv2.imshow(im[0])\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9449f6329bbe2c97fbafd6fa2b41e609391bd310"},"cell_type":"code","source":"w, h = 28, 28\ndata = np.zeros((h, w), dtype=np.uint8)\nimg = Image.fromarray(im2, 'P')\nimg","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65cdbb982258549f26c37edb2a90d65ad288e34c"},"cell_type":"code","source":"classes_dict = {}\nfor key, value in URL_DATA.items():\n    response = requests.get(value)\n    classes_dict[key] = np.load(BytesIO(response.content))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d2173b78d83de5d7ca60942465ce54b52ca66b89"},"cell_type":"code","source":"for i, (key, value) in enumerate(classes_dict.items()):\n    value = value.astype('float32')/255.\n    if i == 0:\n        classes_dict[key] = np.c_[value, np.zeros(len(value))]\n    else:\n        classes_dict[key] = np.c_[value,i*np.ones(len(value))]\n\n# Create a dict with label codes\nlabel_dict = {0:'cannon',1:'eye', 2:'face', 3:'nail', 4:'pear', \n              5:'piana',6:'radio', 7:'spider', 8:'star', 9:'sword'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84b477d35b42e809fe8191a46560ab6627d55176"},"cell_type":"code","source":"lst = []\nfor key, value in classes_dict.items():\n    lst.append(value[:3000])\ndoodles = np.concatenate(lst)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"34780bb74df1b0b17531bdab620f996d238f44f9"},"cell_type":"code","source":"doodles.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c7d772a775c2d4117f3325f6bc6dbe52e3c8b67"},"cell_type":"code","source":"# Split the data into features and class labels (X & y respectively)\ny = doodles[:,-1].astype('float32')\nX = doodles[:,:784]\n\n# Split each dataset into train/test splits\nX_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.3,random_state=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec58836e865915e55bd47d49e5ffa9f53e5fe4e3"},"cell_type":"code","source":"X_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54f516f0cb568e2bda68577f0bb57f7682459e95"},"cell_type":"code","source":"train_X = np.reshape(X_train, (21000, 28, 28))\ntrain_X.shape\nim1 = train_X[20]\nim1 = cv2.resize(im1, (256,256))\n#w, h = 28, 28\n#data = np.zeros((h, w), dtype=np.uint8)\nimg = Image.fromarray(im1, 'P')\nimg","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"62358057794eff759391e148127d0f0614e3fbdd"},"cell_type":"markdown","source":"## loading data and category names\n"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"fnames = glob('../input/train_simplified/*.csv')\ncnames = ['countrycode', 'drawing', 'key_id', 'recognized', 'timestamp', 'word']\ndrawlist = []\nfor f in fnames[0:6]:\n    first = pd.read_csv(f, nrows=10) # make sure we get a recognized drawing\n    first = first[first.recognized==True].head(2)\n    drawlist.append(first)\ndraw_df = pd.DataFrame(np.concatenate(drawlist), columns=cnames)\ndraw_df","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cc883bc6e072549e5b8533f5282e58055244acf8"},"cell_type":"markdown","source":"## Adding underscore"},{"metadata":{"trusted":true,"_uuid":"b7c974763144af95c12659ac6041eedf93af6135"},"cell_type":"code","source":"#%% set label dictionary and params\nclassfiles = os.listdir('../input/train_simplified/')\nnumstonames = {i: v[:-4].replace(\" \", \"_\") for i, v in enumerate(classfiles)} #adds underscores\n\nnum_classes = 340    #340 max \nimheight, imwidth = 64, 64  \nims_per_class = 500  #max?","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"54d8ead9e6feb9d118243388cab709700d482c14"},"cell_type":"markdown","source":"## resampling the images, normalize and enumerate classes"},{"metadata":{"trusted":true,"_uuid":"839511d63a4bef80881f9a5a8f6f68ac5a5ad3db","scrolled":true},"cell_type":"code","source":"def draw_it(strokes):\n    image = Image.new(\"P\", (256,256), color=255)\n    image_draw = ImageDraw.Draw(image)\n    for stroke in ast.literal_eval(strokes):\n        for i in range(len(stroke[0])-1):\n            image_draw.line([stroke[0][i], \n                             stroke[1][i],\n                             stroke[0][i+1], \n                             stroke[1][i+1]],\n                            fill=0, width=5)\n    image = image.resize((imheight, imwidth))\n    return np.array(image)/255.\n\n#%% get train arrays\ntrain_grand = []\nclass_paths = glob('../input/train_simplified/*.csv')\nfor i,c in enumerate(tqdm(class_paths[0: num_classes])):\n    train = pd.read_csv(c, usecols=['drawing', 'recognized'], nrows=ims_per_class*5//4)\n    train = train[train.recognized == True].head(ims_per_class)\n    imagebag = bag.from_sequence(train.drawing.values).map(draw_it) \n    trainarray = np.array(imagebag.compute())  # PARALLELIZE\n    trainarray = np.reshape(trainarray, (ims_per_class, -1))    \n    labelarray = np.full((train.shape[0], 1), i)\n    trainarray = np.concatenate((labelarray, trainarray), axis=1)\n    train_grand.append(trainarray)\n    \ntrain_grand = np.array([train_grand.pop() for i in np.arange(num_classes)]) #less memory than np.concatenate\ntrain_grand = train_grand.reshape((-1, (imheight*imwidth+1)))\n\ndel trainarray\ndel train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"95677ff791249bf887e5fe001a0f4045280bc442"},"cell_type":"code","source":"cv2.imshow('img',draw_it(draw_df.drawing[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"152a487a00b95670421b36d83d00fb4d894a51f2"},"cell_type":"code","source":"print(train_grand.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e75a5f981d1d7f4ca9a816c1a92f61a24c824476"},"cell_type":"markdown","source":"## Train, valid split"},{"metadata":{"trusted":true,"_uuid":"bc12e3260a1117971b804fe6ae47f2e668a5a6fb"},"cell_type":"code","source":"# memory-friendly alternative to train_test_split?\nvalfrac = 0.1\ncutpt = int(valfrac * train_grand.shape[0])\n\nnp.random.shuffle(train_grand)\ny_train, X_train = train_grand[cutpt: , 0], train_grand[cutpt: , 1:]\ny_val, X_val = train_grand[0:cutpt, 0], train_grand[0:cutpt, 1:] #validation set is recognized==True\n\ndel train_grand\n\ny_train = keras.utils.to_categorical(y_train, num_classes)\nX_train = X_train.reshape(X_train.shape[0], imheight, imwidth, 1)\ny_val = keras.utils.to_categorical(y_val, num_classes)\nX_val = X_val.reshape(X_val.shape[0], imheight, imwidth, 1)\n\nprint(y_train.shape, \"\\n\",\n      X_train.shape, \"\\n\",\n      y_val.shape, \"\\n\",\n      X_val.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"061d69cdf2be1ca3dd4e98b26a2cf23be76c30bb"},"cell_type":"code","source":"def to_rgb(img):\n    img = cv2.resize(img, 64, interpolation = cv2.INTER_AREA) \n    img_rgb = np.asarray(np.dstack((img, img, img)), dtype=np.uint8)\n    return img_rgb\n\nrgb_list = []\n#convert X_train data to 48x48 rgb values\nfor i in range(X_train.shape[0]):\n    rgb = to_rgb(X_train[i])\n    rgb_list.append(rgb)\n    #print(rgb.shape)\n    \nrgb_arr = np.stack([rgb_list],axis=4)\nrgb_arr_to_3d = np.squeeze(rgb_arr, axis=4)\nprint(rgb_arr_to_3d.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fa88427ffe3def83f17964cbd49b00b79441bac1"},"cell_type":"markdown","source":"## Initial conv. net, basic, from scratch"},{"metadata":{"trusted":true,"_uuid":"16c478114ebf2d1676faddb362aa0791e7d702b9","scrolled":false},"cell_type":"code","source":"from keras.layers import GlobalAveragePooling2D\nmodel = Sequential()\nmodel.add(Conv2D(filters =16,kernel_size = 2, padding = 'same',activation = 'relu',input_shape=(28,28,1)))\nmodel.add(MaxPooling2D(pool_size=2))\n\nmodel.add(Conv2D(filters =32,kernel_size = 2, padding = 'same',activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=2))\n\nmodel.add(Conv2D(filters =64,kernel_size = 2, padding = 'same',activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Dropout(0.3))\n\nmodel.add(Conv2D(filters =16,kernel_size = 2, padding = 'same',activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Dropout(0.3))\n\n\n\nmodel.add(Conv2D(filters =16,kernel_size = 2, padding = 'same',activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Dropout(0.3))\n\n#model.add(GlobalAveragePooling2D())\nmodel.add(Flatten())\n\nmodel.add(Dense(10, activation='softmax'))\n\n### TODO: Define your architecture.\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bc475c70eec318a4b184e60ef0c0a7b8b8f8db6d"},"cell_type":"code","source":"model.compile(optimizer='rmsprop', loss='categorical_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c3c02e75f956569e5d65a9411721316acdb63ac1","collapsed":true},"cell_type":"code","source":"#from keras.callbacks import ModelCheckpoint  \n\n### TODO: specify the number of epochs that you would like to use to train the model.\n\nepochs = 50\n\n### Do NOT modify the code below this line.\n\ncheckpointer = ModelCheckpoint(filepath='../weights.best.from_scratch.hdf5', \n                               verbose=1, save_best_only=True)\n\nmodel.fit(x=X_train_mobile, y=y_train, \n          validation_data=(X_val_mobile, y_val),\n          epochs=50, batch_size=32, callbacks=[checkpointer], verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"89c0fd6b7fda08e72cf20d5d5bbfdc6da5d3e178"},"cell_type":"code","source":"model.load_weights('../weights.best.from_scratch.hdf5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ad58d048d12e96a121da1aaaa67f81362bafa1b"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b89f0db2cc34414e5e4de2ae311a18446effff84"},"cell_type":"code","source":"model_conv = torchvision.models.resnet18(pretrained=True)\nfor param in model_conv.parameters():\n    param.requires_grad = False\n\n# Parameters of newly constructed modules have requires_grad=True by default\nnum_ftrs = model_conv.fc.in_features\nmodel_conv.fc = nn.Linear(num_ftrs, 2)\n\nmodel_conv = model_conv.to(device)\n\ncriterion = nn.CrossEntropyLoss()\n\n# Observe that only parameters of final layer are being optimized as\n# opoosed to before.\noptimizer_conv = optim.SGD(model_conv.fc.parameters(), lr=0.001, momentum=0.9)\n\n# Decay LR by a factor of 0.1 every 7 epochs\nexp_lr_scheduler = lr_scheduler.StepLR(optimizer_conv, step_size=7, gamma=0.1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"858491c82dd0c3f27a75fdf0afef85bdcc7fa2ee"},"cell_type":"code","source":"model_conv = train_model(model_conv, criterion, optimizer_conv,\n                         exp_lr_scheduler, num_epochs=25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f61752baad7189b7b1a7b0df3cf9fe98c1341d43"},"cell_type":"code","source":"STEPS = 800\nEPOCHS = 16\nsize = 64\nbatchsize = 680","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf928ac345d9a7a408c4cb92b38955d248502b05","scrolled":true},"cell_type":"code","source":"from keras.applications.vgg19 import VGG19\nfrom keras.models import Model\nfrom keras.layers import Dense, GlobalAveragePooling2D\nfrom keras import backend as K\n#import inception with pre-trained weights. do not include fully #connected layers\nmodel = VGG19(input_shape=(size, size, 1), include_top=False,weights='imagenet')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"0d0256398dd5ebfe3e022364adde22c969d1eed0"},"cell_type":"code","source":"model.compile(optimizer='Adam', loss='categorical_crossentropy',\n              metrics=['accuracy'])\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"71fc49a33018898be8c0d2e798338d3d8151cc71"},"cell_type":"code","source":"# add a global spatial average pooling layer\nx = inception_base.output\nx = GlobalAveragePooling2D()(x)\n# add a fully-connected layer\nx = Dense(512, activation='relu')(x)\n# and a fully connected output/classification layer\npredictions = Dense(340, activation='softmax')(x)\n# create the full network so we can train on it\ninception_transfer = Model(input=inception_base.input, output=predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4cea9d09fd5960edaf745094fe2c1aa52f5a5e8e"},"cell_type":"code","source":"model.fit(x=X_train, y=y_train, \n          validation_data=(X_val, y_val),\n          epochs=50, batch_size=32)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"af005ff1cbe24b49c99eebdf41618719ae41fdd7"},"cell_type":"markdown","source":"## loading test data and prediction"},{"metadata":{"trusted":true,"_uuid":"ad999e70efd50b9a21c29794dc4f4ab8477d7ab2"},"cell_type":"code","source":"#%% get test set\nttvlist = []\nreader = pd.read_csv('../input/test_simplified.csv', index_col=['key_id'],\n    chunksize=2048)\nfor chunk in tqdm(reader, total=55):\n    imagebag = bag.from_sequence(chunk.drawing.values).map(draw_it)\n    testarray = np.array(imagebag.compute())\n    testarray = np.reshape(testarray, (testarray.shape[0], imheight, imwidth, 1))\n    testpreds = model.predict(testarray, verbose=0)\n    ttvs = np.argsort(-testpreds)[:, 0:3]  # top 3\n    ttvlist.append(ttvs)\n    \nttvarray = np.concatenate(ttvlist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b4fa4cb0e7d8df6d7de75e372b58cef9c70d324c"},"cell_type":"code","source":"print(testarray.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0d8b8a64d475a5cd744b67ca5938185a1c2e4d98"},"cell_type":"markdown","source":"## Submission"},{"metadata":{"trusted":true,"_uuid":"bb894cd342cf887bab56f7e2a1e54fccede86d4b"},"cell_type":"code","source":"preds_df = pd.DataFrame({'first': ttvarray[:,0], 'second': ttvarray[:,1], 'third': ttvarray[:,2]})\npreds_df = preds_df.replace(numstonames)\npreds_df['words'] = preds_df['first'] + \" \" + preds_df['second'] + \" \" + preds_df['third']\n\nsub = pd.read_csv('../input/sample_submission.csv', index_col=['key_id'])\nsub['word'] = preds_df.words.values\nsub.to_csv('subcnn_small.csv')\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5405ef04e803dad8c9596777a43a46e3b3c53f37"},"cell_type":"code","source":"# get index of predicted dog breed for each image in test set\ndoodle_predictions = [np.argmax(model.predict(np.expand_dims(tensor, axis=0))) for tensor in test_X]\n\n# report test accuracy\ntest_accuracy = 100*np.sum(np.array(doodle_predictions)==np.argmax(test_y, axis=1))/len(doodle_predictions)\nprint('Test accuracy: %.4f%%' % test_accuracy)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"588b1a4fb4139c7bbcede860f5b670729b151322"},"cell_type":"code","source":"from keras.applications.inception_v3 import InceptionV3\nfrom keras.applications.xception import Xception\nfrom keras.applications.resnet50 import ResNet50\nfrom keras.applications.vgg19 import VGG19","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b774d73f07c708af2fa807388d19f77b752a862"},"cell_type":"code","source":"pretrained_model = InceptionV3(include_top=False,input_shape=IN_SHAPE,weights='imagenet')\n\nif pretrained_model.output.shape.ndims > 2:\n    output = Flatten()(pretrained_model.output)\nelse:\n    output = pretrained_model.output\n\noutput = BatchNormalization()(output)\noutput = Dropout(0.5)(output)\noutput = Dense(340, activation='softmax')(output)\n\nmodel = Model(pretrained_model.input, output)\nfor layer in pretrained_model.layers:\n    layer.trainable = False\n\nmodel.summary(line_length=200)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}