{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport json\nimport gc\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom glob import glob\nimport os\ntrain_files = glob(\"../input/quickdraw-doodle-recognition/train_simplified/*.csv\")\nrows = 150000\nrows = rows - (rows % 340)\ncat_size = rows // 340\nprint(cat_size)\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image, ImageDraw\nfrom dask import bag\ndef drawStrokes(matrixOfStrokes):\n    image = Image.new(\"RGB\", (256,256), color=255)\n    image_draw = ImageDraw.Draw(image)\n    for stroke in json.loads(matrixOfStrokes):\n        for i in range(len(stroke[0])-1):\n            image_draw.line([stroke[0][i], \n                             stroke[1][i],\n                             stroke[0][i+1], \n                             stroke[1][i+1]],\n                            fill=0, width=5)\n    return np.array(image.resize((32,32)))/255.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"drawingArray = np.zeros((rows,32,32,3))\ncategories = pd.Series([None] * rows)\ni = 0\nfor f in train_files:\n    for df in pd.read_csv(f, index_col=\"key_id\", chunksize=1000, nrows=cat_size):\n        imagebag = bag.from_sequence(df.drawing.values).map(drawStrokes)\n        imagebag = np.array(imagebag.compute())\n        categories[i:(i + imagebag.shape[0])] = df[\"word\"].replace(\"\\s+\", \"_\", regex=True)\n        drawingArray[i:(i + imagebag.shape[0])] = imagebag\n        i += imagebag.shape[0]\n        print(i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nindecator = pd.get_dummies(categories)\ntr_x,tst_x,tr_indecator,tst_indecator = train_test_split(drawingArray\n                                                           , indecator\n                                                           , test_size=0.2\n                                                           ,random_state=25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del drawingArray,categories\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import keras\nfrom keras.models import Sequential,Input,Model\nfrom keras.layers import Dense,Dropout,Flatten,Conv2D,MaxPooling2D,Activation\n\nmodel = Sequential()\nmodel.add(Conv2D(64, kernel_size=(4,4), strides=1, input_shape=(32,32,3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.3))\n\nmodel.add(Conv2D(128, kernel_size=(4,4), strides=1))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.3))\n\nmodel.add(Flatten())\nmodel.add(Dense(1024))\nmodel.add(Activation('relu'))\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.3))\n\nmodel.add(Dense(340))\nmodel.add(Activation('softmax'))\n\nmodel.compile(loss='categorical_crossentropy',\n              optimizer='adam',\n              metrics=['accuracy'])\n\nhistory = model.fit(tr_x, tr_indecator,batch_size=200,epochs=30\n          ,validation_data=(tst_x,tst_indecator))\n\nprint(history.history.keys())\n# summarize history for accuracy\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()\n# summarize history for loss\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#del tr_x,tst_x,tr_indecator,tst_indecator\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test = pd.read_csv('../input/quickdraw-doodle-recognition/test_simplified.csv', index_col=\"key_id\" ,nrows=100)\nids = test.index\nimagebag = bag.from_sequence(test.drawing.values).map(drawStrokes)\ntest_simplified = np.array(imagebag.compute())\ntest_simplified = test_simplified.reshape(len(test_simplified), 32, 32, 3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del imagebag\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"prediction = model.predict(test_simplified)\nindexOfBigProbability = (-prediction).argsort()[:,:3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline\nimport ast\nimport warnings\nwarnings.filterwarnings('ignore')\n\nraw_images = [ast.literal_eval(lst) for lst in test.loc[test.iloc[:40].index, 'drawing'].values]\nj=0\nfor index, raw_drawing in enumerate(raw_images):\n    plt.figure(figsize=(3,3))\n    for x,y in raw_drawing:\n        title_obj=plt.title(indecator.columns[indexOfBigProbability][j][0]\n                  +\"  \"\n                 +indecator.columns[indexOfBigProbability][j][1]\n                  +\"  \"\n                 +indecator.columns[indexOfBigProbability][j][2], fontsize=22)\n        plt.setp(title_obj, color='green')\n        plt.subplot(1, 1, 1)\n        plt.plot(x,y)\n        plt.axis('off')\n    plt.gca().invert_yaxis()\n    j+=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense,Flatten,Activation\nfrom keras.applications.vgg16 import VGG16\n\nvgg16 = keras.applications.vgg16.VGG16(weights='imagenet',classes=340,include_top=False,input_shape=(32,32,3))\nvgg16.summary()\nm = Sequential()\nfor layer in vgg16.layers:\n    m.add(layer)\nfor layer in m.layers:\n    layer.trainable = False\nm.add(Flatten())\nm.add(Dense(4096,activation='relu'))\nm.add(Dense(4096,activation='relu'))\nm.add(Dense(340,activation='softmax'))\nm.summary()\nm.compile(loss='categorical_crossentropy',optimizer='adam',metrics=['accuracy'])\nhistory1 = m.fit(tr_x, tr_indecator,batch_size=1024,epochs=23,validation_data=(tst_x,tst_indecator))\n\nprint(history1.history.keys())\n# summarize history for accuracy\nplt.plot(history1.history['accuracy'])\nplt.plot(history1.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()\n# summarize history for loss\nplt.plot(history1.history['loss'])\nplt.plot(history1.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred = m.predict(test_simplified)\nind = (-pred).argsort()[:,:3]\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline\nimport ast\nimport warnings\nwarnings.filterwarnings('ignore')\n\nraw_images = [ast.literal_eval(lst) for lst in test.loc[test.iloc[:80].index, 'drawing'].values]\nj=0\nfor index, raw_drawing in enumerate(raw_images):\n    plt.figure(figsize=(3,3))\n    for x,y in raw_drawing:\n        title_obj=plt.title(indecator.columns[ind][j][0]\n                  +\"  \"\n                 +indecator.columns[ind][j][1]\n                  +\"  \"\n                 +indecator.columns[ind][j][2], fontsize=22)\n        plt.setp(title_obj, color='green')\n        plt.subplot(1, 1, 1)\n        plt.plot(x,y)\n        plt.axis('off')\n    plt.gca().invert_yaxis()\n    j+=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.plot(history1.history['accuracy'])\nplt.plot(history1.history['val_accuracy'])\nplt.title('CNN vs VGG16 accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['trainCNN', 'testCNN','trainVGG16','testVGG16'], loc='upper left')\nplt.show()\n# summarize history for loss\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.plot(history1.history['loss'])\nplt.plot(history1.history['val_loss'])\nplt.title('CNN vs VGG16 loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['trainCNN', 'testCNN','trainVGG16','testVGG16'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}