{"cells":[{"metadata":{"_uuid":"441b60727b57fb7f9fd70a9258f34b0990a4d750"},"cell_type":"markdown","source":"<h1 style=\"color:steelblue; font-family:Ewert; font-size:150%;\" class=\"font-effect-3d\">Code Library, Style and Links</h1>\n\nThe previous notebook - [Quick, Draw! Doodle Recognition 1](https://www.kaggle.com/olgabelitskaya/quick-draw-doodle-recognition-1)","execution_count":null},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"9631c90951a73584cd7a0262a5dac92ab5ca79ff"},"cell_type":"code","source":"%%html\n<style>\n@import url('https://fonts.googleapis.com/css?family=Ewert|Roboto&effect=3d|ice|');\nspan {font-family:'Roboto'; color:black; text-shadow: 5px 5px 5px #aaa;}  \ndiv.output_area pre{font-family:'Roboto'; font-size:110%; color: steelblue;}      \n</style>","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np,pandas as pd,keras as ks\nimport os,ast,warnings\nimport pylab as pl\nfrom skimage.transform import resize\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix,\\\nclassification_report\nfrom keras.callbacks import ModelCheckpoint,\\\nReduceLROnPlateau\nfrom keras.models import Sequential\nfrom keras.layers.advanced_activations import LeakyReLU\nfrom keras.layers import Activation,Dropout,Dense,\\\nConv2D, MaxPooling2D, GlobalMaxPooling2D\nwarnings.filterwarnings('ignore')\npl.style.use('seaborn-whitegrid')\nstyle_dict={'background-color':'gainsboro','color':'steelblue', \n            'border-color':'white','font-family':'Roboto'}\nos.listdir(\"../input\")","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"trusted":true,"_uuid":"2c57ebf951bcabcc8773e85e9c803864ce30a20b","collapsed":true},"cell_type":"code","source":"fpath='../input/quickdraw-doodle-recognition/train_simplified/'\nfiles=os.listdir(fpath)\nlabels=[el.replace(\" \",\"_\")[:-4] for el in files]\nprint(sorted(labels)) # 340 labels - 17 sets","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c3cc1a6db60e14276eff3894576af85c6c7133f2"},"cell_type":"code","source":"wpath='../input/quick-draw-model-weights-for-doodle-recognition/weights/'\nweights=sorted(os.listdir(wpath))\nprint(weights) # files with weights for 17 label sets","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6a300203fe003248ffc2d97005e667678fbca692"},"cell_type":"code","source":"I=64 # image size in pixels\nT=20 # number of labels in one set","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"_kg_hide-output":false,"_kg_hide-input":true},"cell_type":"code","source":"# https://stackoverflow.com/questions/25837544/get-all-points-of-a-straight-line-in-python\ndef get_line(x1,y1,x2,y2):\n    steep=abs(y2-y1)>abs(x2-x1)\n    if steep: x1,y1,x2,y2=y1,x1,y2,x2\n    rev=False\n    if x1>x2:\n        x1,x2,y1,y2=x2,x1,y2,y1\n        rev=True\n    dx=x2-x1; dy=abs(y2-y1)\n    error=int(dx/2)\n    xy=[]; y=y1; ystep=None\n    if y1<y2: ystep=1\n    else: ystep=-1\n    for x in range(x1,x2+1):\n        if steep: xy.append([y,x])\n        else: xy.append([x,y])\n        error-=dy\n        if error<0:\n            y+=ystep\n            error+=dx\n    if rev: xy.reverse()\n    return xy\ndef display_drawing():\n    pl.figure(figsize=(10,10))\n    pl.suptitle('Test Pictures')\n    for i in range(20):\n        picture=ast.literal_eval(test_data.drawing.values[i])\n        for x,y in picture:\n            pl.subplot(5,4,i+1)\n            pl.plot(x,y,'-o',markersize=1,color='slategray')\n            pl.xticks([]); pl.yticks([])\n            pl.title(submission.iloc[i][1])\n        pl.gca().invert_yaxis()\n        pl.axis('equal');           \ndef get_image(data,k,I=I):\n    img=np.zeros((280,280))\n    picture=ast.literal_eval(data.values[k])\n    for x,y in picture:\n        for i in range(len(x)):\n            img[y[i]+10][x[i]+10]=1\n            if (i<len(x)-1):\n                x1,y1,x2,y2=x[i],y[i],x[i+1],y[i+1]\n            else:\n                x1,y1,x2,y2=x[i],y[i],x[0],y[0]\n            for [xl,yl] in get_line(x1,y1,x2,y2):\n                img[yl+10][xl+10]=1                \n    return resize(img,(I,I))    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"722f9a75649da365bfd167eefc84b2c5fef2bb3b"},"cell_type":"markdown","source":"<h1 style=\"color:steelblue; font-family:Ewert; font-size:150%;\" class=\"font-effect-3d\">Test Data Exploration</h1>","execution_count":null},{"metadata":{"_kg_hide-output":true,"trusted":true,"_uuid":"1238c5c6ebf9656d985ee70d7baa18d84a5a1892"},"cell_type":"code","source":"ftest='../input/quickdraw-doodle-recognition/test_simplified.csv'\ntest_data=pd.read_csv(ftest,index_col='key_id')\ntest_data.tail(3).T.style.set_properties(**style_dict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc55e1c2602a083a82d47346eb97207ab502c1e7"},"cell_type":"code","source":"# creating test images in pixels\ntest_images=[]\ntest_images.extend([get_image(test_data.drawing,i) \n                    for i in range(len(test_data))])\ntest_images=np.array(test_images)\ntest_images.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"00b5af38d045403aecfc499376d23b0890894c38","_kg_hide-input":true},"cell_type":"code","source":"pl.figure(figsize=(10,5))\npl.subplot(1,2,1); pl.imshow(test_images[0])\npl.subplot(1,2,2); pl.imshow(test_images[10000])\npl.suptitle('Key Points in the Test Pictures');","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"554956eb7052dc0a31ed9e4116e0753f48a969ef"},"cell_type":"markdown","source":"<h1 style=\"color:steelblue; font-family:Ewert; font-size:150%;\" class=\"font-effect-3d\">The Model</h1>","execution_count":null},{"metadata":{"trusted":true,"_uuid":"1ee1f60653b588e95be394711603d9f134bdf584"},"cell_type":"code","source":"def model():\n    model=Sequential()  \n    model.add(Conv2D(32,(5,5),padding='same',\n                     input_shape=(I,I,1)))\n    model.add(LeakyReLU(alpha=.02))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Dropout(.2))\n    model.add(Conv2D(196,(5,5)))\n    model.add(LeakyReLU(alpha=.02))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Dropout(.2))\n    model.add(GlobalMaxPooling2D())\n    model.add(Dense(1024))\n    model.add(LeakyReLU(alpha=.02))\n    model.add(Dropout(.5)) \n    model.add(Dense(T))\n    model.add(Activation('softmax'))\n    model.compile(loss='categorical_crossentropy',\n                  optimizer='adam',metrics=['accuracy'])\n    return model\nmodel=model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cede7f76a00046e868a664418243a5d614e3b845"},"cell_type":"markdown","source":"<h1 style=\"color:steelblue; font-family:Ewert; font-size:200%;\" class=\"font-effect-3d\">Predictions</h1>","execution_count":null},{"metadata":{"trusted":true,"_uuid":"cfa14d6ac7dd73687b9b84140a3edb4bbdd94745"},"cell_type":"code","source":"fn1='../input/quick-draw-model-weights-for-doodle-recognition/'+\\\n    'weights/weights.best.model001_020.hdf5'\nmodel.load_weights(fn1)\ntest_predictions=model.predict(test_images.reshape(-1,I,I,1))\ntest_predictions.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25af01d621cf799cc56b3ed24bc0da12dfbc85b3"},"cell_type":"code","source":"# separated predictions for each label set\nfor w in weights[1:]:\n    w=wpath+w\n    model.load_weights(w)\n    test_predictions2=model.predict(test_images.reshape(-1,I,I,1))\n    test_predictions=np.concatenate((test_predictions,\n                                     test_predictions2),axis=1)\ntest_predictions.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d35b3a790e1d63dc61f785167299fc0290c5a91","_kg_hide-output":true},"cell_type":"code","source":"# 3 best guesses among all label sets\ntest_labels=[[labels[i] \n              for i in test_predictions[k].argsort()[-3:][::-1]] \n             for k in range(len(test_predictions))]\ntest_labels=[\" \".join(test_labels[i]) \n             for i in range(len(test_labels))]\nsubmission=pd.DataFrame({\"key_id\":test_data.index,\n                         \"word\":test_labels})\nsubmission.to_csv('qd_submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7782a5e2d3546e23c036532e879adad99d44e438"},"cell_type":"code","source":"display_drawing()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}