{"cells":[{"metadata":{"_uuid":"81cbb010249759271915879660505b39d26d7fb7"},"cell_type":"markdown","source":"Quick, Draw! is an online game developed by Google that challenges players to draw a picture of an object. You can play the game here https://quickdraw.withgoogle.com/ The game prompts users to draw an image depicting a certain category, such as ”marker,” “table,” etc.  The aim of this competition is to build a better classifier for the existing Quick, Draw! dataset.  The challenging thing is that the data is very noisy.\n\nIn this kernel, I try to visualize all the images in the train data set to get some sense of the type of data we are dealing with\n"},{"metadata":{"_uuid":"5770e6ce0ae1e6a6560d7a4ddef49cc0e2c8d890"},"cell_type":"markdown","source":"# Load Libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport matplotlib.pylab\nfrom matplotlib.backends.backend_agg import FigureCanvasAgg as FigureCanvas\nfrom matplotlib.figure import Figure\n%pylab inline\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.utils import shuffle\nfrom tqdm import tqdm_notebook\nimport ast\n\nsns.set_style(\"white\")\nsns.set_context(\"notebook\", font_scale=1.5, rc={\"lines.linewidth\": 2.5})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7dc80d2bb5bfd6a2a432a7b049462bf25b57189"},"cell_type":"code","source":"train_path = '../input/train_simplified/'\nfiles = os.listdir(train_path)\ncategories = [category.split('.')[0] for category in files]\nprint('Total number of categories: ',len(categories))\nprint('Few Example Categories',categories[0:5])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9c002857e84297f5f923cf07714299ab8ca4db15"},"cell_type":"markdown","source":"# Reading data from all the categories"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_data = pd.DataFrame()\nfor file in tqdm_notebook(files):\n    train_data = train_data.append(pd.read_csv(train_path + file, index_col='word', nrows=10))    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da898a0e9f3f39400c6bf6bffa680b12acf622b5"},"cell_type":"code","source":"train_data.sample(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9e9f05bf658e76fd12ef9506ed440e7a862bff8c"},"cell_type":"code","source":"train_data = train_data.reset_index()\ntrain_data['word_count'] = train_data.groupby('word')['word'].transform('count')\nsns.distplot(train_data['word_count'],kde=False)\nplt.title('Word Count Distribution in Train Set')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3b5bb7f3f913845eb933f71faf5f954424d4c8cd"},"cell_type":"markdown","source":"# Generating the Images"},{"metadata":{"trusted":true,"_uuid":"466d29a1971c47ba480e600039ecb8ced0a477c9"},"cell_type":"code","source":"if train_data.index.name is not 'word':\n    train_data = train_data.set_index('word')\n    \nimg_ar = None\nfor cat in tqdm_notebook(categories):\n    df = train_data[train_data.index==cat]\n    drawings = [ast.literal_eval(pts) for pts in df[:9]['drawing'].values]\n\n    fig = Figure()\n    ax = fig.subplots(1,9)\n    canvas = FigureCanvas(fig)\n    for i, drawing in enumerate(drawings):\n        for x,y in drawing:\n            ax[i].plot(x, y, marker='.')\n            ax[i].axis('off')\n    fig.suptitle(cat,fontsize=30)\n#     plt.show()\n    canvas.draw()       # draw the canvas, cache the renderer\n    image = np.fromstring(canvas.tostring_rgb(), dtype='uint8')\n    width, height = fig.get_size_inches() * fig.get_dpi() \n    img = image.reshape(int(height), int(width), 3)\n    img = np.expand_dims(img,axis=0)\n    if img_ar is None:\n        img_ar = img\n    else:\n        img_ar = np.concatenate([img_ar,img],axis=0)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"419233f043af5c54b87d617d69a904a7e6bcbe74"},"cell_type":"markdown","source":"# Visualizing all the images"},{"metadata":{"trusted":true,"_uuid":"4c1fdb1968bf4d82cc3db5af94d2a68280e056f2"},"cell_type":"code","source":"DataRange = (np.absolute(img_ar)).max() \nEXTENT = [0, width, 0 ,height]\nNORM = matplotlib.colors.Normalize(vmin =-DataRange, vmax= DataRange, clip =True)\n\ngrid_width = 20\ngrid_height = len(categories)//grid_width\nfig,axs = plt.subplots(grid_height,grid_width,figsize=(img_ar.shape[1], img_ar.shape[2]))\nfor i in range(len(categories)):\n    ax = axs[int(i / grid_width), i % grid_width]\n    ax.imshow(img_ar[i], norm = NORM, extent = EXTENT, aspect = 1, interpolation='none')\n    ax.axis('off')\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d07d96117225ab9f004bfe8db7c2449b319a6b2"},"cell_type":"markdown","source":"# You can open the above image in new tab to get better resolution. The image may take some time to load.\nOR you can manually visualize each image like shown below"},{"metadata":{"trusted":true,"_uuid":"6d0af971ed19625d27a03f5fb44eb706fc5f5bb7"},"cell_type":"code","source":"plt.imshow(img_ar[0])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}