{"cells":[{"metadata":{"trusted":true,"_uuid":"e39a4491bd4875ee133bac32ab3f705287f89ea7"},"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport numpy as np\nimport re \nimport os\nfrom glob import glob\nimport matplotlib.pyplot as plt\nimport ast\n\n%matplotlib inline\n\nplt.style.use('seaborn') #make plots prettier","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a2762d16e22b901e435bf50106efb1a4c61cd03b"},"cell_type":"markdown","source":"# EDA and Preprocessing\n\nIn this notebook there is a short EDA, and some preprocessing of the simplified training data to numpy arrays."},{"metadata":{"_uuid":"0728e4c0598186155878b01a4c0d7532ce01e868"},"cell_type":"markdown","source":"# File Paths"},{"metadata":{"trusted":true,"_uuid":"bcfef66315caa00f51187c3534b7aae6f0720ce3"},"cell_type":"code","source":"train_dir =  \"../input/train_simplified/\"\ncsv_files = glob(train_dir + \"*.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7ad07595b8a53fdc7b48e63fd8c58d24912e903d"},"cell_type":"markdown","source":"# Training Classes\n\nI wanted to find out how evenly distributed the classes in the training examples are."},{"metadata":{"trusted":true,"_uuid":"d824efc3d75d8adddeb2fbc3497c916dc7269f8c"},"cell_type":"code","source":"def extract_classname(filename):\n    return re.search( r\"fied/(.+)\\.csv\",filename).group(1) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35266cc375e6f047b61ba215e989a1e468af1b5b"},"cell_type":"code","source":"class_names = [ extract_classname(file) for file in csv_files]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ebb71864ff9b60b9b602d9287f4dcc16409a6f0f"},"cell_type":"code","source":"def count_lines(fname):\n    with open(fname) as f:\n        for i, l in enumerate(f):\n            pass\n    return i - 1 #minus one for header row\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e8ce6e9c1995ece405051daf173a3428ba71141d"},"cell_type":"code","source":"%%time\n#is a little slow, many lines to count\nline_counts = [ count_lines(file) for file in csv_files]\ncounts = pd.DataFrame({\"line_counts\":line_counts})\ncounts[\"class\"] = class_names","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b148e45c51c098aaffe69171cf97dd5a1bae02f4"},"cell_type":"markdown","source":"Certain classes are far more common it seems."},{"metadata":{"trusted":true,"_uuid":"2c855f31bb3733cec07462450b14d056854f1550"},"cell_type":"code","source":"sns.distplot(counts.line_counts.values,kde=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd0e1058c623f385a8985a4eb9c909cf69034c7e"},"cell_type":"code","source":"counts.describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"328fd297d77f11e5d17a7b16bff9bd0c0a8f0d4a"},"cell_type":"markdown","source":"The most common classes are shown bellow"},{"metadata":{"trusted":true,"_uuid":"462de30ea82f54b42660b24fcb27ae46f26f986e"},"cell_type":"code","source":"i = counts.line_counts.nlargest(10).index\nax = counts.iloc[i].line_counts.plot.bar()\nax.set_xticklabels(counts.loc[i,\"class\"]);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3d2eb07bf4e254c347dc90b12d41b040004eda26"},"cell_type":"markdown","source":"# Reading Data\n\nSince each csv file contains a lot of rows, we'll just read in a smaller sample"},{"metadata":{"trusted":true,"_uuid":"2f13e86725611a02c4aa8415566cbfd060ccfd6f"},"cell_type":"code","source":"def read_csvs(csv_files, nrows=1000):\n    df =  pd.concat([ pd.read_csv(file,nrows=nrows) for file in csv_files])\n    df.reset_index(inplace=True,drop=True)\n    df['drawing'] = df.drawing.apply(ast.literal_eval)\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"38ec69b0a52504e46d377bd0718eaaeef35feec5"},"cell_type":"code","source":"%%time\n#takes a litte while\ndf = read_csvs(csv_files)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"632432fca5c3fed9391aa8af3f6e106dfb059e0a"},"cell_type":"markdown","source":"# Number of strokes\n\nI reckon that the number of strokes used could be a good indicator to the model which type of image is being drawn, since certain images will naturally take more or less strokes.\n\n"},{"metadata":{"trusted":true,"_uuid":"f6115a797bb298c4f54e82c96e912bbede129006"},"cell_type":"code","source":"df[\"n_strokes\"] = df.drawing.apply(len)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"330b83c918c60548855294eb0e76d01b3c0537f0"},"cell_type":"code","source":"s = df[df.word.str.contains(\"rabbit|sun|hot dog\")] #look at a subset of rabbit, sun and hot dog drawings","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"14ccff9f85b4c03a0ad0fdf18c9ad8f3047cdf2d"},"cell_type":"markdown","source":"The number of strokes looks almost normally distributed, with differences in the mean, depending on thing being drawn.  "},{"metadata":{"trusted":true,"_uuid":"1d6050c851c7cbf766e3d742fb5ac5e4bf8bdc38"},"cell_type":"code","source":"s = s.groupby('word').n_strokes.plot.kde()\ns.apply(lambda ax: ax.legend());","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6b686aa4a665df95552d481ba3ea896553144813"},"cell_type":"markdown","source":"# List 2 Numpy\n\nIn a later notebook I'll train a CNN but first I need to convert the list of points into an image."},{"metadata":{"trusted":true,"_uuid":"1eebba384a6f8dda1def6fbcab2a23f2b6e7df3b"},"cell_type":"code","source":"def list2numpy(points_list,size=1):\n\n    \"\"\"\n    Takes a list of points and converts it to a boolean\n    numpy array of size 72 by 72. Increase size to\n    double the output size.\n    \"\"\"\n    \n    fig, ax = plt.subplots(figsize=(size,size))\n    fig.tight_layout(pad=0)\n    ax.grid(False)\n    ax.xaxis.set_visible(False)\n    ax.yaxis.set_visible(False)\n    ax.set_axis_off()\n\n\n    for points in points_list:\n        ax.set_xlim(0,255)\n        ax.set_ylim(0,255)\n        ax.invert_yaxis()\n        ax.plot(points[0],points[1])\n\n    fig.canvas.draw()\n\n    X = np.array(fig.canvas.renderer._renderer)\n    plt.close()\n\n    return X[:,:,1] == 255 ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"98c8894a6047b3531b198f8fdf8b556b8df5003b"},"cell_type":"code","source":"plt.imshow(list2numpy(df.drawing[7],size=4),cmap=\"gray\")\nplt.axis('off')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f5aac7e98fd7ca15996e43c522070f12f11aa31"},"cell_type":"code","source":"#default size is 72 b 72\nlist2numpy(df.drawing[7],size=1).shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"82ab85e3ccaa44e84fb5701df747bde4b610d12b"},"cell_type":"code","source":"#but larger size possible\nlist2numpy(df.drawing[7],size=2).shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ba45b26b43f60bf9c9a179c8cdd2310d774d2b5d"},"cell_type":"markdown","source":"# Unrecognized vs Recognized Images\n\nI was curious as to what the unrecognized images looked like? Was it because they are poorly drawn, of the algorithm wasn't smart engough."},{"metadata":{"trusted":true,"_uuid":"80302e5861c8699194abb5c63a28980dcddd5643"},"cell_type":"code","source":"def plot_images(df, w = 5, h =5):\n\n    fig, axes = plt.subplots(w,h, figsize=(10,10))\n\n    for i, ax in enumerate(axes.flatten()):\n            ax.imshow(df.drawing[i], cmap=\"gray\")\n            ax.set_title(df.word[i])\n            ax.set_axis_off()\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"89ccc8e1e26dba229fcdd23cb810733c4eba003f"},"cell_type":"markdown","source":"For the most part it looks like many of these unrecognized images could be guessed by a human.  In some people have written the word instead of drawing a image."},{"metadata":{"trusted":true,"_uuid":"c4e69d9eff90b8a80549d5c689586d9c532eb9ad"},"cell_type":"code","source":"n = 5 # change me to plot more images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"099823009209df03b1eaed8a134bdb4b8c04229e"},"cell_type":"code","source":"#run cell a few times \nunrecognized = df[df.recognized == False].sample(n**2)\nunrecognized.reset_index(inplace=True,drop=True)\nunrecognized[\"drawing\"] =   unrecognized.drawing.apply(list2numpy)\nplot_images(unrecognized)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"897d096508462e1d2f60386d2eb4479bd20c5085"},"cell_type":"markdown","source":"I think mabye I'd say the recognized images tend to be drawn better"},{"metadata":{"trusted":true,"_uuid":"e8aee8c99503f9aa2f91cff5d870f0e5fe48a45a"},"cell_type":"code","source":"recognized = df[df.recognized == True].sample(n**2)\nrecognized.reset_index(inplace=True,drop=True)\nrecognized[\"drawing\"] =   recognized.drawing.apply(list2numpy)\nplot_images(recognized)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.5"}},"nbformat":4,"nbformat_minor":1}