{"cells":[{"metadata":{"_uuid":"3c1b56346612d3d6d9f3bd466feca2a2108ddb56"},"cell_type":"markdown","source":"## References\n1. [kernel: simple eda](https://www.kaggle.com/dimitreoliveira/quick-draw-simple-eda)\n"},{"metadata":{"trusted":true,"_uuid":"117f142a8cda5e44351ae2106e20cbc34f1c784e"},"cell_type":"code","source":"import os\nimport ast\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.utils import shuffle\nimport datetime\n \n\npd.options.display.max_rows = 20\nsns.set(style=\"darkgrid\")\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"train_sample = pd.DataFrame()\nfiles_directory = os.listdir(\"../input/train_simplified\")\nfiles_directory[:10], len(files_directory) # count labels","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"for file in files_directory:\n    train_sample = train_sample.append(pd.read_csv('../input/train_simplified/' + file, index_col='key_id', nrows=10))\n# Shuffle data\ntrain_sample = shuffle(train_sample, random_state=123)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"print(train_sample.shape)\ntrain_sample.head(n=30)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":false},"cell_type":"code","source":"# because of momory and speed problem, limit the numbers of each label data 10000 rows\ntrain = pd.DataFrame()\nfor i, file in enumerate(files_directory):\n    train = train.append(pd.read_csv('../input/train_simplified/' + file, index_col='key_id', usecols=[1, 2, 3, 5], nrows=10000))\n    if i % 20 == 0:\n        print(i, \"[Done]\", datetime.datetime.now(), file)\n\n# Shuffle data\nprint(\"[Start] Shuffle!\")\ntrain = shuffle(train, random_state=123)\nprint(train.shape) # tooooooooooooooooooooooooooooooooooooooooooooooooo large, and this just have only 184 label files\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"e5bf819b0e3fe7125a388a56075cc1fecbff38c9"},"cell_type":"code","source":"print('Train number of rows: ', train.shape[0])\nprint('Train number of columns: ', train_sample.shape[1])\nprint('Train set features: %s' % train_sample.columns.values)\nprint('Train number of label categories: %s' % len(files_directory))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"472da6a149a728bfc13b1a50dad461b5ff6a999a"},"cell_type":"code","source":"sns.countplot(x=\"recognized\", data=train)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46515fb612fa201e1a183bbfd0e8513d044073be"},"cell_type":"code","source":"rec_gp = train.groupby(['word', 'recognized']).size().reset_index(name='count')\nrec_true = rec_gp[(rec_gp['recognized'] == True)].rename(index=str, columns={\"recognized\": \"recognized_true\", \"count\": \"count_true\"})\nrec_false = rec_gp[(rec_gp['recognized'] == False)].rename(index=str, columns={\"recognized\": \"recognized_false\", \"count\": \"count_false\"})\nrec_gp = rec_true.set_index('word').join(rec_false.set_index('word'), on='word')\nrec_gp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"6f21b214822c537e063c74cf923675c5a22278d7"},"cell_type":"code","source":"words = train['word'].tolist()\ndrawings = [ast.literal_eval(pts) for pts in train[:9]['drawing'].values]\nwords[:10], drawings[-1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8517cf37540aff79304956addd4d4f572e2ef9f"},"cell_type":"code","source":"len(words)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a740f43143c444e7aba741ac3f43c11a3f37249"},"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nfor i, drawing in enumerate(drawings):\n    plt.subplot(330 + (i+1))\n    for x,y in drawing:\n        plt.plot(x, y, marker='.')\n        plt.tight_layout()\n        plt.title(words[i]);\n        plt.axis('off')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"098f51e27728413178d1aad4444bdf4b3eb5517b"},"cell_type":"code","source":"print(eval(owls_recognized['drawing'][0])) # drawing is string, so using eval function to convert list types","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}