{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **数据分析**","metadata":{}},{"cell_type":"code","source":"import os\nimport ast\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.utils import shuffle\n\npd.options.display.max_rows = 20\nsns.set(style='darkgrid')","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:13:08.018009Z","iopub.execute_input":"2021-09-18T14:13:08.018349Z","iopub.status.idle":"2021-09-18T14:13:09.039276Z","shell.execute_reply.started":"2021-09-18T14:13:08.018265Z","shell.execute_reply":"2021-09-18T14:13:09.03838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input/quickdraw-doodle-recognition/train_simplified/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        break\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:13:12.056819Z","iopub.execute_input":"2021-09-18T14:13:12.057563Z","iopub.status.idle":"2021-09-18T14:13:12.207311Z","shell.execute_reply.started":"2021-09-18T14:13:12.057517Z","shell.execute_reply":"2021-09-18T14:13:12.20631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 导入简易化处理的数据","metadata":{}},{"cell_type":"code","source":"files_directory = os.listdir('/kaggle/input/quickdraw-doodle-recognition/train_simplified')\nlen(files_directory)","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:13:14.564759Z","iopub.execute_input":"2021-09-18T14:13:14.565184Z","iopub.status.idle":"2021-09-18T14:13:14.572737Z","shell.execute_reply.started":"2021-09-18T14:13:14.565157Z","shell.execute_reply":"2021-09-18T14:13:14.571937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"只提取前50个csv的文件内容保存进train中，不然内存储存空间不够用","metadata":{}},{"cell_type":"code","source":"number_categories = 50\nfiles_directory = os.listdir('/kaggle/input/quickdraw-doodle-recognition/train_simplified')[:number_categories]\ntrain = pd.DataFrame()\nfor file in files_directory:\n    train = train.append(pd.read_csv('/kaggle/input/quickdraw-doodle-recognition/train_simplified/'+file, index_col='key_id', usecols=[1,2,3,5]))","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:13:16.777023Z","iopub.execute_input":"2021-09-18T14:13:16.777439Z","iopub.status.idle":"2021-09-18T14:15:00.211173Z","shell.execute_reply.started":"2021-09-18T14:13:16.777409Z","shell.execute_reply":"2021-09-18T14:15:00.210364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"随机打乱train，random_state为随机打乱的种子","metadata":{}},{"cell_type":"code","source":"train = shuffle(train, random_state=123)\ntrain","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:15:09.887398Z","iopub.execute_input":"2021-09-18T14:15:09.887836Z","iopub.status.idle":"2021-09-18T14:15:15.163157Z","shell.execute_reply.started":"2021-09-18T14:15:09.887806Z","shell.execute_reply":"2021-09-18T14:15:15.162295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train number of rows: \", train.shape[0])\nprint(\"Train number of columns: \", train.shape[1])\nprint(\"Train set features: \", train.columns.values)\nprint(\"Train number of label categories: \", number_categories)\nprint(\"Train label categories: \", train['word'].unique())","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:16:29.546969Z","iopub.execute_input":"2021-09-18T14:16:29.54762Z","iopub.status.idle":"2021-09-18T14:16:30.298059Z","shell.execute_reply.started":"2021-09-18T14:16:29.547573Z","shell.execute_reply":"2021-09-18T14:16:30.297239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 利用SNS柱状图显示数据量前10名和后10名的类别","metadata":{}},{"cell_type":"code","source":"count_gp = train.groupby(['word']).size().reset_index(name='count').sort_values('count', ascending=False)\ntop_10 = count_gp[:10]\nbottom_10 = count_gp[-10:]","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:30:18.117233Z","iopub.execute_input":"2021-09-18T14:30:18.117524Z","iopub.status.idle":"2021-09-18T14:30:19.613379Z","shell.execute_reply.started":"2021-09-18T14:30:18.117495Z","shell.execute_reply":"2021-09-18T14:30:19.610983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax_t10 = sns.barplot(x=\"word\", y=\"count\", data=top_10, palette=\"coolwarm\", ci=500)\nax_t10.set_xticklabels(ax_t10.get_xticklabels(),rotation=40, ha=\"right\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:41:08.55953Z","iopub.execute_input":"2021-09-18T14:41:08.559859Z","iopub.status.idle":"2021-09-18T14:41:08.879505Z","shell.execute_reply.started":"2021-09-18T14:41:08.559828Z","shell.execute_reply":"2021-09-18T14:41:08.878513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax_b10 = sns.barplot(x=\"word\", y=\"count\", data=bottom_10, palette=\"BrBG\")\nax_b10.set_xticklabels(ax_b10.get_xticklabels(), rotation=40, ha='right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:43:39.777366Z","iopub.execute_input":"2021-09-18T14:43:39.777674Z","iopub.status.idle":"2021-09-18T14:43:40.092985Z","shell.execute_reply.started":"2021-09-18T14:43:39.777643Z","shell.execute_reply":"2021-09-18T14:43:40.09202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"recognized\", data=train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:43:41.978984Z","iopub.execute_input":"2021-09-18T14:43:41.979263Z","iopub.status.idle":"2021-09-18T14:43:45.050898Z","shell.execute_reply.started":"2021-09-18T14:43:41.979228Z","shell.execute_reply":"2021-09-18T14:43:45.050066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['recognized'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:50:35.131706Z","iopub.execute_input":"2021-09-18T14:50:35.132736Z","iopub.status.idle":"2021-09-18T14:50:35.173736Z","shell.execute_reply.started":"2021-09-18T14:50:35.13268Z","shell.execute_reply":"2021-09-18T14:50:35.172608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 查看每个类别中认知正确和错误的数量","metadata":{}},{"cell_type":"code","source":"rec_gp = train.groupby([\"word\",\"recognized\"]).size().reset_index(name=\"count\")\nrec_true = rec_gp[rec_gp['recognized']==True].rename(index=str, columns={\"recognized\":\"recognized_true\", \"count\":\"count_true\"})\nrec_false = rec_gp[rec_gp['recognized']==False].rename(index=str, columns={\"recognized\":\"recognized_false\", \"count\":\"count_false\"})\nrec_gp = rec_true.set_index('word').join(rec_false.set_index('word'), on='word')\nrec_gp","metadata":{"execution":{"iopub.status.busy":"2021-09-18T14:57:27.748001Z","iopub.execute_input":"2021-09-18T14:57:27.748323Z","iopub.status.idle":"2021-09-18T14:57:29.0201Z","shell.execute_reply.started":"2021-09-18T14:57:27.748293Z","shell.execute_reply":"2021-09-18T14:57:29.01924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 图像化输入数据","metadata":{}},{"cell_type":"code","source":"train[:1]['drawing'].values[0]","metadata":{"execution":{"iopub.status.busy":"2021-09-18T15:20:07.440084Z","iopub.execute_input":"2021-09-18T15:20:07.441106Z","iopub.status.idle":"2021-09-18T15:20:07.447693Z","shell.execute_reply.started":"2021-09-18T15:20:07.441062Z","shell.execute_reply":"2021-09-18T15:20:07.446892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words = train['word'].tolist()\ndrawings = [ast.literal_eval(pts) for pts in train[:9]['drawing'].values]\n\nplt.figure(figsize=(10,10))\nfor i, drawing in enumerate(drawings):\n    plt.subplot(330 + (i+1))\n    for x, y in drawing:\n        plt.plot(x, y, marker=\".\")\n        plt.tight_layout()\n        plt.title(words[i])\n        plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2021-09-18T15:27:23.07716Z","iopub.execute_input":"2021-09-18T15:27:23.07808Z","iopub.status.idle":"2021-09-18T15:27:27.348176Z","shell.execute_reply.started":"2021-09-18T15:27:23.078022Z","shell.execute_reply":"2021-09-18T15:27:27.347099Z"},"trusted":true},"execution_count":null,"outputs":[]}]}