{"cells":[{"metadata":{},"cell_type":"markdown","source":"1. 프레임워크\n    : Keras를 사용하였습니다. Tensorflow보다는 개발을 좀 더 편리하게 하기 위해서 Keras를 사용하였습니다.\n2. 데이터 전처리\n    : 이미지 파일로 제공되는 줄 알았으나, 그림을 그린 벡터형식으로 주어졌습니다. 이 벡터 형식을 이미지 파일로 변환하는 전처리 과정을 수행하였습니다. 벡터 형식은 CV2.line를 사용하여 그림을 그렸습니다.\n    또한 데이터가 256x256 크기였으나, 각 Label 당 100개정도 밖에 학습을 못할 정도로 주어진 자원(RAM)이 부족하여, 크기를 64x64로 축소하였습니다.\n3. 학습 모델\n    : CNN의 기본 모델을 사용하여 학습을 하였습니다. Convolution 레이어는 2개를 추가하였습니다. BatchSize = 400, epoch = 20 을 설정하고 학습하였습니다.\n4. 결과\n    : loss = 2.57703574775247 , acc = 0.43591177463531494 의 결과가 나왔습니다. 학습을 하는 데에 사용되는 데이터의 양이 상대적으로 부족하여 학습이 덜 된것 같습니다. RAM을 좀 더 효율적으로 사용하는 방법을 추가적으로 생각해 보아야 할 것 같습니다.\n    "},{"metadata":{"_uuid":"25b8c16e-3699-4c3a-b006-e24389450f47","_cell_guid":"74ad7158-0520-448d-b7cc-47574da0342c","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2 as cv # image lining\nimport matplotlib.pyplot as plt # image\nimport os\nimport ast # string to list\nfrom sklearn import model_selection # Train : Test set\nfrom sklearn.preprocessing import OneHotEncoder # OneHotEncoding\n\nfrom keras.models import Sequential # Keras\nfrom keras.layers import Dense, Dropout, Activation, Flatten, Conv2D, pooling\nfrom keras.utils import np_utils\n\nplt.gray() #graycolor setting","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"데이터 확인"},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(\"../input/quickdraw-doodle-recognition/train_simplified/cat.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"vector drawing"},{"metadata":{"trusted":true},"cell_type":"code","source":"def vector_to_img(vector):\n    image = np.zeros((256,256), np.uint8)\n    for line in vector:\n        for i in range(len(line[0])-1):\n            cv.line(image,(line[0][i],line[1][i]),(line[0][i+1],line[1][i+1]),color=255)\n    \n    temp_image= cv.resize(image, dsize=(64, 64), interpolation=cv.INTER_AREA)\n    del image\n    return temp_image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['drawing'][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"vector_img = ast.literal_eval(df['drawing'][0])\na=vector_to_img(vector_img)\nplt.imshow(a)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"전처리\n\nX - img\n\nY - label"},{"metadata":{"trusted":true},"cell_type":"code","source":"X = []\nY = []\nLabel = []\nLabel_num = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def File_Preprocessing(filename,X,Y,Label_num):\n    df = pd.read_csv(\"../input/quickdraw-doodle-recognition/train_simplified/{}\".format(filename))\n    for i in range(500):\n        img = vector_to_img(ast.literal_eval(df['drawing'][i])).reshape(64,64,1)\n        X.append(img)\n        Y.append([Label_num])\n        del img\n    del df\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"train_simplified에 대해 전처리 실행"},{"metadata":{"trusted":true},"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input/quickdraw-doodle-recognition/train_simplified'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        df = pd.read_csv(\"../input/quickdraw-doodle-recognition/train_simplified/{}\".format(filename))\n        Label.append(df['word'][0])\n        for i in range(500):\n            img = vector_to_img(ast.literal_eval(df['drawing'][i])).reshape(64,64,1)\n            X.append(img)\n            Y.append([Label_num])\n        Label_num = Label_num + 1\n        \nenc = OneHotEncoder()\nenc.fit(Y)\nY = enc.transform(Y).toarray()\n\nprint(\"Fin\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, X_test, Y_train, Y_test = model_selection.train_test_split(np.array(X), np.array(Y), test_size = 0.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(X_train.shape,X_test.shape,Y_train.shape,Y_test.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"CNN 학습"},{"metadata":{"trusted":true},"cell_type":"code","source":"#normalization\nX_train = X_train.astype('float32')\nX_test = X_test.astype('float32')\nX_train /= 255\nX_test /= 255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(32, 3, 3, activation='relu', input_shape=(64,64, 1)))\nprint(model.output_shape)\n\nmodel.add(Conv2D(32, 3, 3, activation='relu'))\nmodel.add(pooling.MaxPooling2D(pool_size=(2,2)))\nmodel.add(Dropout(0.25))\nprint(model.output_shape)\n\nmodel.add(Conv2D(32, 3, 3, activation='relu'))\nmodel.add(pooling.MaxPooling2D(pool_size=(2,2)))\nmodel.add(Dropout(0.25))\nprint(model.output_shape)\n\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(340, activation='softmax'))\nprint(model.output_shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(X_train, Y_train, batch_size=400, epochs=20, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"loss,acc = model.evaluate(X_test, Y_test, verbose=0)\nprint(\"loss = {} , acc = {}\".format(loss,acc))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('kmj.hdf5')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"test_simplified 에 대해 전처리 실행"},{"metadata":{"trusted":true},"cell_type":"code","source":"test_simplified = pd.read_csv(\"../input/quickdraw-doodle-recognition/test_simplified.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_simplified","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_simplified.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = []","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in range(112199):\n    img = vector_to_img(ast.literal_eval(df['drawing'][i])).reshape(64,64,1)\n    X.append(img)\nprint(\"Fin\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X=np.array(X)\nX.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(X[0].reshape(64,64))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"test_simplified Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"pred = model.predict(X, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"top_3 = np.argsort(-pred)[:, 0:3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"top_3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"top_3_Label = []\nfor i in top_3:\n    top_3_Label.append(\"{} {} {}\".format(Label[i[0]],Label[i[1]],Label[i[2]]))\nprint(\"Fin\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_submission = pd.read_csv(\"../input/quickdraw-doodle-recognition/sample_submission.csv\",index_col=['key_id'])\nsample_submission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = sample_submission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission['word'] = top_3_Label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv')\nsubmission.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}