# -*- coding: utf-8 -*-
"""quickDrawCNN.ipynb

Automatically generated by Colaboratory.

Original file is located at
    https://colab.research.google.com/drive/1Lkl_K4SKpEA6pwInliw1gpw2Es7ERdi5

QUICK DRAW CNN
Hyungguk Kim

----------------------------------------------------------------------------------------

Main Idea : Ensemble of several CNNs (ideally the more the better, but due to time constraint only have 5).

The architecture of each model is identical, but since we cannot use the whole data at once due to memory constraint, randomly sample 1,000 from each word every time we train a model.

----------------------------------------------------------------------------------------

동일한 구조의 CNN을 여러개 training 시켜서 그 중 가장 많은 '득표'를 한 단어 3가지로 예측하였습니다. 처음에는 최소 10개 이상의 모델을 훈련시켜 ensemble 을 하려고 하였으나 시간이 너무 없는 관계로 5개만 훈련하였습니다.
각 모델을 훈련할 시, 각 단어 마다 1000개의 샘플을 랜덤하게 골라 트레이닝 데이터로 사용하였습니다. 이 방법으로 한번에 전체의 데이터를 사용하지 못하지만
batch normalization 마냥 여러번 샘플링을 하여 메모리 제한을 극복하고 최대한 많은 데이터로 훈련을 시키려고 했습니다.

(요번주(~5/17)가 기말고사여서 시험끝나고 밤새서 하느라 시간적 제약이 조금 있었습니다...)

GPU로 더 빨리 트레이닝을 하기 위해 구글에서 제공하는 Google colab을 이용하였습니다.

# Mount data files from my google drive.
"""

from google.colab import drive
drive.mount('/content/gdrive')


"""**Load training data from my google drive. Convert the csv files to numpy arrays.**

Randomly sample 1,000 data for each word. 

numpy array로 변환하는 과정은 
https://www.kaggle.com/jpmiller/image-based-cnn
을 참고였습니다.

** Some basic parameters of the input data.**
"""

ims_per_class = 1000
num_classes = 340
height, width = 32, 32

import numpy as np
import pandas as pd
import tensorflow as tf
from tensorflow import keras
import os
import glob
import matplotlib.pyplot as plt
import cv2
from PIL import Image, ImageDraw 
import re
import ast
from tqdm import tqdm
from dask import bag

def to_drawing(strokes):
    img = Image.new("P", (256,256), color=255)
    img_draw = ImageDraw.Draw(img)
    for stroke in ast.literal_eval(strokes):
        for i in range(len(stroke[0])-1):
            img_draw.line([stroke[0][i], 
                           stroke[1][i],
                           stroke[0][i+1], 
                           stroke[1][i+1]],
                           fill=0, width=5)
    img = img.resize((height, width))
    return np.array(img)/255
  
train_grand = []
class_paths = glob.glob('./train_simplified/*.csv')
for i,c in enumerate(tqdm(class_paths[0: 340])):
    n = sum(1 for line in open(c))
    indices = range(1, n)
    s = ims_per_class*5//4            # number of samples we want from each word.
    skip = sorted(np.random.choice(indices, size=n-s-1, replace=False))
    train = pd.read_csv(c, usecols=['drawing', 'recognized'], skiprows=skip, nrows=ims_per_class*5//4)
    train = train[train.recognized == True].head(ims_per_class)
    imagebag = bag.from_sequence(train.drawing.values).map(to_drawing) 
    trainarray = np.array(imagebag.compute())  # PARALLELIZE
    trainarray = np.reshape(trainarray, (ims_per_class, -1))    
    labelarray = np.full((train.shape[0], 1), i)
    trainarray = np.concatenate((labelarray, trainarray), axis=1)
    train_grand.append(trainarray)
    
train_grand = np.array([train_grand.pop() for i in np.arange(num_classes)]) #less memory than np.concatenate
train_grand = train_grand.reshape((-1, (height*width+1)))

del trainarray
del train

valfrac = 0.1
cutpt = int(valfrac * train_grand.shape[0])

np.random.shuffle(train_grand)
y_train, X_train = train_grand[cutpt: , 0], train_grand[cutpt: , 1:]
y_val, X_val = train_grand[0:cutpt, 0], train_grand[0:cutpt, 1:] #validation set is recognized==True

del train_grand

y_train = keras.utils.to_categorical(y_train, num_classes)
X_train = X_train.reshape(X_train.shape[0], height, width, 1)
y_val = keras.utils.to_categorical(y_val, num_classes)  
X_val = X_val.reshape(X_val.shape[0], height, width, 1)

np.random.seed(95)
np.random.shuffle(X_train)
np.random.seed(95)
np.random.shuffle(y_train)
np.random.seed(95)
np.random.shuffle(X_val)
np.random.seed(95)
np.random.shuffle(y_val)

"""Design a Convolutional Neural Netword"""

from tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping

def top_3_accuracy(x,y): 
    t3 = top_k_categorical_accuracy(x,y, 3)
    return t3

reduceLROnPlat = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, 
                                   verbose=1, mode='auto', min_delta=0.005, cooldown=5, min_lr=0.0001)
earlystop = EarlyStopping(monitor='val_top_3_accuracy', mode='max', patience=5) 
callbacks = [reduceLROnPlat, earlystop]

import tensorflow as tf
from tensorflow import keras
from tensorflow.keras.models import Sequential
from tensorflow.keras.layers import Dense, Dropout, Flatten
from tensorflow.keras.layers import Conv2D, MaxPooling2D, BatchNormalization
from tensorflow.keras.metrics import top_k_categorical_accuracy
from tensorflow.keras.optimizers import Adam


model = Sequential()
model.add(Conv2D(32, kernel_size=(3, 3), padding='same', activation='relu', input_shape=(height, width, 1)))
model.add(Conv2D(64, kernel_size=(3, 3), padding='same', activation='relu', input_shape=(height, width, 1)))
model.add(Conv2D(128, kernel_size=(3, 3), padding='same', activation='relu', input_shape=(height, width, 1)))
model.add(Conv2D(256, kernel_size=(3, 3), padding='same', activation='relu', input_shape=(height, width, 1)))
model.add(MaxPooling2D(pool_size=(2, 2)))
model.add(Conv2D(512, kernel_size=(3, 3), padding='same', activation='relu', input_shape=(height, width, 1)))
model.add(MaxPooling2D(pool_size=(2, 2)))
model.add(Conv2D(1024, kernel_size=(3, 3), padding='same', activation='relu', input_shape=(height, width, 1)))
model.add(MaxPooling2D(pool_size=(2, 2)))
model.add(Flatten())
model.add(Dense(680, activation='relu'))
model.add(Dropout(0.5))
model.add(Dense(540, activation='relu'))
model.add(Dropout(0.5))
model.add(Dense(340, activation='softmax'))

model.compile(loss='categorical_crossentropy',
              optimizer=Adam(lr=1e-4),
              metrics=['accuracy', top_3_accuracy])

model.summary()

"""**Train the model with the sampled data.**"""

model.fit(X_train, y_train, batch_size=32, validation_data=(X_val, y_val), epochs=20, callbacks=callbacks)

"""**With the trained model, make predictions for the test data.**"""

wordlist = []
reader = pd.read_csv('./test_simplified.csv', index_col=['key_id'],
    chunksize=2048)
for chunk in tqdm(reader, total=55):
    imagebag = bag.from_sequence(chunk.drawing.values).map(to_drawing)
    test = np.array(imagebag.compute())
    test = np.reshape(test, (test.shape[0], height, width, 1))
    preds = model.predict(test, verbose=0)
    ttvs = np.argsort(-preds)[:, 0:3]
    wordlist.append(ttvs)
    
wordarray = np.concatenate(wordlist)

"""**Save the predictions to csv file.**"""

classfiles = os.listdir('./train_simplified/')
numstonames = {i: v[:-4].replace(" ", "_") for i, v in enumerate(classfiles)}

preds_df = pd.DataFrame({'first': wordvarray[:,0], 'second': wordvarray[:,1], 'third': wordvarray[:,2]})
preds_df = preds_df.replace(numstonames)
preds_df['words'] = preds_df['first'] + " " + preds_df['second'] + " " + preds_df['third']

sub = pd.read_csv('./sample_submission.csv', index_col=['key_id'])
sub['word'] = preds_df.words.values
sub.to_csv('model5.csv')
sub.head()

"""**Take majority votes from the predictions we have made.
Intended to have at least more than 10 predictions from different models,
but due to time constraint only had 5.**
"""

from collections import Counter

for file in os.listdir('./predictions'):
      print(file)

model_1 = pd.read_csv('./predictions/model1.csv')
model_2 = pd.read_csv('./predictions/model2.csv')
model_3 = pd.read_csv('./predictions/model3.csv')
model_4 = pd.read_csv('./predictions/model4.csv')
model_5 = pd.read_csv('./predictions/model5.csv')

d= {'key_id': model_2['key_id'], 
    'mod1': model_1['word'], 
    'mod2':model_2['word'],
    'mod3':model_3['word'],
    'mod4':model_4['word'],
    'mod5':model_5['word']}

ensemble = pd.DataFrame(d)

col_words = []
for i, row in enumerate(ensemble.to_numpy()):
  r = row[1:]
  lst = []
  for w in r:
    lst = lst + w.split()
  t3 = Counter(lst).most_common(3)
  words = [x[0] for x in t3]
  words = words[0] + " " + words[1] + " " + words[2]
  col_words.append(words)
  
ensemble['words'] = col_words

ensemble.head()
ensemble.to_csv('./predictions/ensemble.csv', columns=['key_id', 'words'], header=['key_id', 'word'], index=False, quoting=csv.QUOTE_NONNUMERIC)rrent directory are saved as output.