{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport random\nimport tensorflow as tf\nimport tensorflow.keras.layers as L\nimport tensorflow.keras.backend as K\nimport joblib\nimport seaborn as sns\nimport pandas as pd\nimport cv2\nimport os\nimport matplotlib.pyplot as plt\nfrom utilities_x_ray import read_xray,showXray\nfrom tqdm import tqdm\nimport pydicom\nfrom sklearn.model_selection import KFold\nimport pydicom as dicom\nimport warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def seedAll(seed=355):\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    random.seed(seed)\nseedAll()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\nss = pd.read_csv('../input/vinbigdata-chest-xray-abnormalities-detection/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ss.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(8,10))\nplt.imshow(read_xray('../input/vinbigdata-chest-xray-abnormalities-detection/train/0108949daa13dc94634a7d650a05c0bb.dicom'),cmap=plt.cm.bone)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"showXray('../input/vinbigdata-chest-xray-abnormalities-detection/train/0108949daa13dc94634a7d650a05c0bb.dicom',train,with_boxes=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of rows in train dataframe: {}\".format(train.shape[0]))\nprint(\"Number of Unique images in train set: {}\".format(train.image_id.nunique()))\nprint(\"Number of Classes: {}\\n\".format(train.class_name.nunique()))\nprint(\"Class Names: {}\".format(list(train.class_name.unique())))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Null Values:\")\ntrain.isna().sum().to_frame().rename(columns={0:'Null Value count'}).style.background_gradient('viridis')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(9,6))\nsns.countplot(train[\"class_id\"]);\nplt.title(\"Class Distributions\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(9,6))\nsns.countplot(train[\"rad_id\"]);\nplt.title(\"rad_id Distributions\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_names = sorted(train.class_name.unique())\ndel class_names[class_names.index('No finding')]\nclass_names = class_names+['No finding']\nclasses = dict(zip(list(range(15)),class_names))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def prepareDataFrame(train_df= train):\n    train_df = train_df.fillna(0)\n    cols = ['image_id','label']+list(range(4*len(class_names[:-1])))\n    return_df = pd.DataFrame(columns=cols)\n    \n    for image in tqdm(train_df.image_id.unique()):\n        df = train_df.query(\"image_id==@image\")\n        label = np.zeros(15)\n        for cls in df.class_id.unique():\n            label[int(cls)]=1\n        bboxes_df = df.groupby('class_id')[['x_min','y_min','x_max','y_max']].mean().round()\n        \n        bboxes_list = [0 for i in range(60)]\n        for ind in list(bboxes_df.index):\n            bboxes_list[4*ind:4*ind+4] = list(bboxes_df.loc[ind,:].values)\n        return_df.loc[len(return_df),:] = [image]+[label]+bboxes_list[:-4]\n    return return_df\ntrain_df = prepareDataFrame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def generateFolds(n_splits = None):\n    kf = KFold(n_splits= n_splits)\n    for id,(tr_,val_) in enumerate(kf.split(train_df[\"image_id\"],train_df[\"label\"])):\n        train_df.loc[val_,'kfold'] = int(id)\n    train_df[\"kfold\"].astype(int)\n\ngenerateFolds(n_splits=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class DataLoader:\n    def __init__(self,path = None,train_df=train_df,val_df=None):\n        self.path = path\n        self.df = train_df\n        self.val_df = val_df\n        self.train_list = [f'{img}.npy' for img in train_df[\"image_id\"].unique()]\n        np.random.shuffle(self.train_list)\n        self.test_list = [f'{img}.npy' for img in val_df[\"image_id\"].unique()]\n        np.random.shuffle(self.test_list)\n    \n    def read_image(self):\n        for img in self.train_list:\n            im_name = img.split('.npy')[0]\n            image = np.load(self.path+img)\n            temp = self.df[self.df.image_id==im_name]\n            c_label,bb = temp.iloc[0,1],temp.iloc[0,2:].values.astype('float')\n            yield image,c_label,bb\n    \n    \n    def batch_generator(self,items,batch_size):\n        a=[]\n        i=0\n        for item in items:\n            a.append(item)\n            i+=1\n\n            if i%batch_size==0:\n                yield a\n                a=[]\n        if len(a) is not 0:\n            yield a\n            \n    def flow(self,batch_size):\n        \"\"\"\n        flow from given directory in batches\n        ==========================================\n        batch_size: size of the batch\n        \"\"\"\n        while True:\n            for bat in self.batch_generator(self.read_image(),batch_size):\n                batch_images = []\n                batch_c_labels = []\n                batch_bb = []\n                for im,im_c_label,im_bb in bat:\n                    batch_images.append(im)\n                    batch_c_labels.append(im_c_label)\n                    batch_bb.append(im_bb)\n                batch_images = np.stack(batch_images,axis=0)\n                batch_labels =  (np.stack(batch_c_labels,axis=0),np.stack(batch_bb,axis=0))\n                yield batch_images,batch_labels\n    \n    def getVal(self):\n        images = []\n        c_labels = []\n        bb_labels = []\n        for img in self.test_list:\n            im_name = img.split('.npy')[0]\n            image = np.load(self.path+img)\n            temp = self.val_df[self.val_df.image_id==im_name]\n            c_label,bb = temp.iloc[0,1],temp.iloc[0,2:].values.astype('float')\n            images.append(image)\n            c_labels.append(c_label)\n            bb_labels.append(bb)\n        return np.stack(images,axis=0),(np.stack(c_labels,axis=0),np.stack(bb_labels,axis=0))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def build():\n    in1 = L.Input(shape=(256,256,1))\n    \n    out1 = L.Conv2D(32,(3,3),activation=\"relu\")(in1)\n    out1 = L.Conv2D(32,(3,3),activation=\"relu\")(out1)\n    out1 = L.MaxPooling2D((2,2))(out1)\n    \n    out1 = L.Conv2D(64,(3,3),activation=\"relu\")(out1)\n    out1 = L.Conv2D(64,(3,3),activation=\"relu\")(out1)\n    out1 = L.MaxPooling2D((2,2))(out1)\n    \n    out1 = L.Conv2D(128,(3,3),activation=\"relu\")(out1)\n    out1 = L.Conv2D(128,(3,3),activation=\"relu\")(out1)\n    out1 = L.MaxPooling2D((2,2))(out1)\n    out1 = L.Flatten()(out1)\n    \n    out2 = L.Dense(50,activation=\"relu\",kernel_initializer=\"lecun_normal\")(out1)\n    out2 = L.Dense(30,activation=\"relu\",kernel_initializer=\"lecun_normal\")(out2)\n    out2 = L.Dense(15,activation=\"sigmoid\",kernel_initializer=\"lecun_normal\",name='class_out')(out2)\n    \n    out3 = L.Dense(50,activation=\"relu\",kernel_initializer=\"lecun_normal\")(out1)\n    out3 = L.Dense(30,activation=\"relu\",kernel_initializer=\"lecun_normal\")(out3)\n    out3 = L.Dense(56,activation=\"relu\",kernel_initializer=\"lecun_normal\",name=\"bb_out\")(out3)\n    \n    model = tf.keras.Model(inputs=in1,outputs=[out2,out3])\n    model.compile(loss={'class_out':'categorical_crossentropy','bb_out':'mse'},optimizer=\"adam\")\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = build()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tf.keras.utils.plot_model(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def getTest(path=None):\n    images = []\n    for img in tqdm(os.listdir(path)):\n        im_name = img.split('.npy')[0]\n        image = np.load(path+img)\n        images.append(image)\n    return np.stack(images,axis=0)\n\nX_test = getTest('../input/xraynumpy/images/test/')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_label = np.zeros((len(X_test),15))\nbb_label = np.zeros((len(X_test),56))\n\nfor fold in range(5):\n    print(f'\\nFold: {fold}\\n')\n    \n    X_train = train_df[train_df.kfold!=fold].drop('kfold',axis=1)\n    X_val = train_df[train_df.kfold==fold].drop('kfold',axis=1)\n    \n    dl = DataLoader('../input/xraynumpy/images/train/',X_train,X_val)\n    train_set = dl.flow(batch_size=32)\n    X_eval,Y_eval = dl.getVal()\n    \n    chckpt = tf.keras.callbacks.ModelCheckpoint(f'./model_f{fold}.hdf5',monitor='val_loss',mode='min',save_best_only=True)\n    \n    K.clear_session()\n    model = build()\n    \n    model.fit(train_set,\n             epochs=1,\n              steps_per_epoch=int(15000/32),\n              validation_data = (X_eval,Y_eval),\n              callbacks = [chckpt]\n             )\n    \n    c,b = model.predict(X_test)\n    class_label+=c\n    bb_label+=b\nclass_label = class_label/5\nbb_label = bb_label/5\nnp.save('./class_label.npy',class_label)\nnp.save('./bb_label.npy',bb_label)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Y_test = model.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"joblib.dump(Y_test, 'y_test')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cls, b = Y_test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_labels = []\n\nfor lab in cls:\n    lab = np.argmax(lab)\n    pred_labels.append(lab)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}