{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport PIL.Image as Image\nimport cv2\nfrom IPython.display import display\nfrom sklearn.model_selection import train_test_split\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib import rc\nimport pandas as pd\n!pip install iterative-stratification\nfrom iterstrat.ml_stratifiers import MultilabelStratifiedKFold\n%matplotlib inline\nnp.random.seed(42)\nimport shutil as sh\nimport os\n\nprint('Setup Completed')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Preprocessing(input_path, output_path):\n    #Example image\n    #Read image\n    img = cv2.imread(input_path)\n    gray_image = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    \n    #Histogram Equlization\n    # create a CLAHE object\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8,8))\n    cl1 = clahe.apply(gray_image)\n    img_f = cv2.cvtColor(cl1, cv2.COLOR_GRAY2BGR)\n    \n    #Normalization\n    norm_img = np.zeros((800,800))\n    n_img = cv2.normalize(img_f,  norm_img, 0, 255, cv2.NORM_MINMAX)\n    \n    final_image = Image.fromarray(n_img)\n    final_image.save(output_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Read Original Dataset\ntrain_df = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\nimage_ids = train_df.image_id.unique()\ndisplay(train_df .head())\nprint(train_df .shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Normal and Abnormal data\ntrain_normal = train_df[train_df['class_name']=='No finding'].reset_index(drop=True)\n\n\ntrain_abnormal = train_df[train_df['class_name']!='No finding'].reset_index(drop=True)\n\ndisplay(train_abnormal.tail())\n\nprint(f'Number of Normal data: {train_normal.shape[0]}')\nprint(f'Number of Abnormal: {train_abnormal.shape[0]}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Split function to fold\ndef split_df(df):\n    kf = MultilabelStratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    df['id'] = df.index\n    annot_pivot = pd.pivot_table(df, index=['image_id'], columns=['class_id'],\n                                 values='id', fill_value=0, aggfunc='count') \\\n    .reset_index().rename_axis(None, axis=1)\n    for fold, (train_idx, val_idx) in enumerate(kf.split(annot_pivot,\n                                                         annot_pivot.iloc[:, 1:(1+df['class_id'].nunique())])):\n        annot_pivot[f'fold_{fold}'] = 0\n        annot_pivot.loc[val_idx, f'fold_{fold}'] = 1\n    return annot_pivot\n\n\n\nfold_csv = split_df(train_df)\nfold_csv.head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(fold_csv[fold_csv.fold_0 == 1].shape)\nprint(fold_csv[fold_csv.fold_1 == 1].shape)\nprint(fold_csv[fold_csv.fold_2 == 1].shape)\nprint(fold_csv[fold_csv.fold_3 == 1].shape)\nprint(fold_csv[fold_csv.fold_4 == 1].shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create Folder in kaggle/working for 5 fold\n\nimport shutil as sh\nfrom tqdm.notebook import tqdm\n\nDir_origin_image = '/kaggle/input/vindatafold/VinData/images'\nDir_origin_labels = '/kaggle/input/vindatafold/VinData/labels'\n\n#fold 0-3 use to train and fold 4 use to test\n\n#Create file for fold\nfor i in range(0,5):\n    os.makedirs(f'/kaggle/working/custom_data/fold{i}/labels/train', exist_ok = True)\n    os.makedirs(f'/kaggle/working/custom_data/fold{i}/labels/test', exist_ok = True)\n    os.makedirs(f'/kaggle/working/custom_data/fold{i}/images/train', exist_ok = True)\n    os.makedirs(f'/kaggle/working/custom_data/fold{i}/images/test', exist_ok = True)\n    \n#Copy Image and txt from Data to fold \nfor fold in range(0,5):\n    print(f'\\n Fold {fold}:\\n')\n    list_image_train = fold_csv[fold_csv[f'fold_{fold}']==0]['image_id']\n    list_image_test = fold_csv[fold_csv[f'fold_{fold}']==1]['image_id']\n    Dir_Train_image = f'/kaggle/working/custom_data/fold{fold}/images/train'\n    Dir_Test_image = f'/kaggle/working/custom_data/fold{fold}/images/test'\n    Dir_Train_label = f'/kaggle/working/custom_data/fold{fold}/labels/train'\n    Dir_Test_label = f'/kaggle/working/custom_data/fold{fold}/labels/test'\n    print(Dir_Train_image)\n    print(Dir_Test_image)\n    print(Dir_Train_label)\n    print(Dir_Test_label)\n    \n    for image_id in tqdm(list_image_train,total=len(list_image_train)):\n        path_origin_images = os.path.join(Dir_origin_image, image_id +\".jpg\")\n        path_origin_label = os.path.join(Dir_origin_labels, image_id +\".txt\")\n        path_images = os.path.join(Dir_Train_image, image_id +\".jpg\")\n        path_label = os.path.join(Dir_Train_label, image_id +\".txt\")\n        Preprocessing(path_origin_images, path_images)\n        sh.copy(path_origin_label, path_label)\n    for image_id in tqdm(list_image_test,total=len(list_image_test)):\n        path_origin_images = os.path.join(Dir_origin_image, image_id +\".jpg\")\n        path_origin_label = os.path.join(Dir_origin_labels, image_id +\".txt\")\n        path_images = os.path.join(Dir_Test_image, image_id +\".jpg\")\n        path_label = os.path.join(Dir_Test_label, image_id +\".txt\")\n        Preprocessing(path_origin_images, path_images)\n        sh.copy(path_origin_label, path_label)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_ids, class_names = list(zip(*set(zip(train_df.class_id, train_df.class_name))))\nclasses = list(np.array(class_names)[np.argsort(class_ids)])\nclasses = list(map(lambda x: str(x), classes))\nclasses.pop()\nclasses","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create train.txt and test.txt for each Fold\n\nfor fold in range(0,5):\n    list_image_train = fold_csv[fold_csv[f'fold_{fold}']==0]['image_id']\n    list_image_test = fold_csv[fold_csv[f'fold_{fold}']==1]['image_id']\n    print(f'\\n Fold {fold}:\\n')\n    DIR = f'/kaggle/working/custom_data/fold{fold}'\n    DIR_images = f'/kaggle/working/custom_data/fold{fold}/images'\n    DIR_labels = f'/kaggle/working/custom_data/fold{fold}/labels'\n    train = 'train'\n    test = 'test'\n    train_txt_path = os.path.join(DIR,'train.txt')\n    test_txt_path = os.path.join(DIR,'test.txt') \n    train_images_path = os.path.join(DIR_images, train)\n    test_images_path = os.path.join(DIR_images, test)\n    train_labels_path = os.path.join(DIR_labels, train)\n    test_labels_path = os.path.join(DIR_labels, test)\n    print(train_images_path)\n    print(test_images_path)\n    print(train_labels_path)\n    print(test_labels_path)\n    file_train = open(train_txt_path, \"w+\")\n    for image_id in tqdm(list_image_train,total=len(list_image_train)):\n        file_path = os.path.join(train_images_path, image_id +\".jpg\")\n        file_train.write(f'{file_path}\\n')\n    file_train.close()\n    file_test = open(test_txt_path, \"w+\")\n    for image_id in tqdm(list_image_test,total=len(list_image_test)):\n        file_path = os.path.join(test_images_path, image_id +\".jpg\")\n        file_test.write(f'{file_path}\\n')\n    file_test.close()\n    print(train_txt_path)\n    print(test_txt_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from os.path import isfile, join\nimport yaml\nDIR = '/kaggle/working/custom_data/'\nfor fold in range(0,5):\n    train_txt_path = f'/kaggle/working/custom_data/fold{fold}/train.txt'\n    test_txt_path = f'/kaggle/working/custom_data/fold{fold}/test.txt'\n    data = dict(\n        train =  train_txt_path ,\n        val   =  test_txt_path,\n        nc    = 14,\n        names = classes\n        )\n    with open(join( DIR , f'custom{fold}.yaml'), 'w') as outfile:\n        yaml.dump(data, outfile, default_flow_style=False)\n    print(f'\\n Fold {fold}:\\n')\n    f = open(join( DIR , f'custom{fold}.yaml'), 'r')\n    print('\\nyaml:')\n    print(f.read())\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#cloning yolov5 model\n!git clone https://github.com/ultralytics/yolov5\n\n#cloning NVIDIA/apex to speed up the process\n!git clone https://github.com/NVIDIA/apex.git","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom IPython.display import Image, clear_output  # to display images\nprint('Setup complete. Using torch %s %s' % (torch.__version__, torch.cuda.get_device_properties(0) if torch.cuda.is_available() else 'CPU'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mv yolov5/* ./","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -r requirements.txt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python detect.py --weights yolov5s.pt --img 640 --conf 0.25 --source /kaggle/working/data/images/zidane.jpg","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nImage(filename='/kaggle/working/runs/detect/exp/zidane.jpg', width=600)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fold 0\n!WANDB_MODE=\"dryrun\" python train.py --img 640 --batch 16 --epochs 20 --data /kaggle/working/custom_data/custom0.yaml --cfg /kaggle/input/vindatafold/yolov5m.yaml --weights yolov5m.pt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(30,15))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp2/results.png'));","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'runs/train/exp2/weights/best.pt')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fold 1\n!WANDB_MODE=\"dryrun\" python train.py --img 640 --batch 16 --epochs 20 --data /kaggle/working/custom_data/custom1.yaml --cfg /kaggle/input/vindatafold/yolov5m.yaml --weights runs/train/exp2/weights/best.pt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'runs/train/exp3/weights/best.pt')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fold 2\n!WANDB_MODE=\"dryrun\" python train.py --img 640 --batch 16 --epochs 30 --data /kaggle/working/custom_data/custom2.yaml --cfg /kaggle/input/vindatafold/yolov5m.yaml --weights /kaggle/input/vindatafold/best_fold1.pt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(30,15))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp/results.png'));","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fold 3\n!WANDB_MODE=\"dryrun\" python train.py --img 640 --batch 16 --epochs 30 --data /kaggle/working/custom_data/custom3.yaml --cfg /kaggle/input/vindatafold/yolov5m.yaml --weights runs/train/exp/weights/last.pt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(30,15))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp2/results.png'));","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'runs/train/exp2/weights/last.pt')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fold 4\n!WANDB_MODE=\"dryrun\" python train.py --img 640 --batch 16 --epochs 40 --data /kaggle/working/custom_data/custom4.yaml --cfg /kaggle/input/vindatafold/yolov5m.yaml --weights /kaggle/input/vintraintest/0152.pt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(30,15))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp6/results.png'));","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}