{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Version\n* `v13`: Fold4\n* `v12`: Fold3\n* `v10`: Fold2\n* `v09`: Fold1\n* `v03`: Fold0"},{"metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.execute_input":"2021-01-01T09:44:43.843489Z","iopub.status.busy":"2021-01-01T09:44:43.842712Z","iopub.status.idle":"2021-01-01T09:44:53.448523Z","shell.execute_reply":"2021-01-01T09:44:53.447971Z"},"papermill":{"duration":9.633907,"end_time":"2021-01-01T09:44:53.448657","exception":false,"start_time":"2021-01-01T09:44:43.81475","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"!pip install --upgrade seaborn\n!pip install ensemble_boxes","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:44:53.50848Z","iopub.status.busy":"2021-01-01T09:44:53.50769Z","iopub.status.idle":"2021-01-01T09:44:54.403472Z","shell.execute_reply":"2021-01-01T09:44:54.402433Z"},"papermill":{"duration":0.926929,"end_time":"2021-01-01T09:44:54.403588","exception":false,"start_time":"2021-01-01T09:44:53.476659","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.notebook import tqdm\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.options.display.max_columns = None\npd.options.display.max_rows = None\npd.set_option('max_colwidth',1600)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dim = 512 #512, 256, 'original'\nfold = 4","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:44:54.468231Z","iopub.status.busy":"2021-01-01T09:44:54.467706Z","iopub.status.idle":"2021-01-01T09:44:54.690894Z","shell.execute_reply":"2021-01-01T09:44:54.69183Z"},"papermill":{"duration":0.262045,"end_time":"2021-01-01T09:44:54.691965","exception":false,"start_time":"2021-01-01T09:44:54.42992","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(f'../input/vinbigdata-{dim}-image-dataset/vinbigdata/train.csv')\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:44:54.7506Z","iopub.status.busy":"2021-01-01T09:44:54.749926Z","iopub.status.idle":"2021-01-01T09:44:54.80575Z","shell.execute_reply":"2021-01-01T09:44:54.804838Z"},"papermill":{"duration":0.086788,"end_time":"2021-01-01T09:44:54.805857","exception":false,"start_time":"2021-01-01T09:44:54.719069","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df['image_path'] = f'/kaggle/input/vinbigdata-{dim}-image-dataset/vinbigdata/train/'+train_df.image_id+('.png' if dim!='original' else '.jpg')\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.027478,"end_time":"2021-01-01T09:44:54.861374","exception":false,"start_time":"2021-01-01T09:44:54.833896","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Only 14 Class"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:44:54.922292Z","iopub.status.busy":"2021-01-01T09:44:54.921364Z","iopub.status.idle":"2021-01-01T09:44:54.943995Z","shell.execute_reply":"2021-01-01T09:44:54.943575Z"},"papermill":{"duration":0.05543,"end_time":"2021-01-01T09:44:54.944088","exception":false,"start_time":"2021-01-01T09:44:54.888658","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_df = train_df[train_df.class_id!=14].reset_index(drop = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train_df[(train_df.image_id=='9a5094b2563a1ef3ff50dc5c7ff71345')]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# IoU = Calculate_IoU( (692.0,1375.0,1657.0,1799.0), (689.0,1313.0,1666.0,1763.0))\n# print(\"IoU是：{}\".format(IoU))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train_df[(train_df.image_id=='051132a778e61a86eb147c7c6f564dfe')]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_name_ids = ['Aortic enlargement','Atelectasis','Calcification','Cardiomegaly','Consolidation','ILD','Infiltration','Lung Opacity','Nodule/Mass','Other lesion','Pleural effusion','Pleural thickening','Pneumothorax','Pulmonary fibrosis']\nvalues = [0,1,2,3,4,5,6,7,8,9,10,11,12,13]\nclass_dictionary = dict(zip(class_name_ids, values))\nprint (class_dictionary)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['class_id'].value_counts(normalize = False, dropna = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# class_name_ids = ['Aortic Enlargement','Atelectasis','Calcification','Cardiomegaly','Consolidation','ILD','Infiltration','Lung Opacity','Nodule/Mass','Other Lesion','Pleural Effusion','Pleural Thickening','Pneumothorax','Pulmonary Fibrosis']\n# values = [0,1,2,3,4,5,6,7,8,9,10,11,12,13]\n# class_dictionary = dict(zip(class_name_ids, values))\n# print (class_dictionary)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # import numpy as np\n# import pandas as pd\n\n# from tqdm import tqdm\n# from ensemble_boxes import *\n\n# consist_iou_threshold=0.3\n# # ===============================\n# # Default WBF config (you can change these)\n# iou_thr = 0.5\n# skip_box_thr = 0.0001\n# sigma = 0.1\n# # ===============================\n\n# # Loading the train DF\n# # df = pd.read_csv(\"../input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")\n# df = train_df\n# df.fillna(0, inplace=True)\n# # df.loc[df[\"class_id\"] == 14, ['x_max', 'y_max']] = 1.0\n\n# results = []\n# image_ids = df[\"image_id\"].unique()\n\n\n# count=3#\n\n\n# for image_id in tqdm(image_ids, total=len(image_ids)):\n#     count-=1\n#     print('count',count)\n#     if count==0:#\n#         break#\n    \n#     print('image_id',image_id)\n        \n#     # All annotations for the current image.\n#     data = df[df[\"image_id\"] == image_id]\n#     data = data.reset_index(drop=True)\n#     annotations = {}\n#     weights = []\n    \n    \n#     width=data.iloc[0].width\n#     height=data.iloc[0].height\n#     image_path=data.iloc[0].image_path\n#     class_name=data.iloc[0].class_name\n\n#     # WBF expects the coordinates in 0-1 range.\n#     max_value = data.iloc[:, 4:8].values.max()\n#     data.loc[:, [\"x_min\", \"y_min\", \"x_max\", \"y_max\"]] = data.iloc[:, 4:8]\n\n#     # Loop through all of the annotations\n#     for idx, row in data.iterrows():\n\n#         class_name_id = row[\"class_name\"]\n\n#         if class_name_id not in annotations:\n#             annotations[class_name_id] = {\n#                 \"boxes_list\": [],\n#                 \"rad_id_list\": [],\n#                 \"labels_list\": [],\n#             }\n\n#             # We consider all of the radiologists as equal.\n#             weights.append(1.0)\n\n#         annotations[class_name_id][\"boxes_list\"].append([row[\"x_min\"], row[\"y_min\"], row[\"x_max\"], row[\"y_max\"]])\n#         annotations[class_name_id][\"rad_id_list\"].append(row[\"rad_id\"])\n#         annotations[class_name_id][\"labels_list\"].append(class_name_id)\n#         print('class_name_id',class_name_id)\n\n#     boxes_list = []\n#     rad_list = []\n#     labels_list = []\n    \n#     print('annotations',annotations)\n\n#     for annotator in annotations.keys():\n#         boxes_list.append(annotations[annotator][\"boxes_list\"])\n#         rad_list.append(annotations[annotator][\"rad_id_list\"])\n#         labels_list.append(annotations[annotator][\"labels_list\"])\n        \n#     print()\n#     print('boxes_list',boxes_list)\n#     print()\n#     print('rad_list',rad_list)\n#     print()\n#     print('labels_list',labels_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# import pandas as pd\n# df = pd.DataFrame({'BoolCol': [1, 2, 3, 3, 4],'attr': [22, 33, 22, 44, 66]},  \n#        index=[10,20,30,40,50])  \n# print(df)  \n# a = df[(df.BoolCol==3)&(df.attr==22)].index.tolist()  \n# print(a)  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Calculate_IoU(predicted_bound, ground_truth_bound):\n    pxmin, pymin, pxmax, pymax = predicted_bound\n    print(\"预测框P的坐标是：({}, {}, {}, {})\".format(pxmin, pymin, pxmax, pymax))\n    gxmin, gymin, gxmax, gymax = ground_truth_bound\n    print(\"原标记框G的坐标是：({}, {}, {}, {})\".format(gxmin, gymin, gxmax, gymax))\n    \n#  \"\"\"\n#  computing the IoU of two boxes.\n#  Args:\n#   box: (xmin, ymin, xmax, ymax),通过左下和右上两个顶点坐标来确定矩形位置\n#  Return:\n#   IoU: IoU of box1 and box2.\n#  \"\"\"\n    \n\n    parea = (pxmax - pxmin) * (pymax - pymin) # 计算P的面积\n    garea = (gxmax - gxmin) * (gymax - gymin) # 计算G的面积\n    print(\"预测框P的面积是：{}；原标记框G的面积是：{}\".format(parea, garea))\n    print('mark1')\n\n    # 求相交矩形的左下和右上顶点坐标(xmin, ymin, xmax, ymax)\n    xmin = max(pxmin, gxmin) # 得到左下顶点的横坐标\n    ymin = max(pymin, gymin) # 得到左下顶点的纵坐标\n    xmax = min(pxmax, gxmax) # 得到右上顶点的横坐标\n    ymax = min(pymax, gymax) # 得到右上顶点的纵坐标\n    print('mark2')\n    # 计算相交矩形的面积\n    w = xmax - xmin\n    h = ymax - ymin\n    if w <=0 or h <= 0:\n        return 0,(0,0,0,0)\n    print('mark3')\n    area = w * h # G∩P的面积\n    # area = max(0, xmax - xmin) * max(0, ymax - ymin) # 可以用一行代码算出来相交矩形的面积\n    print(\"G∩P的坐标是：\",xmin,ymin,xmax,ymax)\n    print(\"G∩P的面积是：{}\".format(area))\n\n    # 并集的面积 = 两个矩形面积 - 交集面积\n    IoU = area / (parea + garea - area)\n\n    return IoU,(xmin,ymin,xmax,ymax)\n \nif __name__ == '__main__':\n    IoU = Calculate_IoU( (-1, -1, 1, 1), (0, 0, 2, 2))\n    print(\"IoU是：{}\".format(IoU))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# import numpy as np\nimport pandas as pd\n\nfrom tqdm import tqdm\nfrom ensemble_boxes import *\n\nconsist_iou_threshold=0.5\n# ===============================\n# Default WBF config (you can change these)\niou_thr = 0.5\nskip_box_thr = 0.0001\nsigma = 0.1\n# ===============================\n\n# Loading the train DF\n# df = pd.read_csv(\"../input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")\ndf = train_df\ndf.fillna(0, inplace=True)\n# df.loc[df[\"class_id\"] == 14, ['x_max', 'y_max']] = 1.0\n\nresults = []\nimage_ids = df[\"image_id\"].unique()\n\n\n# count=3000#\n\n\nfor image_id in tqdm(image_ids, total=len(image_ids)):\n#     count-=1\n#     print('count',count)\n#     if count==0:#\n#         break#\n    \n    print('image_id',image_id)\n        \n    # All annotations for the current image.\n    data = df[df[\"image_id\"] == image_id]\n    data = data.reset_index(drop=True)\n    annotations = {}\n    weights = []\n    \n    \n    width=data.iloc[0].width\n    height=data.iloc[0].height\n    image_path=data.iloc[0].image_path\n    class_name=data.iloc[0].class_name\n\n    # WBF expects the coordinates in 0-1 range.\n    max_value = data.iloc[:, 4:8].values.max()\n    data.loc[:, [\"x_min\", \"y_min\", \"x_max\", \"y_max\"]] = data.iloc[:, 4:8]\n\n    # Loop through all of the annotations\n    for idx, row in data.iterrows():\n#         print('index',row.index.tolist()[0]) \n        class_name_id = row[\"class_name\"]\n#         print('index',row.index)\n        if class_name_id not in annotations:\n            annotations[class_name_id] = {\n                \"boxes_list\": [],\n                \"rad_id_list\": [],\n                \"labels_list\": [],\n            }\n\n            # We consider all of the radiologists as equal.\n            weights.append(1.0)\n\n        annotations[class_name_id][\"boxes_list\"].append([row[\"x_min\"], row[\"y_min\"], row[\"x_max\"], row[\"y_max\"]])\n        annotations[class_name_id][\"rad_id_list\"].append(row[\"rad_id\"])\n        annotations[class_name_id][\"labels_list\"].append(class_name_id)\n        print('class_name_id',class_name_id)\n\n    boxes_list = []\n    rad_list = []\n    labels_list = []\n    \n    print('annotations',annotations)\n\n    for annotator in annotations.keys():\n        boxes_list.append(annotations[annotator][\"boxes_list\"])\n        rad_list.append(annotations[annotator][\"rad_id_list\"])\n        labels_list.append(annotations[annotator][\"labels_list\"])\n        \n    print()\n    print('boxes_list',boxes_list)\n    print()\n    print('rad_list',rad_list)\n    print()\n    print('labels_list',labels_list)\n    print()\n    print()\n    \n#     for i in range(len(boxes_list)):\n        \n\n#         for j in range(0,len(boxes_list[i])-1):\n#             iou_value_list=[]\n# #             G_P_box_list=[]\n# #             box_pass_sign=False\n#             for k in range(1+j,len(boxes_list[i])):\n#                 gt_ground=(boxes_list[i][j][0], boxes_list[i][j][1], boxes_list[i][j][2], boxes_list[i][j][3])\n#                 pr_ground=(boxes_list[i][k][0], boxes_list[i][k][1], boxes_list[i][k][2], boxes_list[i][k][3])\n#                 print('gt_ground',gt_ground)\n#                 print('pr_ground',pr_ground)\n#                 iou_value,G_P_box=Calculate_IoU(gt_ground,pr_ground)\n#                 iou_value_list.append(iou_value)\n#                 # 如果两个框iou大于阈值consist_iou_threshold\n#                 if iou_value>consist_iou_threshold:\n#                     results.append({\n#                     \"image_id\": image_id,\n#                     \"class_id\": int(class_dictionary[labels_list[i][0]]),\n#                     \"rad_id\": \"wbf\",\n#                     \"x_min\": G_P_box[0],\n#                     \"y_min\": G_P_box[1],\n#                     \"x_max\": G_P_box[2],\n#                     \"y_max\": G_P_box[3],\n#                     \"width\": width,\n#                     \"height\": height,\n#                     \"image_path\": image_path,\n#                     \"class_name\":  labels_list[i][0]\n#                 })\n\n    #                     G_P_box_list.append(G_P_box)\n    #             for p in range(len(iou_value_list)):\n    #                 pass\n    #                 np.mean(iou_value_list)\n    #             if box_pass_sign:\n\n    for i in range(len(boxes_list)):\n        \n        if (labels_list[i][0]=='Aortic enlargement')|(labels_list[i][0]=='Cardiomegaly')|(labels_list[i][0]=='Pleural effusion')|(labels_list[i][0]=='Pleural thickening')|(labels_list[i][0]=='Pulmonary fibrosis'):\n            for j in range(0,len(boxes_list[i])-1):\n                iou_value_list=[]\n    #             G_P_box_list=[]\n    #             box_pass_sign=False\n                for k in range(1+j,len(boxes_list[i])):\n                    gt_ground=(boxes_list[i][j][0], boxes_list[i][j][1], boxes_list[i][j][2], boxes_list[i][j][3])\n                    pr_ground=(boxes_list[i][k][0], boxes_list[i][k][1], boxes_list[i][k][2], boxes_list[i][k][3])\n                    print('gt_ground',gt_ground)\n                    print('pr_ground',pr_ground)\n                    iou_value,G_P_box=Calculate_IoU(gt_ground,pr_ground)\n                    iou_value_list.append(iou_value)\n                    # 如果两个框iou大于阈值consist_iou_threshold\n                    if iou_value>consist_iou_threshold:\n                        results.append({\n                        \"image_id\": image_id,\n                        \"class_id\": int(class_dictionary[labels_list[i][0]]),\n                        \"rad_id\": \"wbf\",\n                        \"x_min\": G_P_box[0],\n                        \"y_min\": G_P_box[1],\n                        \"x_max\": G_P_box[2],\n                        \"y_max\": G_P_box[3],\n                        \"width\": width,\n                        \"height\": height,\n                        \"image_path\": image_path,\n                        \"class_name\":  labels_list[i][0]\n                    })\n\n    #                     G_P_box_list.append(G_P_box)\n    #             for p in range(len(iou_value_list)):\n    #                 pass\n    #                 np.mean(iou_value_list)\n    #             if box_pass_sign:\n        else:\n            for j in range(0,len(boxes_list[i])-1):\n                results.append({\n                    \"image_id\": image_id,\n                    \"class_id\": int(class_dictionary[labels_list[i][0]]),\n                    \"rad_id\": \"wbf\",\n                    \"x_min\": boxes_list[i][j][0],\n                    \"y_min\": boxes_list[i][j][1],\n                    \"x_max\": boxes_list[i][j][2],\n                    \"y_max\": boxes_list[i][j][3],\n                    \"width\": width,\n                    \"height\": height,\n                    \"image_path\": image_path,\n                    \"class_name\":  labels_list[i][0]\n                })\n\n                \n#             'Aortic enlargement''Cardiomegaly''Pleural effusion''Pleural thickening''Pulmonary fibrosis'\n        \n                \n                \n\nresults = pd.DataFrame(results)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results = pd.DataFrame(results)\nresults['class_id'].value_counts(normalize = False, dropna = False)\ntrain_df=results\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['class_id'].value_counts(normalize = False, dropna = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# import numpy as np\n# a = [2,4,6,8,10]\n# average_a = np.mean(a)\n# average_a","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.052492,"end_time":"2021-01-01T09:47:56.110766","exception":false,"start_time":"2021-01-01T09:47:56.058274","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Split"},{"metadata":{"trusted":true},"cell_type":"code","source":"# !pip install ensemble_boxes","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:47:56.234311Z","iopub.status.busy":"2021-01-01T09:47:56.233425Z","iopub.status.idle":"2021-01-01T09:47:56.297206Z","shell.execute_reply":"2021-01-01T09:47:56.297652Z"},"papermill":{"duration":0.134603,"end_time":"2021-01-01T09:47:56.297774","exception":false,"start_time":"2021-01-01T09:47:56.163171","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"gkf  = GroupKFold(n_splits = 5)\ntrain_df['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(train_df, groups = train_df.image_id.tolist())):\n    train_df.loc[val_idx, 'fold'] = fold\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:47:56.420454Z","iopub.status.busy":"2021-01-01T09:47:56.419355Z","iopub.status.idle":"2021-01-01T09:47:56.443157Z","shell.execute_reply":"2021-01-01T09:47:56.443661Z"},"papermill":{"duration":0.086817,"end_time":"2021-01-01T09:47:56.443789","exception":false,"start_time":"2021-01-01T09:47:56.356972","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"train_files = []\nval_files   = []\nval_files += list(train_df[train_df.fold==fold].image_path.unique())\ntrain_files += list(train_df[train_df.fold!=fold].image_path.unique())\nlen(train_files), len(val_files)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# os.makedirs('/kaggle/working/vinbigdata/labels/train', exist_ok = True)\n# os.makedirs('/kaggle/working/vinbigdata/labels/val', exist_ok = True)\n# os.makedirs('/kaggle/working/vinbigdata/images/train', exist_ok = True)\n# os.makedirs('/kaggle/working/vinbigdata/images/val', exist_ok = True)\n# label_dir = '/kaggle/input/vinbigdata-yolo-labels-dataset/labels'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os.path as osp\nfrom path import Path\nfrom collections import Counter\nimport cv2\nfrom ensemble_boxes import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imagepaths = train_df['image_path'].unique()\ntrain_annotations=train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# img_array  = cv2.imread('/kaggle/input/vinbigdata-512-image-dataset/vinbigdata/train/d3637a1935a905b3c326af31389cb846.png')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Create_nms_box_txt(desktop_path,name):\n    iou_thr = 0.5\n    skip_box_thr = 0.0001\n    viz_images = []\n#     image_basename = Path(path).stem\n    image_basename = name\n#     print(f\"(\\'{image_basename}\\', \\'{path}\\')\")\n    img_annotations = train_annotations[train_annotations.image_id==image_basename]\n\n    boxes_viz = img_annotations[['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().tolist()\n    labels_viz = img_annotations['class_id'].to_numpy().tolist()\n\n    print(\"Bboxes before nms:\\n\", boxes_viz)\n    print(\"Labels before nms:\\n\", labels_viz)\n\n    boxes_list = []\n    scores_list = []\n    labels_list = []\n    weights = []\n\n    boxes_single = []\n    labels_single = []\n\n    cls_ids = img_annotations['class_id'].unique().tolist()\n    count_dict = Counter(img_annotations['class_id'].tolist())\n    print(count_dict)\n\n    for cid in cls_ids:       \n        ## Performing Fusing operation only for multiple bboxes with the same label\n        if count_dict[cid]==1:\n            labels_single.append(cid)\n            boxes_single.append(img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().squeeze().tolist())\n\n        else:\n            cls_list =img_annotations[img_annotations.class_id==cid]['class_id'].tolist()\n            labels_list.append(cls_list)\n            bbox = img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy()\n\n            ## Normalizing Bbox by Image Width and Height\n            bbox = bbox/(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n\n\n            bbox = np.clip(bbox, 0, 1)\n            boxes_list.append(bbox.tolist())\n\n            scores_list.append(np.ones(len(cls_list)).tolist())\n\n            weights.append(1)\n\n            \n    # Perform NMS\n    if len(boxes_list)==0:\n        boxes=boxes_single\n        box_labels=labels_single\n        print(\"Bboxes after nms:\\n\", boxes)\n        print(\"Labels after nms:\\n\", box_labels)\n        \n        count_dict = Counter(box_labels)\n        print(count_dict)\n\n        text_create(desktop_path,image_basename,boxes,box_labels,img_annotations.iloc[0][\"width\"],img_annotations.iloc[0][\"height\"])\n        \n    else:\n        boxes, scores, box_labels = nms(boxes_list, scores_list, labels_list, weights=weights,\n                                    iou_thr=iou_thr)\n\n\n        #img_array.shape[1]是宽度\n        boxes = boxes*(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n        boxes = boxes.round(1).tolist()\n        box_labels = box_labels.astype(int).tolist()\n\n        boxes.extend(boxes_single)\n        box_labels.extend(labels_single)\n\n        print(\"Bboxes after nms:\\n\", boxes)\n        print(\"Labels after nms:\\n\", box_labels)\n\n        count_dict = Counter(box_labels)\n        print(count_dict)\n\n        text_create(desktop_path,image_basename,boxes,box_labels,img_annotations.iloc[0][\"width\"],img_annotations.iloc[0][\"height\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# iou_thr = 0.5\n# skip_box_thr = 0.0001\n# viz_images = []\n# for i, path in tqdm(enumerate(imagepaths[5:6])):\n#     image_basename = Path(path).stem\n#     print(f\"(\\'{image_basename}\\', \\'{path}\\')\")\n#     img_annotations = train_annotations[train_annotations.image_id==image_basename]\n\n#     boxes_viz = img_annotations[['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().tolist()\n#     labels_viz = img_annotations['class_id'].to_numpy().tolist()\n\n#     print(\"Bboxes before nms:\\n\", boxes_viz)\n#     print(\"Labels before nms:\\n\", labels_viz)\n\n#     boxes_list = []\n#     scores_list = []\n#     labels_list = []\n#     weights = []\n\n#     boxes_single = []\n#     labels_single = []\n\n#     cls_ids = img_annotations['class_id'].unique().tolist()\n#     count_dict = Counter(img_annotations['class_id'].tolist())\n#     print(count_dict)\n\n#     for cid in cls_ids:       \n#         ## Performing Fusing operation only for multiple bboxes with the same label\n#         if count_dict[cid]==1:\n#             labels_single.append(cid)\n#             boxes_single.append(img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().squeeze().tolist())\n\n#         else:\n#             cls_list =img_annotations[img_annotations.class_id==cid]['class_id'].tolist()\n#             labels_list.append(cls_list)\n#             bbox = img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy()\n\n#             ## Normalizing Bbox by Image Width and Height\n#             bbox = bbox/(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n\n\n#             bbox = np.clip(bbox, 0, 1)\n#             boxes_list.append(bbox.tolist())\n\n#             scores_list.append(np.ones(len(cls_list)).tolist())\n\n#             weights.append(1)\n\n\n#     # Perform NMS\n#     boxes, scores, box_labels = nms(boxes_list, scores_list, labels_list, weights=weights,\n#                                     iou_thr=iou_thr)\n    \n#     print(\"Bboxes without multipy:\\n\", boxes)\n\n#     #img_array.shape[1]是宽度\n#     boxes = boxes*(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n#     boxes = boxes.round(1).tolist()\n#     box_labels = box_labels.astype(int).tolist()\n\n#     boxes.extend(boxes_single)\n#     box_labels.extend(labels_single)\n\n#     print(\"Bboxes after nms:\\n\", boxes)\n#     print(\"Labels after nms:\\n\", box_labels)\n\n#     count_dict = Counter(box_labels)\n#     print(count_dict)\n\n#     text_create('./',image_basename,boxes,box_labels,img_annotations.iloc[0][\"width\"],img_annotations.iloc[0][\"height\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Create_softnms_box_txt(desktop_path,name):\n    iou_thr = 0.5\n    skip_box_thr = 0.0001\n    viz_images = []\n    sigma = 0.1\n#     image_basename = Path(path).stem\n    image_basename = name\n#     print(f\"(\\'{image_basename}\\', \\'{path}\\')\")\n    img_annotations = train_annotations[train_annotations.image_id==image_basename]\n\n    boxes_viz = img_annotations[['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().tolist()\n    labels_viz = img_annotations['class_id'].to_numpy().tolist()\n\n    print(\"Bboxes before nms:\\n\", boxes_viz)\n    print(\"Labels before nms:\\n\", labels_viz)\n\n    boxes_list = []\n    scores_list = []\n    labels_list = []\n    weights = []\n\n    boxes_single = []\n    labels_single = []\n\n    cls_ids = img_annotations['class_id'].unique().tolist()\n    count_dict = Counter(img_annotations['class_id'].tolist())\n    print(count_dict)\n\n    for cid in cls_ids:       \n        ## Performing Fusing operation only for multiple bboxes with the same label\n        if count_dict[cid]==1:\n            labels_single.append(cid)\n            boxes_single.append(img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().squeeze().tolist())\n\n        else:\n            cls_list =img_annotations[img_annotations.class_id==cid]['class_id'].tolist()\n            labels_list.append(cls_list)\n            bbox = img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy()\n\n            ## Normalizing Bbox by Image Width and Height\n            bbox = bbox/(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n\n\n            bbox = np.clip(bbox, 0, 1)\n            boxes_list.append(bbox.tolist())\n\n            scores_list.append(np.ones(len(cls_list)).tolist())\n\n            weights.append(1)\n\n            \n    # Perform NMS\n    if len(boxes_list)==0:\n        boxes=boxes_single\n        box_labels=labels_single\n        print(\"Bboxes after nms:\\n\", boxes)\n        print(\"Labels after nms:\\n\", box_labels)\n        \n        count_dict = Counter(box_labels)\n        print(count_dict)\n\n        text_create(desktop_path,image_basename,boxes,box_labels,img_annotations.iloc[0][\"width\"],img_annotations.iloc[0][\"height\"])\n        \n    else:\n        boxes, scores, box_labels = soft_nms(boxes_list, scores_list, labels_list, weights=weights,\n                                    iou_thr=iou_thr)\n\n\n        #img_array.shape[1]是宽度\n        boxes = boxes*(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n        boxes = boxes.round(1).tolist()\n        box_labels = box_labels.astype(int).tolist()\n\n        boxes.extend(boxes_single)\n        box_labels.extend(labels_single)\n\n        print(\"Bboxes after nms:\\n\", boxes)\n        print(\"Labels after nms:\\n\", box_labels)\n\n        count_dict = Counter(box_labels)\n        print(count_dict)\n\n        text_create(desktop_path,image_basename,boxes,box_labels,img_annotations.iloc[0][\"width\"],img_annotations.iloc[0][\"height\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Create_non_maximum_weighted_box_txt(desktop_path,name):\n    iou_thr = 0.5\n    skip_box_thr = 0.0001\n    viz_images = []\n#     image_basename = Path(path).stem\n    image_basename = name\n#     print(f\"(\\'{image_basename}\\', \\'{path}\\')\")\n    img_annotations = train_annotations[train_annotations.image_id==image_basename]\n\n    boxes_viz = img_annotations[['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().tolist()\n    labels_viz = img_annotations['class_id'].to_numpy().tolist()\n\n    print(\"Bboxes before nms:\\n\", boxes_viz)\n    print(\"Labels before nms:\\n\", labels_viz)\n\n    boxes_list = []\n    scores_list = []\n    labels_list = []\n    weights = []\n\n    boxes_single = []\n    labels_single = []\n\n    cls_ids = img_annotations['class_id'].unique().tolist()\n    count_dict = Counter(img_annotations['class_id'].tolist())\n    print(count_dict)\n\n    for cid in cls_ids:       \n        ## Performing Fusing operation only for multiple bboxes with the same label\n        if count_dict[cid]==1:\n            labels_single.append(cid)\n            boxes_single.append(img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().squeeze().tolist())\n\n        else:\n            cls_list =img_annotations[img_annotations.class_id==cid]['class_id'].tolist()\n            labels_list.append(cls_list)\n            bbox = img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy()\n\n            ## Normalizing Bbox by Image Width and Height\n            bbox = bbox/(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n\n\n            bbox = np.clip(bbox, 0, 1)\n            boxes_list.append(bbox.tolist())\n\n            scores_list.append(np.ones(len(cls_list)).tolist())\n\n            weights.append(1)\n\n            \n    # Perform NMS\n    if len(boxes_list)==0:\n        boxes=boxes_single\n        box_labels=labels_single\n        print(\"Bboxes after nms:\\n\", boxes)\n        print(\"Labels after nms:\\n\", box_labels)\n        \n        count_dict = Counter(box_labels)\n        print(count_dict)\n\n        text_create(desktop_path,image_basename,boxes,box_labels,img_annotations.iloc[0][\"width\"],img_annotations.iloc[0][\"height\"])\n        \n    else:\n        boxes, scores, box_labels = non_maximum_weighted(boxes_list, scores_list, labels_list, weights=weights,\n                                    iou_thr=iou_thr)\n\n\n        #img_array.shape[1]是宽度\n        boxes = boxes*(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n        boxes = boxes.round(1).tolist()\n        box_labels = box_labels.astype(int).tolist()\n\n        boxes.extend(boxes_single)\n        box_labels.extend(labels_single)\n\n        print(\"Bboxes after nms:\\n\", boxes)\n        print(\"Labels after nms:\\n\", box_labels)\n\n        count_dict = Counter(box_labels)\n        print(count_dict)\n\n        text_create(desktop_path,image_basename,boxes,box_labels,img_annotations.iloc[0][\"width\"],img_annotations.iloc[0][\"height\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Create_weighted_boxes_fusion_box_txt(desktop_path,name):\n    iou_thr = 0.5\n    skip_box_thr = 0.0001\n    viz_images = []\n#     image_basename = Path(path).stem\n    image_basename = name\n#     print(f\"(\\'{image_basename}\\', \\'{path}\\')\")\n    img_annotations = train_annotations[train_annotations.image_id==image_basename]\n\n    boxes_viz = img_annotations[['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().tolist()\n    labels_viz = img_annotations['class_id'].to_numpy().tolist()\n\n    print(\"Bboxes before nms:\\n\", boxes_viz)\n    print(\"Labels before nms:\\n\", labels_viz)\n\n    boxes_list = []\n    scores_list = []\n    labels_list = []\n    weights = []\n\n    boxes_single = []\n    labels_single = []\n\n    cls_ids = img_annotations['class_id'].unique().tolist()\n    count_dict = Counter(img_annotations['class_id'].tolist())\n    print(count_dict)\n\n    for cid in cls_ids:       \n        ## Performing Fusing operation only for multiple bboxes with the same label\n        if count_dict[cid]==1:\n            labels_single.append(cid)\n            boxes_single.append(img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy().squeeze().tolist())\n\n        else:\n            cls_list =img_annotations[img_annotations.class_id==cid]['class_id'].tolist()\n            labels_list.append(cls_list)\n            bbox = img_annotations[img_annotations.class_id==cid][['x_min', 'y_min', 'x_max', 'y_max']].to_numpy()\n\n            ## Normalizing Bbox by Image Width and Height\n            bbox = bbox/(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n\n\n            bbox = np.clip(bbox, 0, 1)\n            boxes_list.append(bbox.tolist())\n\n            scores_list.append(np.ones(len(cls_list)).tolist())\n\n            weights.append(1)\n\n            \n    # Perform NMS\n    if len(boxes_list)==0:\n        boxes=boxes_single\n        box_labels=labels_single\n        print(\"Bboxes after nms:\\n\", boxes)\n        print(\"Labels after nms:\\n\", box_labels)\n        \n        count_dict = Counter(box_labels)\n        print(count_dict)\n\n        text_create(desktop_path,image_basename,boxes,box_labels,img_annotations.iloc[0][\"width\"],img_annotations.iloc[0][\"height\"])\n        \n    else:\n        boxes, scores, box_labels = weighted_boxes_fusion(boxes_list, scores_list, labels_list, weights=weights,\n                                    iou_thr=iou_thr)\n\n\n        #img_array.shape[1]是宽度\n        boxes = boxes*(img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"], img_annotations.iloc[0][\"width\"], img_annotations.iloc[0][\"height\"])\n        boxes = boxes.round(1).tolist()\n        box_labels = box_labels.astype(int).tolist()\n\n        boxes.extend(boxes_single)\n        box_labels.extend(labels_single)\n\n        print(\"Bboxes after nms:\\n\", boxes)\n        print(\"Labels after nms:\\n\", box_labels)\n\n        count_dict = Counter(box_labels)\n        print(count_dict)\n\n        text_create(desktop_path,image_basename,boxes,box_labels,img_annotations.iloc[0][\"width\"],img_annotations.iloc[0][\"height\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def text_create(desktop_path,name, boxes,box_labels,w,h):\n    \n    full_path = os.path.join(desktop_path, name+'.txt')  # 也可以创建一个.doc的word文档\n    print('full_path',full_path)\n    file = open(full_path, 'w')\n    \n    dw = 1. / (w)\n    dh = 1. / (h)\n    \n    for i in range(len(boxes)):\n\n        \n        x = (boxes[i][0] + boxes[i][2]) / 2.0\n        y = (boxes[i][1] + boxes[i][3]) / 2.0\n        w = boxes[i][2] - boxes[i][0]\n        h = boxes[i][3] - boxes[i][1]\n\n        \n        x = x * dw\n        w = w * dw\n        y = y * dh\n        h = h * dh\n        \n        \n        file.write(str(box_labels[i])+ ' '+ str(x)+ ' '+ str(y)+ ' '+ str(w)+ ' '+ str(h)+ ' '+ '\\n') \n        print(str(box_labels[i])+ ' '+ str(x)+ ' '+ str(y)+ ' '+ str(w)+ ' '+ str(h)+ ' ')\n    file.close()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.083752,"end_time":"2021-01-01T09:47:56.584924","exception":false,"start_time":"2021-01-01T09:47:56.501172","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Copying Files"},{"metadata":{"trusted":true},"cell_type":"code","source":"os.makedirs('/kaggle/working/vinbigdata/labels/train', exist_ok = True)\nos.makedirs('/kaggle/working/vinbigdata/labels/val', exist_ok = True)\nos.makedirs('/kaggle/working/vinbigdata/images/train', exist_ok = True)\nos.makedirs('/kaggle/working/vinbigdata/images/val', exist_ok = True)\nlabel_dir = '/kaggle/input/vinbigdata-yolo-labels-dataset/labels'\nfor file in tqdm(train_files):\n    shutil.copy(file, '/kaggle/working/vinbigdata/images/train')\n    filename = file.split('/')[-1].split('.')[0]\n    \n    # nms stuff\n    print(filename)\n    Create_softnms_box_txt('/kaggle/working/vinbigdata/labels/train',filename)\n    \n#     shutil.copy(os.path.join(label_dir, filename+'.txt'), '/kaggle/working/vinbigdata/labels/train')\n    \nfor file in tqdm(val_files):\n    shutil.copy(file, '/kaggle/working/vinbigdata/images/val')\n    filename = file.split('/')[-1].split('.')[0]\n    \n    # nms stuff\n    Create_softnms_box_txt('/kaggle/working/vinbigdata/labels/val',filename)\n#     shutil.copy(os.path.join(label_dir, filename+'.txt'), '/kaggle/working/vinbigdata/labels/val')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# import os\n# path = os.getcwd()#获取当前路径\n# print(path)\n\n# # all_files = [f for f in os.listdir('/kaggle/working/vinbigdata/labels/val' )]#输出根path下的所有文件名到一个列表中\n# all_files = [f for f in os.listdir(path )]#输出根path下的所有文件名到一个列表中\n# #对各个文件进行处理\n# print(all_files)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.068822,"end_time":"2021-01-01T09:50:01.458337","exception":false,"start_time":"2021-01-01T09:50:01.389515","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Get Class Name"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:50:01.595454Z","iopub.status.busy":"2021-01-01T09:50:01.594289Z","iopub.status.idle":"2021-01-01T09:50:01.60074Z","shell.execute_reply":"2021-01-01T09:50:01.601395Z"},"papermill":{"duration":0.082234,"end_time":"2021-01-01T09:50:01.601574","exception":false,"start_time":"2021-01-01T09:50:01.51934","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"class_ids, class_names = list(zip(*set(zip(train_df.class_id, train_df.class_name))))\nclasses = list(np.array(class_names)[np.argsort(class_ids)])\nclasses = list(map(lambda x: str(x), classes))\nclasses","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.056257,"end_time":"2021-01-01T09:50:01.716608","exception":false,"start_time":"2021-01-01T09:50:01.660351","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# [YOLOv5](https://github.com/ultralytics/yolov5)\n![](https://user-images.githubusercontent.com/26833433/98699617-a1595a00-2377-11eb-8145-fc674eb9b1a7.jpg)\n![](https://user-images.githubusercontent.com/26833433/90187293-6773ba00-dd6e-11ea-8f90-cd94afc0427f.png)"},{"metadata":{"papermill":{"duration":0.055699,"end_time":"2021-01-01T09:50:01.82747","exception":false,"start_time":"2021-01-01T09:50:01.771771","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# YOLOv5 Stuff"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:50:01.950234Z","iopub.status.busy":"2021-01-01T09:50:01.949285Z","iopub.status.idle":"2021-01-01T09:50:01.995866Z","shell.execute_reply":"2021-01-01T09:50:01.996316Z"},"papermill":{"duration":0.113001,"end_time":"2021-01-01T09:50:01.996448","exception":false,"start_time":"2021-01-01T09:50:01.883447","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"from os import listdir\nfrom os.path import isfile, join\nimport yaml\n\ncwd = '/kaggle/working/'\n\nwith open(join( cwd , 'train.txt'), 'w') as f:\n    for path in glob('/kaggle/working/vinbigdata/images/train/*'):\n        f.write(path+'\\n')\n            \nwith open(join( cwd , 'val.txt'), 'w') as f:\n    for path in glob('/kaggle/working/vinbigdata/images/val/*'):\n        f.write(path+'\\n')\n\ndata = dict(\n    train =  join( cwd , 'train.txt') ,\n    val   =  join( cwd , 'val.txt' ),\n    nc    = 14,\n    names = classes\n    )\n\nwith open(join( cwd , 'vinbigdata.yaml'), 'w') as outfile:\n    yaml.dump(data, outfile, default_flow_style=False)\n\nf = open(join( cwd , 'vinbigdata.yaml'), 'r')\nprint('\\nyaml:')\nprint(f.read())","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2021-01-01T09:50:02.170487Z","iopub.status.busy":"2021-01-01T09:50:02.169672Z","iopub.status.idle":"2021-01-01T09:50:08.782533Z","shell.execute_reply":"2021-01-01T09:50:08.783883Z"},"papermill":{"duration":6.702428,"end_time":"2021-01-01T09:50:08.784153","exception":false,"start_time":"2021-01-01T09:50:02.081725","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/ultralytics/yolov5\n# !git clone https://github.com/ultralytics/yolov5  # clone repo\n# %cd yolov5\nshutil.copytree('/kaggle/input/yolov5-official-v31-dataset/yolov5', '/kaggle/working/yolov5')\nos.chdir('/kaggle/working/yolov5')\n# %pip install -qr requirements.txt # install dependencies\n\nimport torch\nfrom IPython.display import Image, clear_output  # to display images\n\nclear_output()\nprint('Setup complete. Using torch %s %s' % (torch.__version__, torch.cuda.get_device_properties(0) if torch.cuda.is_available() else 'CPU'))","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:50:08.996106Z","iopub.status.busy":"2021-01-01T09:50:08.995246Z","iopub.status.idle":"2021-01-01T09:50:19.303278Z","shell.execute_reply":"2021-01-01T09:50:19.302769Z"},"papermill":{"duration":10.410768,"end_time":"2021-01-01T09:50:19.303402","exception":false,"start_time":"2021-01-01T09:50:08.892634","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"!python detect.py --weights yolov5s.pt --img 640 --conf 0.25 --source data/images/\nImage(filename='runs/detect/exp/zidane.jpg', width=600)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.064911,"end_time":"2021-01-01T09:50:19.435746","exception":false,"start_time":"2021-01-01T09:50:19.370835","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Pretrained Checkpoints:\n\n| Model | AP<sup>val</sup> | AP<sup>test</sup> | AP<sub>50</sub> | Speed<sub>GPU</sub> | FPS<sub>GPU</sub> || params | FLOPS |\n|---------- |------ |------ |------ | -------- | ------| ------ |------  |  :------: |\n| [YOLOv5s](https://github.com/ultralytics/yolov5/releases/tag/v3.0)    | 37.0     | 37.0     | 56.2     | **2.4ms** | **416** || 7.5M   | 13.2B\n| [YOLOv5m](https://github.com/ultralytics/yolov5/releases/tag/v3.0)    | 44.3     | 44.3     | 63.2     | 3.4ms     | 294     || 21.8M  | 39.4B\n| [YOLOv5l](https://github.com/ultralytics/yolov5/releases/tag/v3.0)    | 47.7     | 47.7     | 66.5     | 4.4ms     | 227     || 47.8M  | 88.1B\n| [YOLOv5x](https://github.com/ultralytics/yolov5/releases/tag/v3.0)    | **49.2** | **49.2** | **67.7** | 6.9ms     | 145     || 89.0M  | 166.4B\n| | | | | | || |\n| [YOLOv5x](https://github.com/ultralytics/yolov5/releases/tag/v3.0) + TTA|**50.8**| **50.8** | **68.9** | 25.5ms    | 39      || 89.0M  | 354.3B\n| | | | | | || |\n| [YOLOv3-SPP](https://github.com/ultralytics/yolov5/releases/tag/v3.0) | 45.6     | 45.5     | 65.2     | 4.5ms     | 222     || 63.0M  | 118.0B"},{"metadata":{"papermill":{"duration":0.064016,"end_time":"2021-01-01T09:50:19.564859","exception":false,"start_time":"2021-01-01T09:50:19.500843","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Selecting Models\nIn this notebok I'm using `v5s`. To select your prefered model just replace `--cfg models/yolov5s.yaml --weights yolov5s.pt` with the following command:\n* `v5s` : `--cfg models/yolov5s.yaml --weights yolov5s.pt`\n* `v5m` : `--cfg models/yolov5m.yaml --weights yolov5m.pt`\n* `v5l` : `--cfg models/yolov5l.yaml --weights yolov5l.pt`\n* `v5x` : `--cfg models/yolov5x.yaml --weights yolov5x.pt`"},{"metadata":{"papermill":{"duration":0.064553,"end_time":"2021-01-01T09:50:19.6938","exception":false,"start_time":"2021-01-01T09:50:19.629247","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Train"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T09:50:19.916161Z","iopub.status.busy":"2021-01-01T09:50:19.915216Z","iopub.status.idle":"2021-01-01T15:22:16.288743Z","shell.execute_reply":"2021-01-01T15:22:16.289579Z"},"papermill":{"duration":19916.498298,"end_time":"2021-01-01T15:22:16.289734","exception":false,"start_time":"2021-01-01T09:50:19.791436","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"# !WANDB_MODE=\"dryrun\" python train.py --img 640 --batch 16 --epochs 3 --data coco128.yaml --weights yolov5s.pt --nosave --cache \n!WANDB_MODE=\"dryrun\" python train.py --img 640 --batch 16 --epochs 50 --data /kaggle/working/vinbigdata.yaml --weights yolov5s.pt --cache","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":4.919442,"end_time":"2021-01-01T15:22:26.398681","exception":false,"start_time":"2021-01-01T15:22:21.479239","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Class Distribution"},{"metadata":{"trusted":true},"cell_type":"code","source":"a=1","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:22:36.714816Z","iopub.status.busy":"2021-01-01T15:22:36.713892Z","iopub.status.idle":"2021-01-01T15:22:37.752475Z","shell.execute_reply":"2021-01-01T15:22:37.752939Z"},"papermill":{"duration":6.511035,"end_time":"2021-01-01T15:22:37.753063","exception":false,"start_time":"2021-01-01T15:22:31.242028","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (20,20))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp/labels_correlogram.jpg'));","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:22:47.848164Z","iopub.status.busy":"2021-01-01T15:22:47.847303Z","iopub.status.idle":"2021-01-01T15:22:48.613974Z","shell.execute_reply":"2021-01-01T15:22:48.614481Z"},"papermill":{"duration":5.977042,"end_time":"2021-01-01T15:22:48.614609","exception":false,"start_time":"2021-01-01T15:22:42.637567","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (20,20))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp/labels.jpg'));","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":5.378338,"end_time":"2021-01-01T15:22:59.482837","exception":false,"start_time":"2021-01-01T15:22:54.104499","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Batch Image"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:23:10.192425Z","iopub.status.busy":"2021-01-01T15:23:10.19175Z","iopub.status.idle":"2021-01-01T15:23:11.776947Z","shell.execute_reply":"2021-01-01T15:23:11.777415Z"},"papermill":{"duration":7.317416,"end_time":"2021-01-01T15:23:11.777544","exception":false,"start_time":"2021-01-01T15:23:04.460128","status":"completed"},"tags":[],"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread('runs/train/exp/train_batch0.jpg'))\n\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread('runs/train/exp/train_batch1.jpg'))\n\nplt.figure(figsize = (15, 15))\nplt.imshow(plt.imread('runs/train/exp/train_batch2.jpg'))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# GT Vs Pred"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:23:22.104906Z","iopub.status.busy":"2021-01-01T15:23:22.10403Z","iopub.status.idle":"2021-01-01T15:23:23.51416Z","shell.execute_reply":"2021-01-01T15:23:23.514596Z"},"papermill":{"duration":6.453975,"end_time":"2021-01-01T15:23:23.514717","exception":false,"start_time":"2021-01-01T15:23:17.060742","status":"completed"},"tags":[],"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(3, 2, figsize = (2*5,3*5), constrained_layout = True)\nfor row in range(3):\n    ax[row][0].imshow(plt.imread(f'runs/train/exp/test_batch{row}_labels.jpg'))\n    ax[row][0].set_xticks([])\n    ax[row][0].set_yticks([])\n    ax[row][0].set_title(f'runs/train/exp/test_batch{row}_labels.jpg', fontsize = 12)\n    \n    ax[row][1].imshow(plt.imread(f'runs/train/exp/test_batch{row}_pred.jpg'))\n    ax[row][1].set_xticks([])\n    ax[row][1].set_yticks([])\n    ax[row][1].set_title(f'runs/train/exp/test_batch{row}_pred.jpg', fontsize = 12)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# (Loss, Map) Vs Epoch"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(30,15))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp/results.png'));","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Confusion Matrix"},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"plt.figure(figsize=(30,15))\nplt.axis('off')\nplt.imshow(plt.imread('runs/train/exp/confusion_matrix.png'));","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 0     3625\n# 3     2860\n# 13    1299\n# 11     763 0.5\n# 8      728\n# 10     699\n# 7      448\n# 9      357 0.67\n# 6      291\n# 5      290\n# 2      193\n# 4      121\n# 12      97\n# 1       46","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":4.941983,"end_time":"2021-01-01T15:23:33.765831","exception":false,"start_time":"2021-01-01T15:23:28.823848","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Inference"},{"metadata":{"_kg_hide-output":true,"execution":{"iopub.execute_input":"2021-01-01T15:23:44.50211Z","iopub.status.busy":"2021-01-01T15:23:44.501304Z","iopub.status.idle":"2021-01-01T15:23:49.800287Z","shell.execute_reply":"2021-01-01T15:23:49.799352Z"},"papermill":{"duration":10.763143,"end_time":"2021-01-01T15:23:49.800461","exception":false,"start_time":"2021-01-01T15:23:39.037318","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"!python detect.py --weights 'runs/train/exp/weights/best.pt'\\\n--img 640\\\n--conf 0.15\\\n--iou 0.5\\\n--source /kaggle/working/vinbigdata/images/val\\\n--exist-ok","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":5.225725,"end_time":"2021-01-01T15:24:00.706026","exception":false,"start_time":"2021-01-01T15:23:55.480301","status":"completed"},"tags":[]},"cell_type":"markdown","source":"# Inference Plot"},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:24:10.963448Z","iopub.status.busy":"2021-01-01T15:24:10.962552Z","iopub.status.idle":"2021-01-01T15:24:11.210595Z","shell.execute_reply":"2021-01-01T15:24:11.211731Z"},"papermill":{"duration":5.31015,"end_time":"2021-01-01T15:24:11.211904","exception":false,"start_time":"2021-01-01T15:24:05.901754","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom mpl_toolkits.axes_grid1 import ImageGrid\nimport numpy as np\nimport random\nimport cv2\nfrom glob import glob\nfrom tqdm import tqdm\n\nfiles = glob('runs/detect/exp/*')\nfor _ in range(3):\n    row = 4\n    col = 4\n    grid_files = random.sample(files, row*col)\n    images     = []\n    for image_path in tqdm(grid_files):\n        img          = cv2.cvtColor(cv2.imread(image_path), cv2.COLOR_BGR2RGB)\n        images.append(img)\n\n    fig = plt.figure(figsize=(col*5, row*5))\n    grid = ImageGrid(fig, 111,  # similar to subplot(111)\n                     nrows_ncols=(col, row),  # creates 2x2 grid of axes\n                     axes_pad=0.05,  # pad between axes in inch.\n                     )\n\n    for ax, im in zip(grid, images):\n        # Iterating over the grid returns the Axes.\n        ax.imshow(im)\n        ax.set_xticks([])\n        ax.set_yticks([])\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"a=1","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-01-01T15:24:21.60059Z","iopub.status.busy":"2021-01-01T15:24:21.599598Z","iopub.status.idle":"2021-01-01T15:24:22.413063Z","shell.execute_reply":"2021-01-01T15:24:22.411761Z"},"papermill":{"duration":5.709202,"end_time":"2021-01-01T15:24:22.413173","exception":false,"start_time":"2021-01-01T15:24:16.703971","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"# shutil.rmtree('/kaggle/working/vinbigdata')\n# shutil.rmtree('runs/detect')\n# for file in (glob('runs/train/exp/**/*.png', recursive = True)+glob('runs/train/exp/**/*.jpg', recursive = True)):\n#     os.remove(file)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dim = 512 #1024, 256, 'original'\ntest_dir = f'/kaggle/input/vinbigdata-{dim}-image-dataset/vinbigdata/test'\n# weights_dir = '/kaggle/input/vinbigdata-cxr-ad-yolov5-14-class-train/yolov5/runs/train/exp/weights/best.pt'\nweights_dir = f'runs/train/exp/weights/best.pt'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = pd.read_csv(f'/kaggle/input/vinbigdata-{dim}-image-dataset/vinbigdata/test.csv')\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.chdir('/kaggle/working/yolov5') # install dependencies\n\nimport torch\nfrom IPython.display import Image, clear_output  # to display images\n\nclear_output()\nprint('Setup complete. Using torch %s %s' % (torch.__version__, torch.cuda.get_device_properties(0) if torch.cuda.is_available() else 'CPU'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!python detect.py --weights $weights_dir\\\n--img 640\\\n--conf 0.15\\\n--iou 0.4\\\n--source $test_dir\\\n--save-txt --save-conf --exist-ok","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def yolo2voc(image_height, image_width, bboxes):\n    \"\"\"\n    yolo => [xmid, ymid, w, h] (normalized)\n    voc  => [x1, y1, x2, y1]\n    \n    \"\"\" \n#     print('bboxes',bboxes)\n    bboxes = bboxes.copy().astype(float) # otherwise all value will be 0 as voc_pascal dtype is np.int\n    \n    bboxes[..., [0, 2]] = bboxes[..., [0, 2]]* image_width\n    bboxes[..., [1, 3]] = bboxes[..., [1, 3]]* image_height\n    \n    bboxes[..., [0, 1]] = bboxes[..., [0, 1]] - bboxes[..., [2, 3]]/2\n    bboxes[..., [2, 3]] = bboxes[..., [0, 1]] + bboxes[..., [2, 3]]\n    \n    return bboxes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_ids = []\nPredictionStrings = []\n\nfor file_path in tqdm(glob('runs/detect/exp/labels/*txt')):\n    image_id = file_path.split('/')[-1].split('.')[0]\n    w, h = test_df.loc[test_df.image_id==image_id,['width', 'height']].values[0]\n    f = open(file_path, 'r')\n    data = np.array(f.read().replace('\\n', ' ').strip().split(' ')).astype(np.float32).reshape(-1, 6)\n    data = data[:, [0, 5, 1, 2, 3, 4]]\n    bboxes = list(np.round(np.concatenate((data[:, :2], np.round(yolo2voc(h, w, data[:, 2:]))), axis =1).reshape(-1), 1).astype(str))\n    for idx in range(len(bboxes)):\n        bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n    image_ids.append(image_id)\n    PredictionStrings.append(' '.join(bboxes))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_df_pure = pd.DataFrame({'image_id':image_ids,\n                        'PredictionString':PredictionStrings})\nsub_df_pure = pd.merge(test_df, pred_df_pure, on = 'image_id', how = 'left').fillna(\"14 1 0 0 1 1\")\nsub_df_pure = sub_df_pure[['image_id', 'PredictionString']]\nsub_df_pure.to_csv('/kaggle/working/sub_df_pure.csv',index = False)\nsub_df_pure.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"display(sub_df_pure)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"a=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom mpl_toolkits.axes_grid1 import ImageGrid\nimport numpy as np\nimport random\nimport cv2\nfrom glob import glob\nfrom tqdm import tqdm\n\nfiles = glob('runs/detect/exp/*png')\nfor _ in range(6):\n    row = 4\n    col = 4\n    grid_files = random.sample(files, row*col)\n    images     = []\n    for image_path in tqdm(grid_files):\n        img          = cv2.cvtColor(cv2.imread(image_path), cv2.COLOR_BGR2RGB)\n        images.append(img)\n\n    fig = plt.figure(figsize=(col*5, row*5))\n    grid = ImageGrid(fig, 111,  # similar to subplot(111)\n                     nrows_ncols=(col, row),  # creates 2x2 grid of axes\n                     axes_pad=0.05,  # pad between axes in inch.\n                     )\n\n    for ax, im in zip(grid, images):\n        # Iterating over the grid returns the Axes.\n        ax.imshow(im)\n        ax.set_xticks([])\n        ax.set_yticks([])\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.notebook import tqdm\nimport seaborn as sns\nfrom scipy.stats import gaussian_kde\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom scipy.stats import gaussian_kde\n\ndim = 512 #512, 256, 'original'\nfold = 4\n\ntrain_df = pd.read_csv(f'/kaggle/input/vinbigdata-{dim}-image-dataset/vinbigdata/train.csv')\ntrain_df.head()\n\ntrain_df['x_min'] = train_df.apply(lambda row: (row.x_min)/row.width, axis =1)\ntrain_df['y_min'] = train_df.apply(lambda row: (row.y_min)/row.height, axis =1)\n\ntrain_df['x_max'] = train_df.apply(lambda row: (row.x_max)/row.width, axis =1)\ntrain_df['y_max'] = train_df.apply(lambda row: (row.y_max)/row.height, axis =1)\n\ntrain_df['x_mid'] = train_df.apply(lambda row: (row.x_max+row.x_min)/2, axis =1)\ntrain_df['y_mid'] = train_df.apply(lambda row: (row.y_max+row.y_min)/2, axis =1)\n\ntrain_df['w'] = train_df.apply(lambda row: (row.x_max-row.x_min), axis =1)\ntrain_df['h'] = train_df.apply(lambda row: (row.y_max-row.y_min), axis =1)\n\ntrain_df['area'] = train_df['w']*train_df['h']\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = train_df[train_df.class_id!=14].reset_index(drop = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list_mid_gaussian_kde=[]\nfor i in range(14):\n    \n\n    train_df_0 = train_df[train_df.class_id==i]\n    x_val = train_df_0.x_mid\n    y_val = train_df_0.y_mid\n    print(len(x_val))\n\n    # Calculate the point density\n    xy = np.vstack([x_val,y_val])\n    z = gaussian_kde(xy)(xy)\n\n    fig, ax = plt.subplots(figsize = (10, 10))\n    # ax.axis('off')\n    ax.axis([0,1,1,0])\n    ax.scatter(x_val, y_val, c=z, s=100, cmap='viridis')\n    # ax.set_xlabel('x_mid')\n    # ax.set_ylabel('y_mid')\n    plt.show()\n\n    evalutor=gaussian_kde(xy)\n    # evalutor.evaluate([[0.6],[0.4]])\n    list_mid_gaussian_kde.append(evalutor)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list_w_h_gaussian_kde=[]\nfor i in range(14):\n    \n\n    train_df_0 = train_df[train_df.class_id==i]\n    x_val = train_df_0.w\n    y_val = train_df_0.h\n    print(len(x_val))\n\n    # Calculate the point density\n    xy = np.vstack([x_val,y_val])\n    z = gaussian_kde(xy)(xy)\n\n    fig, ax = plt.subplots(figsize = (10, 10))\n    # ax.axis('off')\n    ax.axis([0,1,1,0])\n    ax.scatter(x_val, y_val, c=z, s=100, cmap='viridis')\n    # ax.set_xlabel('x_mid')\n    # ax.set_ylabel('y_mid')\n    plt.show()\n\n    evalutor=gaussian_kde(xy)\n    # evalutor.evaluate([[0.6],[0.4]])\n    list_w_h_gaussian_kde.append(evalutor)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.pythonheidong.com/blog/article/436765/a0facf9464d9337dc1eb/\nlist_area_gaussian_kde=[]\nfor i in range(14):\n    train_df_0 = train_df[train_df.class_id==i]\n    data = train_df_0.w\n\n    density = gaussian_kde(data)\n    xs = np.linspace(0,1,200)\n    plt.plot(xs,density(xs))\n    plt.show()\n    list_area_gaussian_kde.append(density)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Hotmask_mid_value(class_id,normalized_xmid_ymid_w_h):\n    evaluator = list_mid_gaussian_kde[int(class_id)]\n    value = evaluator.evaluate([[normalized_xmid_ymid_w_h[0]],[normalized_xmid_ymid_w_h[1]]])\n    \n    return value\n\ndef Hotmask_w_h_value(class_id,normalized_xmid_ymid_w_h):\n    evaluator = list_w_h_gaussian_kde[int(class_id)]\n    value = evaluator.evaluate([[normalized_xmid_ymid_w_h[2]],[normalized_xmid_ymid_w_h[3]]])\n    \n    return value\n\ndef Hotmask_area_value(class_id,normalized_xmid_ymid_w_h):\n    evaluator = list_area_gaussian_kde[int(class_id)]\n    value = evaluator.evaluate(normalized_xmid_ymid_w_h[2]*normalized_xmid_ymid_w_h[3])\n    \n    return value","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_ids = []\nPredictionStrings = []\nsum_delte_on=0\n\n# threshold=0.001\nthreshold_mid=0.001\nthreshold_w_h=0.001\nthreshold_area=0.001\n\nfor file_path in tqdm(glob('runs/detect/exp/labels/*txt')):\n    delte_on=0\n    \n    image_id = file_path.split('/')[-1].split('.')[0]\n    w, h = test_df.loc[test_df.image_id==image_id,['width', 'height']].values[0]\n    f = open(file_path, 'r')\n    data = np.array(f.read().replace('\\n', ' ').strip().split(' ')).astype(np.float32).reshape(-1, 6)\n    data = data[:, [0, 5, 1, 2, 3, 4]]\n#     print('data',data.type)\n    \n    list_classid_probility = data[:, :2]\n    list_normalized_box = data[:, 2:]\n    delte_sign=[]\n#     filtered_list_classid_probility=[]\n#     filtered_list_normalized_box=[]\n    \n#     print('data[:, :2]',data[:, :2]) \n    \n    for i in range(len(list_classid_probility)):\n        if (Hotmask_mid_value(list_classid_probility[i][0] , list_normalized_box[i])>threshold_mid and Hotmask_w_h_value(list_classid_probility[i][0] , list_normalized_box[i])>threshold_w_h and Hotmask_area_value(list_classid_probility[i][0] , list_normalized_box[i])>threshold_area) or list_classid_probility[i][0]==14:\n            delte_sign.append(True)\n        else:\n            delte_sign.append(False)\n            delte_on+=1\n            \n#             filtered_list_classid_probility.append(list_classid_probility[i][0])\n#             filtered_list_normalized_box.append(list_classid_probility[i].tolist())\n            \n#     prin**t('filtered_list_normalized_box',filtered_list_normalized_box) \n    \n    print('delte_sign',delte_sign) \n    bboxes = list(np.round(np.concatenate((data[:, :2][delte_sign], np.round(yolo2voc(h, w, data[:, 2:][delte_sign]))), axis =1).reshape(-1), 1).astype(str))\n#     bboxes = np.array(list(np.round(np.concatenate((filtered_list_classid_probility, np.round(yolo2voc(h, w, filtered_list_normalized_box))), axis =1).reshape(-1), 1).astype(str)),dtype=np.float32)\n#     print('data[:, 2:]',data[:, 2:])\n    for idx in range(len(bboxes)):\n        bboxes[idx] = str(int(float(bboxes[idx]))) if idx%6!=1 else bboxes[idx]\n    image_ids.append(image_id)\n    PredictionStrings.append(' '.join(bboxes))\n    \n    if delte_on!=0:\n        sum_delte_on+=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_df = pd.DataFrame({'image_id':image_ids,\n                        'PredictionString':PredictionStrings})\nsub_df = pd.merge(test_df, pred_df, on = 'image_id', how = 'left').fillna(\"14 1 0 0 1 1\")\nsub_df = sub_df[['image_id', 'PredictionString']]\n\nsub_df.loc[sub_df['PredictionString'] =='','PredictionString']=\"14 1 0 0 1 1\"\nsub_df.to_csv('/kaggle/working/submission_softnms.csv',index = False)\nsub_df.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"display(sub_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom glob import glob\nimport shutil","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os \nos.getcwd()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_14cls =sub_df\npred_2cls = pd.read_csv('/kaggle/input/vinbigdata-2class-prediction/2-cls test pred.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"count_low=0\ncount_mid=0\ncount_high=0\ndef filter_2cls_raw(row, low_thr=0.08, high_thr=0.95):\n    global count_low\n    global count_mid\n    global count_high\n    prob = row['target']\n    if prob<low_thr:\n        ## Less chance of having any disease\n        row['PredictionString'] = '14 1 0 0 1 1'\n        count_low+=1\n    elif low_thr<=prob<high_thr:\n        ## More change of having any diesease\n        row['PredictionString']+=f' 14 {prob} 0 0 1 1'\n        count_mid+=1\n    elif high_thr<=prob:\n        ## Good chance of having any disease so believe in object detection model\n        row['PredictionString'] = row['PredictionString']\n        count_high+=1\n    else:\n        raise ValueError('Prediction must be from [0-1]')\n    return row","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_raw = pd.merge(pred_14cls, pred_2cls, on = 'image_id', how = 'left')\npred_raw.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_raw = pred_raw.apply(filter_2cls_raw, axis=1)\nprint(count_low/3000,count_mid/3000,count_high/3000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_raw[['image_id', 'PredictionString']].to_csv('submission_raw.csv',index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print_df=sub_raw[['image_id', 'PredictionString']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"display(print_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# for i in range(sub_raw.shape[0]):\n# #     sub_raw.iloc[i]['image_id', 'PredictionString']\n#     print(print_df.loc[i])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}