{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -U ensemble-boxes","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:04:55.147127Z","iopub.execute_input":"2023-01-20T19:04:55.14768Z","iopub.status.idle":"2023-01-20T19:05:01.26435Z","shell.execute_reply.started":"2023-01-20T19:04:55.147643Z","shell.execute_reply":"2023-01-20T19:05:01.262689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom ensemble_boxes import *\nfrom glob import glob\nimport copy\nfrom tqdm import tqdm\nimport shutil","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-20T19:05:01.266009Z","iopub.execute_input":"2023-01-20T19:05:01.266276Z","iopub.status.idle":"2023-01-20T19:05:01.272082Z","shell.execute_reply.started":"2023-01-20T19:05:01.266247Z","shell.execute_reply":"2023-01-20T19:05:01.270396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()\nsubs = [\n    pd.read_csv('/kaggle/input/first-ensembling-scores/54_96.csv'),\n    pd.read_csv('/kaggle/input/first-ensembling-scores/55_22.csv'),\n    pd.read_csv('/kaggle/input/first-ensembling-scores/55_26.csv'),\n    pd.read_csv('/kaggle/input/first-ensembling-scores/55_40.csv'),\n    pd.read_csv('/kaggle/input/first-ensembling-scores/55_81.csv') \n       ]","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:05:01.273295Z","iopub.execute_input":"2023-01-20T19:05:01.273627Z","iopub.status.idle":"2023-01-20T19:05:01.345321Z","shell.execute_reply.started":"2023-01-20T19:05:01.273594Z","shell.execute_reply":"2023-01-20T19:05:01.343847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def update(row):\n    if (row['xmax']>1920):\n        row['xmax'] = 1920\n    if (row['xmin']<0):\n        row['xmin'] = 0\n    if (row['ymax']>1080):\n        row['ymax'] = 1080\n    if (row['ymin']<0):\n        row['ymin'] = 0\n    return row\nsubs[4] = subs[4].progress_apply(update,axis=1)\nsubs[3] = subs[3].progress_apply(update,axis=1)\nsubs[2] = subs[2].progress_apply(update,axis=1)\nsubs[1] = subs[1].progress_apply(update,axis=1)\nsubs[0] = subs[0].progress_apply(update,axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:05:01.346714Z","iopub.execute_input":"2023-01-20T19:05:01.346997Z","iopub.status.idle":"2023-01-20T19:05:03.675661Z","shell.execute_reply.started":"2023-01-20T19:05:01.346962Z","shell.execute_reply":"2023-01-20T19:05:03.673953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_paths = list(subs[0].image_path.unique())\nimage_paths[:5]","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:05:03.678848Z","iopub.execute_input":"2023-01-20T19:05:03.67924Z","iopub.status.idle":"2023-01-20T19:05:03.687041Z","shell.execute_reply.started":"2023-01-20T19:05:03.679202Z","shell.execute_reply":"2023-01-20T19:05:03.686047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxes_dict = {}\nscores_dict = {}\nlabels_dict = {}\nwhwh_dict = {}\n\nfor i in tqdm(subs[0].image_path.unique()):\n    if not i in boxes_dict.keys():\n        boxes_dict[i] = []\n        scores_dict[i] = []\n        labels_dict[i] = []\n        whwh_dict[i] = []\n\n#     size_ratio = fnl_dict.get(i)\n    size_ratio = [1920.0,1080.0,1920.0,1080.0]\n    whwh_dict[i].append(size_ratio) \n    tmp_df = [subs[x][subs[x]['image_path']==i] for x in range(len(subs))]\n    for j in range(len(tmp_df)):\n        tmp = tmp_df[j]\n        tmp['score']=1\n#     print(tmp_df)\n    for x in range(len(tmp_df)):\n        boxes_dict[i].append(((tmp_df[x][['xmin','ymin','xmax','ymax']].values)/size_ratio).tolist())\n        scores_dict[i].append(tmp_df[x]['score'].values.tolist())\n        labels_dict[i].append(tmp_df[x]['class'].values.tolist())","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:05:03.688837Z","iopub.execute_input":"2023-01-20T19:05:03.689401Z","iopub.status.idle":"2023-01-20T19:05:27.909085Z","shell.execute_reply.started":"2023-01-20T19:05:03.68936Z","shell.execute_reply":"2023-01-20T19:05:27.9084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs[1]['name'].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:05:27.910354Z","iopub.execute_input":"2023-01-20T19:05:27.910649Z","iopub.status.idle":"2023-01-20T19:05:27.916559Z","shell.execute_reply.started":"2023-01-20T19:05:27.910621Z","shell.execute_reply":"2023-01-20T19:05:27.915773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes = ['GRAFFITI',\n 'FADED_SIGNAGE',\n 'POTHOLES',\n 'GARBAGE',\n 'CONSTRUCTION_ROAD',\n 'BROKEN_SIGNAGE',\n 'BAD_STREETLIGHT',\n 'BAD_BILLBOARD',\n 'SAND_ON_ROAD',\n 'CLUTTER_SIDEWALK',\n 'UNKEPT_FACADE']\nclasses","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:05:27.918456Z","iopub.execute_input":"2023-01-20T19:05:27.918886Z","iopub.status.idle":"2023-01-20T19:05:27.933764Z","shell.execute_reply.started":"2023-01-20T19:05:27.918837Z","shell.execute_reply":"2023-01-20T19:05:27.931817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/first-ensembling-scores/54_96.csv').head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T18:10:00.173124Z","iopub.execute_input":"2023-01-20T18:10:00.17356Z","iopub.status.idle":"2023-01-20T18:10:00.199541Z","shell.execute_reply.started":"2023-01-20T18:10:00.173521Z","shell.execute_reply":"2023-01-20T18:10:00.198572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# weights = [1]*5\n# weights += [3]\nweights = [1,2,3,4,6]\niou_thr = 0.2875\nskip_box_thr = 0.0\nsigma = 0.1\n\nfnl = {}\n\nfor i in tqdm(boxes_dict.keys()):\n    \n    \n    boxes, scores, labels = weighted_boxes_fusion(boxes_dict[i], scores_dict[i], labels_dict[i],\\\n                                                  weights=weights, iou_thr=iou_thr, skip_box_thr=skip_box_thr)\n    \n    \n    \n    \n    \n#     boxes1, scores1, labels1 = nms(boxes_dict[i], scores_dict[i], labels_dict[i], weights=weights, iou_thr=iou_thr)\n\n    \n    \n    \n    \n    \n#     boxes0, scores0, labels0 = soft_nms(boxes_dict[i], scores_dict[i], labels_dict[i], weights=weights,\\\n#                                      iou_thr=iou_thr, sigma=sigma, thresh=skip_box_thr)\n    \n#     boxes2, scores2, labels2 = non_maximum_weighted(boxes_dict[i], scores_dict[i], labels_dict[i], weights=weights,\n#                                                                 skip_box_thr=skip_box_thr)\n\n    \n    \n    \n#     boxes, scores, labels = weighted_boxes_fusion([boxes0,boxes1,boxes2,boxes3],\\\n#                                                   [scores0,scores1,scores2,scores3],\\\n#                                                   [labels0,labels1,labels2,labels3],\\\n#                                                   weights=weights1, iou_thr=iou_thr, skip_box_thr=skip_box_thr)\n    \n    if not i in fnl.keys():\n        fnl[i] = {'boxes':[],'scores':[],'labels':[]}\n        \n    fnl[i]['boxes'] = boxes*whwh_dict[i]\n    fnl[i]['scores'] = scores\n    fnl[i]['labels'] = labels","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:16:21.393532Z","iopub.execute_input":"2023-01-20T19:16:21.393873Z","iopub.status.idle":"2023-01-20T19:16:22.843374Z","shell.execute_reply.started":"2023-01-20T19:16:21.393841Z","shell.execute_reply":"2023-01-20T19:16:22.842395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fnl","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:16:22.845009Z","iopub.execute_input":"2023-01-20T19:16:22.845305Z","iopub.status.idle":"2023-01-20T19:16:22.848551Z","shell.execute_reply.started":"2023-01-20T19:16:22.845273Z","shell.execute_reply":"2023-01-20T19:16:22.847702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/first-ensembling-scores/54_96.csv').head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:16:22.849665Z","iopub.execute_input":"2023-01-20T19:16:22.849961Z","iopub.status.idle":"2023-01-20T19:16:22.880152Z","shell.execute_reply.started":"2023-01-20T19:16:22.849926Z","shell.execute_reply":"2023-01-20T19:16:22.878714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs[0]['class'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:16:22.881817Z","iopub.execute_input":"2023-01-20T19:16:22.882132Z","iopub.status.idle":"2023-01-20T19:16:22.889406Z","shell.execute_reply.started":"2023-01-20T19:16:22.882101Z","shell.execute_reply":"2023-01-20T19:16:22.888474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd_form = []\nfor i in fnl.keys():\n    b = fnl[i]\n    for j in range(len(b['boxes'])):\n        class_id = int(b['labels'][j])\n        image_path = i\n        name = classes[int(b['labels'][j])]\n        xmax = int(b['boxes'][j][2])\n        xmin = int(b['boxes'][j][0])\n        ymax = int(b['boxes'][j][3])\n        ymin = int(b['boxes'][j][1])\n        pd_form.append([class_id,image_path,name,xmax,xmin,ymax,ymin])\n        \nfinal_df = pd.DataFrame(pd_form,columns = ['class','image_path','name','xmax','xmin','ymax','ymin'])\n# final_df = final_df.drop_duplicates(keep = 'first')\nfinal_df","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:16:22.891061Z","iopub.execute_input":"2023-01-20T19:16:22.89136Z","iopub.status.idle":"2023-01-20T19:16:22.948806Z","shell.execute_reply.started":"2023-01-20T19:16:22.891329Z","shell.execute_reply":"2023-01-20T19:16:22.947816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_img = list(final_df['image_path'].unique())\nfinal_img[:5]","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:16:23.528622Z","iopub.execute_input":"2023-01-20T19:16:23.528986Z","iopub.status.idle":"2023-01-20T19:16:23.540693Z","shell.execute_reply.started":"2023-01-20T19:16:23.528949Z","shell.execute_reply":"2023-01-20T19:16:23.539315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for img_path in image_paths:\n    if (img_path not in final_img):\n        final_df = final_df.append({\n                    'class':6,\n                    'image_path':img_path,\n                    'name':classes[6],\n                    'xmax':0,\n                    'xmin':0,\n                    'ymax':0,\n                    'ymin':0,\n        },ignore_index=True)\nfinal_df","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:16:23.711606Z","iopub.execute_input":"2023-01-20T19:16:23.711931Z","iopub.status.idle":"2023-01-20T19:16:24.133555Z","shell.execute_reply.started":"2023-01-20T19:16:23.711905Z","shell.execute_reply":"2023-01-20T19:16:24.132897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.to_csv('final_df12.csv',index=None)","metadata":{"execution":{"iopub.status.busy":"2023-01-20T19:16:28.496342Z","iopub.execute_input":"2023-01-20T19:16:28.496906Z","iopub.status.idle":"2023-01-20T19:16:28.517343Z","shell.execute_reply.started":"2023-01-20T19:16:28.496864Z","shell.execute_reply":"2023-01-20T19:16:28.515884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submission_encoder(df:pd.DataFrame) -> np.ndarray:\n    dct = {}\n    for i in tqdm(df['image_id'].unique()):\n        if not i in dct.keys():\n            dct[i] = []\n        tmp = df[df['image_id'] == i].values\n        for j in tmp:\n            dct[i].append(int(j[1]))\n            dct[i].append(float(j[2]))\n            dct[i].append(int(j[3]))\n            dct[i].append(int(j[4]))\n            dct[i].append(int(j[5]))\n            dct[i].append(int(j[6]))\n        \n        dct[i] = map(str,dct[i])\n        dct[i] = ' '.join(dct[i])\n    dct = [[k, v] for k, v in dct.items()]\n    return pd.DataFrame(dct,columns = ['image_id','PredictionString']).reset_index(drop = True)\n\ndf = submission_encoder(final_df)\ndf.to_csv('Fold5Yolo.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NORMAL = \"14 1 0 0 1 1\"\nlow_threshold = 0.00\nhigh_threshold = 0.99\npred_det_df = df  # You can load from another submission.csv here too.\nn_normal_before = len(pred_det_df.query(\"PredictionString == @NORMAL\"))\nmerged_df = pd.merge(pred_det_df, pred_2cls, on=\"image_id\", how=\"left\")\n\nif \"target\" in merged_df.columns:\n    merged_df[\"class0\"] = 1 - merged_df[\"target\"]\n\nc0, c1, c2 = 0, 0, 0\nfor i in range(len(merged_df)):\n    p0 = merged_df.loc[i, \"class0\"]\n    if p0 < low_threshold:\n        # Keep, do nothing.\n        c0 += 1\n    elif low_threshold <= p0 and p0 < high_threshold:\n        # Add, keep \"det\" preds and add normal pred.\n        if ' 14 ' not in merged_df.loc[i, \"PredictionString\"]:\n            merged_df.loc[i, \"PredictionString\"] += f\" 14 {p0} 0 0 1 1\"\n            \n        c1 += 1\n    else:\n        # Replace, remove all \"det\" preds.\n        merged_df.loc[i, \"PredictionString\"] = NORMAL\n        c2 += 1\n\nn_normal_after = len(merged_df.query(\"PredictionString == @NORMAL\"))\nprint(\n    f\"n_normal: {n_normal_before} -> {n_normal_after} with threshold {low_threshold} & {high_threshold}\"\n)\nprint(f\"Keep {c0} Add {c1} Replace {c2}\")\nsubmission_filepath = str(\"submission.csv\")\nsubmission_df = merged_df[[\"image_id\", \"PredictionString\"]]\nsubmission_df.to_csv(submission_filepath, index=False)\nprint(f\"Saved to {submission_filepath}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}