{"cells":[{"metadata":{},"cell_type":"markdown","source":"### Credit of this notebook goes entirely to below public notebook, kindly upvote and appreciate the original author\n\n* https://www.kaggle.com/muhammad4hmed/lets-overfit-together"},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2021-03-10T12:24:05.989303Z","iopub.status.busy":"2021-03-10T12:24:05.988642Z","iopub.status.idle":"2021-03-10T12:24:05.99164Z","shell.execute_reply":"2021-03-10T12:24:05.991116Z"},"papermill":{"duration":0.012986,"end_time":"2021-03-10T12:24:05.991832","exception":false,"start_time":"2021-03-10T12:24:05.978846","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-03-10T12:24:06.003559Z","iopub.status.busy":"2021-03-10T12:24:06.002982Z","iopub.status.idle":"2021-03-10T12:24:06.058383Z","shell.execute_reply":"2021-03-10T12:24:06.05879Z"},"papermill":{"duration":0.06342,"end_time":"2021-03-10T12:24:06.058979","exception":false,"start_time":"2021-03-10T12:24:05.995559","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"pred_2class = pd.read_csv(\"../input/vinbigdata-2class-prediction/2-cls test pred.csv\")\nlow_threshold = 0.001\nhigh_threshold = 0.88\npred_2class","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.003752,"end_time":"2021-03-10T12:24:06.067026","exception":false,"start_time":"2021-03-10T12:24:06.063274","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Apply 2class filter"},{"metadata":{"execution":{"iopub.execute_input":"2021-03-10T12:24:06.083599Z","iopub.status.busy":"2021-03-10T12:24:06.082655Z","iopub.status.idle":"2021-03-10T12:24:07.134898Z","shell.execute_reply":"2021-03-10T12:24:07.135344Z"},"papermill":{"duration":1.064509,"end_time":"2021-03-10T12:24:07.135516","exception":false,"start_time":"2021-03-10T12:24:06.071007","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"NORMAL = \"14 1 0 0 1 1\"\n\npred_det_df = pd.read_csv(\"../input/vinbigdatastack/submission_postprocessed.csv\")\nn_normal_before = len(pred_det_df.query(\"PredictionString == @NORMAL\"))\nmerged_df = pd.merge(pred_det_df, pred_2class, on=\"image_id\", how=\"left\")\n\n\nif \"target\" in merged_df.columns:\n    merged_df[\"class0\"] = 1 - merged_df[\"target\"]\n\nc0, c1, c2 = 0, 0, 0\nfor i in range(len(merged_df)):\n    p0 = merged_df.loc[i, \"class0\"]\n    if p0 < low_threshold:\n\n        c0 += 1\n    elif low_threshold <= p0 and p0 < high_threshold:\n\n        merged_df.loc[i, \"PredictionString\"] += f\" 14 {p0} 0 0 1 1\"\n        c1 += 1\n    else:\n\n        merged_df.loc[i, \"PredictionString\"] = NORMAL\n        c2 += 1\n\nn_normal_after = len(merged_df.query(\"PredictionString == @NORMAL\"))\nprint(\n    f\"n_normal: {n_normal_before} -> {n_normal_after} with threshold {low_threshold} & {high_threshold}\"\n)\nprint(f\"Keep {c0} Add {c1} Replace {c2}\")\nsubmission_filepath = str(\"submission.csv\")\nsubmission_df = merged_df[[\"image_id\", \"PredictionString\"]]\nsubmission_df.to_csv(submission_filepath, index=False)\nprint(f\"Saved to {submission_filepath}\")\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}