{"cells":[{"metadata":{},"cell_type":"markdown","source":"# [VinBigData Chest X-ray Abnormalities Detection](https://www.kaggle.com/c/vinbigdata-chest-xray-abnormalities-detection)"},{"metadata":{},"cell_type":"markdown","source":"### Acknowledgements:\n\n* notebook [VinBigData Chest X-ray Abnormalities Detection](https://www.kaggle.com/mahmudds/vinbigdata-chest-x-ray-abnormalities-detection)\n* notebook [Lets overfit together](https://www.kaggle.com/muhammad4hmed/lets-overfit-together)\n* dataset [VinBigData Chest X-ray models](https://www.kaggle.com/bryanb/vinbigdatastack)\n* dataset [VinBigData 2-class Prediction](https://www.kaggle.com/awsaf49/vinbigdata-2class-prediction) "},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2021-03-10T12:24:05.989303Z","iopub.status.busy":"2021-03-10T12:24:05.988642Z","iopub.status.idle":"2021-03-10T12:24:05.99164Z","shell.execute_reply":"2021-03-10T12:24:05.991116Z"},"papermill":{"duration":0.012986,"end_time":"2021-03-10T12:24:05.991832","exception":false,"start_time":"2021-03-10T12:24:05.978846","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt \nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-03-10T12:24:06.003559Z","iopub.status.busy":"2021-03-10T12:24:06.002982Z","iopub.status.idle":"2021-03-10T12:24:06.058383Z","shell.execute_reply":"2021-03-10T12:24:06.05879Z"},"papermill":{"duration":0.06342,"end_time":"2021-03-10T12:24:06.058979","exception":false,"start_time":"2021-03-10T12:24:05.995559","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"pred_2class = pd.read_csv(\"../input/vinbigdata-2class-prediction/2-cls test pred.csv\")\npred_2class","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Tuning by LB private - the best option: my version (commit) 5:\nlow_threshold = 0.005\nhigh_threshold = 0.95","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Tuning by LB private\ncommits_df = pd.DataFrame(columns = ['n_commit', 'low', 'high', 'LB_score'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 0th version - [the starting point](https://www.kaggle.com/mahmudds/vinbigdata-chest-x-ray-abnormalities-detection) "},{"metadata":{"trusted":true},"cell_type":"code","source":"# From https://www.kaggle.com/mahmudds/vinbigdata-chest-x-ray-abnormalities-detection\nn=0\ncommits_df.loc[n, 'n_commit'] = 0                        # I didn't run it - that's the starting point\ncommits_df.loc[n, 'low'] = 0.001                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.87                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.246                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### My versions (commits)"},{"metadata":{"trusted":true},"cell_type":"code","source":"n=1\ncommits_df.loc[n, 'n_commit'] = n                        # Number of commit\ncommits_df.loc[n, 'low'] = 0.001                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.90                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.244                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=2\ncommits_df.loc[n, 'n_commit'] = n                        # Number of commit\ncommits_df.loc[n, 'low'] = 0.001                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.94                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.241                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.228                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=3\ncommits_df.loc[n, 'n_commit'] = n                        # Number of commit\ncommits_df.loc[n, 'low'] = 0.0                           # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.91                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.243                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=4\ncommits_df.loc[n, 'n_commit'] = n                        # Number of commit\ncommits_df.loc[n, 'low'] = 0.002                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.9                          # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.244                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=5\ncommits_df.loc[n, 'n_commit'] = n                        # Number of commit\ncommits_df.loc[n, 'low'] = 0.005                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.95                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.241                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.228                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=6\ncommits_df.loc[n, 'n_commit'] = n+1                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.001                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.9                          # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.244                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=7\ncommits_df.loc[n, 'n_commit'] = n+3                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.001                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.88                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.246                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=8\ncommits_df.loc[n, 'n_commit'] = n+3                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.001                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.86                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.245                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=9\ncommits_df.loc[n, 'n_commit'] = n+3                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.001                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.875                        # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.246                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.225                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=10\ncommits_df.loc[n, 'n_commit'] = n+3                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.0                           # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.875                        # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.246                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.225                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=11\ncommits_df.loc[n, 'n_commit'] = n+3                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.002                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.885                        # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.246                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=12\ncommits_df.loc[n, 'n_commit'] = n+3                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.003                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.88                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.246                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=13\ncommits_df.loc[n, 'n_commit'] = n+3                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.005                         # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.88                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.246                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=14\ncommits_df.loc[n, 'n_commit'] = n+3                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.01                          # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.88                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.246                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=15\ncommits_df.loc[n, 'n_commit'] = n+3                      # Number of commit\ncommits_df.loc[n, 'low'] = 0.1                           # Number of low_threshold\ncommits_df.loc[n, 'high'] = 0.88                         # Number of high_threshold\ncommits_df.loc[n, 'LB_score'] = 0.246                    # LB score after submitting\ncommits_df.loc[n, 'LB_private'] = 0.226                  # LB private","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Find and mark minimun value of LB score\ncommits_df['LB_score'] = pd.to_numeric(commits_df['LB_score'])\ncommits_df['LB_private'] = pd.to_numeric(commits_df['LB_private'])\ncommits_df = commits_df.sort_values(by=['LB_score'], ascending = False).reset_index(drop=True)\n\n# LB public\ncommits_df['max'] = 0\ncommits_df.loc[15, 'max'] = 1   # LB private maximum\ncommits_df.loc[0, 'max'] = 2   # LB public maximum\n\n# LB private\ncommits_df['max_private'] = 0\ncommits_df.loc[0, 'max_private'] = 1   # LB public maximum\ncommits_df.loc[15, 'max_private'] = 2   # LB private maximum","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# LB public\ncommits_df.sort_values(by=['LB_score'], ascending = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# LB private\ncommits_df.sort_values(by=['LB_private'], ascending = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Interactive plot with results of parameters tuning - LB public\nfig = px.scatter_3d(commits_df, x='low', y='high', z='LB_score', color='max',\n                    title='Parameters and LB score PUBLIC - visualization of solutions (start - yellow, private optimum - red)')\nfig.update(layout=dict(title=dict(x=0.1)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Interactive plot with results of parameters tuning - LB private\nfig = px.scatter_3d(commits_df, x='low', y='high', z='LB_private', color='max_private',\n                    title='Parameters and LB score PRIVATE - visualization of solutions (start - red, private optimum - yellow)')\nfig.update(layout=dict(title=dict(x=0.1)))","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.003752,"end_time":"2021-03-10T12:24:06.067026","exception":false,"start_time":"2021-03-10T12:24:06.063274","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Apply 2class filter"},{"metadata":{"execution":{"iopub.execute_input":"2021-03-10T12:24:06.083599Z","iopub.status.busy":"2021-03-10T12:24:06.082655Z","iopub.status.idle":"2021-03-10T12:24:07.134898Z","shell.execute_reply":"2021-03-10T12:24:07.135344Z"},"papermill":{"duration":1.064509,"end_time":"2021-03-10T12:24:07.135516","exception":false,"start_time":"2021-03-10T12:24:06.071007","status":"completed"},"tags":[],"trusted":true},"cell_type":"code","source":"# Thanks to https://www.kaggle.com/muhammad4hmed/lets-overfit-together\n# Thanks to https://www.kaggle.com/mahmudds/vinbigdata-chest-x-ray-abnormalities-detection\n\nNORMAL = \"14 1 0 0 1 1\"\n\npred_det_df = pd.read_csv(\"../input/vinbigdatastack/submission_postprocessed.csv\")\nn_normal_before = len(pred_det_df.query(\"PredictionString == @NORMAL\"))\nmerged_df = pd.merge(pred_det_df, pred_2class, on=\"image_id\", how=\"left\")\n\n\nif \"target\" in merged_df.columns:\n    merged_df[\"class0\"] = 1 - merged_df[\"target\"]\n\nc0, c1, c2 = 0, 0, 0\nfor i in range(len(merged_df)):\n    p0 = merged_df.loc[i, \"class0\"]\n    if p0 < low_threshold:\n\n        c0 += 1\n    elif low_threshold <= p0 and p0 < high_threshold:\n\n        merged_df.loc[i, \"PredictionString\"] += f\" 14 {p0} 0 0 1 1\"\n        c1 += 1\n    else:\n\n        merged_df.loc[i, \"PredictionString\"] = NORMAL\n        c2 += 1\n\nn_normal_after = len(merged_df.query(\"PredictionString == @NORMAL\"))\nprint(\n    f\"n_normal: {n_normal_before} -> {n_normal_after} with threshold {low_threshold} & {high_threshold}\"\n)\nprint(f\"Keep {c0} Add {c1} Replace {c2}\")\nsubmission_filepath = str(\"submission.csv\")\nsubmission_df = merged_df[[\"image_id\", \"PredictionString\"]]\nsubmission_df.to_csv(submission_filepath, index=False)\nprint(f\"Saved to {submission_filepath}\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}