{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-02T15:27:51.779266Z","iopub.execute_input":"2021-09-02T15:27:51.779701Z","iopub.status.idle":"2021-09-02T15:28:48.502347Z","shell.execute_reply.started":"2021-09-02T15:27:51.779603Z","shell.execute_reply":"2021-09-02T15:28:48.500324Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # # LIBRARY","metadata":{}},{"cell_type":"code","source":"import os\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport pandas as pd\nimport time\nimport tqdm\nimport pickle\nimport pydicom\nimport cv2\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport tqdm\nfrom tqdm.notebook import tqdm\nimport numpy as np\nimport gc\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:29:20.926849Z","iopub.execute_input":"2021-09-02T15:29:20.927239Z","iopub.status.idle":"2021-09-02T15:29:23.854272Z","shell.execute_reply.started":"2021-09-02T15:29:20.927205Z","shell.execute_reply":"2021-09-02T15:29:23.853251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # # DIR","metadata":{}},{"cell_type":"code","source":"DATA_DIR = '../input/vinbigdata-chest-xray-abnormalities-detection'\n\ntrain_dir = os.path.join(DATA_DIR, 'train')\ntest_dir = os.path.join(DATA_DIR, 'test')\n\ntrain_dicom_path = [os.path.join(train_dir, f_name) for f_name in os.listdir(train_dir)]\ntest_dicom_path = [os.path.join(test_dir, f_name) for f_name in os.listdir(test_dir)]\n\nsb_dir = os.path.join(DATA_DIR, 'sample_submission.csv')\ntraincsv_dir = os.path.join(DATA_DIR, 'train.csv')\n\nsb_df = pd.read_csv(sb_dir)\ntrain_df = pd.read_csv(traincsv_dir)\n\nprint(f\"The number of training file is: {len(train_dicom_path)}\")\nprint(f\"The number of testing file is: {len(test_dicom_path)}\")\n\nprint(\"Train datafram\")\ndisplay(train_df.head(5))\n\nprint(\"Sample submission datafram\")\ndisplay(sb_df.head(5))\n","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:29:26.466871Z","iopub.execute_input":"2021-09-02T15:29:26.467273Z","iopub.status.idle":"2021-09-02T15:29:26.79689Z","shell.execute_reply.started":"2021-09-02T15:29:26.467237Z","shell.execute_reply":"2021-09-02T15:29:26.795889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # # HELPER FUNCTION","metadata":{}},{"cell_type":"code","source":"FIG_FONT = dict(family = \"Helvetica, Arial\", size = 14, color = \"#7f7f7f\")\nLABEL_COLORS = [px.colors.label_rgb(px.colors.convert_to_RGB_255(x)) for x in sns.color_palette(\"Spectral\", 15)]\nLABEL_COLORS_WOUT_NO_FINDING = LABEL_COLORS[:8]+LABEL_COLORS[9:]","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:29:31.195547Z","iopub.execute_input":"2021-09-02T15:29:31.195893Z","iopub.status.idle":"2021-09-02T15:29:31.20264Z","shell.execute_reply.started":"2021-09-02T15:29:31.195865Z","shell.execute_reply":"2021-09-02T15:29:31.201895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom2array(path, voi_lut = True, fix_monochrom = True):\n    dicom = pydicom.read_file(path)\n\n# VOI LUT (nếu có trên thiết bị DICOM) được sử dụng để \n# chuyển đổi dữ liệu DICOM thô sang chế độ xem \"thân thiện với con người\"\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # Neu anh trong bi nguoc\n    if fix_monochrom and dicom.PhotometricInterpretation == 'MONOCHROM1':\n        data = np.amax(data) - data\n    # Chuan hoa mang hinh anh va trả vè\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\n# Hien thi hinh anh   \ndef plot_image(img, title = '', figsize = (8,8), cmap = None):\n    plt.figue(figsize = figsize)\n    if cmap:\n        plt.imhshow(img, cmap = cmap)\n    else:\n        plt.imshow(img)\n    plt.title(title, fontweight = ' bold')\n    plt.axis(False)\n    plt.show()\n    \n# Hien thi id hinh anh\ndef get_image_id(path):\n    return path.rsplit(\"/\", 1)[1].rsplit(\".\", 1)[0]\n\n# Tạo ra toa do hinh vuong\ndef create_fractional_bbox_coordinates(row):\n    frac_x_min = row['x_min']/row['img_width']\n    frac_x_max = row['x_max']/row['img_width']\n    frac_y_min = row['y_min']/row['img_height']\n    frac_y_max = row['y_max']/row['img_height']\n    return frac_x_min, frac_x_max, frac_y_min, frac_y_max\n\n# Ve hinh vuong \ndef draw_bboxes(img, tl, br, rgb, label = '', label_location = 'tl', opacity = 0.85, line_thickness = 0 ):\n    rect = np.uint8(np.ones((br[1]-tl[1], br[0]-tl[0],3))*rgb)\n    sub_combo = cv2.addWeighted(img[tl[1]:br[1], tl[0]:br[0], :], 1-opacity, rect, opacity, 1.0)\n    img[tl[1]:br[1], tl[0]:br[0], :] = sub_combo\n    \n    if line_thickness > 0:\n        img = cv2.rectangle(img, tuple(tl), tuple(br), rgb, line_thickness)\n    \n    if label:\n        FONT = cv2.FONT_HERSHEY_SIMPLEX\n        FONT_SCALE = 1.666\n        FONT_THICKNESS = 3\n        FONT_LINE_STYLE = cv2.LINE_AA\n    if type(label) == str:\n        LABEL = label.upper().replace(\"\", \"\")\n    else:\n        LABEL = f\"CLASS_{label: 02}\"\n        \n    text_width, text_height = cv2.getTextSize(LABEL, FONT, FONT_SCALE, FONT_THICKNESS)[0]\n    label_original = {\"tl\":tl, \"br\":br, \"tr\":(br[0], tl[1]), \"bl\":(br[1], tl[0])}[label_location]\n    label_offset = {\"tl\":np.array([0, -10]), \"br\":np.array([-text_width, text_height+10]),\n                   \"tr\":np.array([-text_width, -10]), \"bl\":np.array([0, text_height+10])}[label_location]\n    img = cv2.putText(img, LABEL, tuple(label_original+label_offset),\n                     FONT, FONT_SCALE, rgb, FONT_THICKNESS, FONT_LINE_STYLE)\n    return img","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:29:33.436967Z","iopub.execute_input":"2021-09-02T15:29:33.437496Z","iopub.status.idle":"2021-09-02T15:29:33.456657Z","shell.execute_reply.started":"2021-09-02T15:29:33.437463Z","shell.execute_reply":"2021-09-02T15:29:33.455877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# # # DỮ LIỆU DẠNG BẢNG","metadata":{}},{"cell_type":"markdown","source":"# TỔNG SỐ THÔNG BÁO ĐỐI TƯỢNG MỖI HÌNH ẢNH","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(train_df.image_id.value_counts(),\n                  log_y =True, color_discrete_sequence = ['indianred'], opacity = 0.85,\n                  labels = {\"value\":\"Number of Annotations per image\"}, \n                  title = \"<b>DISTRIBUTION OF ANNOTATION PER IMAGE</b>\")\nfig.update_layout(showlegend = False,\n                  xaxis_title = '<b> Number of unique image</b>',\n                  yaxis_title = '<b> Count of all object annotation</b>',\n                  font = FIG_FONT)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:29:37.393679Z","iopub.execute_input":"2021-09-02T15:29:37.39421Z","iopub.status.idle":"2021-09-02T15:29:38.776388Z","shell.execute_reply.started":"2021-09-02T15:29:37.394159Z","shell.execute_reply":"2021-09-02T15:29:38.775332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# THÔNG BÁO ĐỐI TƯỢNG ĐỘC ĐÁO MỖI HÌNH ẢNH","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(train_df.groupby('image_id')[\"class_id\"].unique().apply(lambda x: len(x)),\n                  log_y = True, color_discrete_sequence = ['green'], opacity = 0.85,\n                  labels = {\"value\": \"Number of unique abnormalities\"},\n                  title = \"<b>DISTIBUTION OF ANNOTATIONS PER PATIENT\")\nfig.update_layout(showlegend = False,\n                 xaxis_title = \"<b> Number of unique abnormalities</b>\",\n                 yaxis_title = \"<b> Count of unique patient</b>\", font = FIG_FONT)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:29:42.863045Z","iopub.execute_input":"2021-09-02T15:29:42.863427Z","iopub.status.idle":"2021-09-02T15:29:44.266978Z","shell.execute_reply.started":"2021-09-02T15:29:42.863397Z","shell.execute_reply":"2021-09-02T15:29:44.265745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KHAI THÁC CỘT CLASS_NAME","metadata":{}},{"cell_type":"code","source":"fig = px.bar(train_df.class_name.value_counts().sort_index(),\n             color = train_df.class_name.value_counts().sort_index().index, opacity = 0.85,\n             labels = {\"y\":\"Annotation per class\", \"x\":\"\"},\n             title = \"<b>Annotation per class</b>\")\nfig.update_layout(showlegend = False,\n                 xaxis_title = \"\",\n                 yaxis_title = \"<b> Annotation per class\", font = FIG_FONT)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:29:48.106544Z","iopub.execute_input":"2021-09-02T15:29:48.10695Z","iopub.status.idle":"2021-09-02T15:29:48.300842Z","shell.execute_reply.started":"2021-09-02T15:29:48.106915Z","shell.execute_reply":"2021-09-02T15:29:48.299937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KHAI THÁC CỘT CLASS_ID","metadata":{}},{"cell_type":"code","source":"int_2_str = {i:train_df[train_df[\"class_id\"]==i].iloc[0][\"class_name\"] for i in range(15)}\nstr_2_int = {v:k for k,v in int_2_str.items()}\nint_2_clr = {str_2_int[k]:LABEL_COLORS[i] for i,k in enumerate(sorted(str_2_int.keys()))}\nprint(\"\\n... Dictionary Mapping Class Integer to Class String Representation [int_2_str]...\\n\")\ndisplay(int_2_str)\n\nprint(\"\\n... Dictionary Mapping Class String to Class Integer Representation [str_2_int]...\\n\")\ndisplay(str_2_int)\n\nprint(\"\\n... Dictionary Mapping Class Integer to Color Representation [str_2_clr]...\\n\")\ndisplay(int_2_clr)\n\nprint(\"\\n... Head of Train Dataframe After Dropping The Class Name Column...\\n\")\n\ntrain_df.drop(columns=[\"class_name\"], inplace=True)\ndisplay(train_df.head(5))","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:29:54.154532Z","iopub.execute_input":"2021-09-02T15:29:54.154901Z","iopub.status.idle":"2021-09-02T15:29:54.231565Z","shell.execute_reply.started":"2021-09-02T15:29:54.15487Z","shell.execute_reply":"2021-09-02T15:29:54.23088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KHAI THÁC CỘT RAD_ID","metadata":{}},{"cell_type":"code","source":"fig = px.histogram(train_df, x = \"rad_id\", color = \"rad_id\", opacity = 0.85,\n                  labels = {\"rad_id\":\"Radiologist ID\"}, \n                  title = \"<b>DISTRIBUTION OF # OF ANNOTATIONS PER RADIOLOGIST</b>\").update_xaxes(categoryorder = 'total descending')\nfig.update_layout(showlegend = False,\n                 legend_title = \"<b> RADIOLOGIST ID</b>\",\n                 xaxis_title = \"Radiologist ID\",\n                 yaxis_title = \"<b>Number of annotations made</b>\", font = FIG_FONT)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:29:58.91005Z","iopub.execute_input":"2021-09-02T15:29:58.910817Z","iopub.status.idle":"2021-09-02T15:29:59.835563Z","shell.execute_reply.started":"2021-09-02T15:29:58.910771Z","shell.execute_reply":"2021-09-02T15:29:59.834624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SỐ NHÃN ĐÁNH DẤU CỦA MỖI BÁC SĨ","metadata":{}},{"cell_type":"code","source":"fig = go.Figure()\n\nfor i in range(15):\n    fig.add_trace(go.Histogram(x = train_df[train_df[\"class_id\"]==i][\"rad_id\"], marker_color = int_2_clr[i],\n                                name = f\"<b>{int_2_str[i]}</b>\")).update_xaxes(categoryorder=\"total descending\")\nfig.update_layout(title = \"<b>DISTRIBUTION OF CLASS LABEL ANNOTATIONS BY RADIOLOGIST</b>\",\n                 barmode = 'stack',\n                 xaxis_title = \"<b> Radiologist ID </b>\",\n                 yaxis_title = \"<b>Number of annotations made</b>\", font = FIG_FONT)\nfig.show()\n\nfig = go.Figure()\n\nfor i in range(15):\n    fig.add_trace(go.Histogram(x=train_df[(train_df[\"class_id\"]==i) & (~train_df[\"rad_id\"].isin([\"R8\",\"R9\",\"R10\"]))][\"rad_id\"], \n                               marker_color = int_2_clr[i],\n                               name = f\"<b>{int_2_str[i]}</b>\")).update_xaxes(categoryorder=\"total descending\")\nfig.update_layout(title = \"<b>DISTRIBUTION OF CLASS LABEL ANNOTATIONS BY RADIOLOGIST</b>\",\n                 barmode = 'stack',\n                 xaxis_title = \"<b> Radiologist ID </b>\",\n                 yaxis_title = \"<b>Number of annotations made</b>\", font = FIG_FONT)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:30:03.943174Z","iopub.execute_input":"2021-09-02T15:30:03.943569Z","iopub.status.idle":"2021-09-02T15:30:04.913169Z","shell.execute_reply.started":"2021-09-02T15:30:03.943527Z","shell.execute_reply":"2021-09-02T15:30:04.912161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KHAI THÁC CÁC CỘT PHỐI HỢP CÁC BBOX","metadata":{}},{"cell_type":"code","source":"bbox_df = train_df[train_df.class_id != 14].reset_index(drop=True)\nBBOX_PATH = [os.path.join(train_dir, name + \".dicom\") for name in bbox_df.image_id.unique()]\n\nsizes_of_images_w_bboxes = {}\n\nfor path in tqdm(BBOX_PATH, total = len(BBOX_PATH)):\n    dicom = pydicom.read_file(path)\n    sizes_of_images_w_bboxes[path[:-6].rsplit(\"/\", 1)[1]] = (dicom.Rows, dicom.Columns)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:30:12.835549Z","iopub.execute_input":"2021-09-02T15:30:12.836194Z","iopub.status.idle":"2021-09-02T15:43:24.223804Z","shell.execute_reply.started":"2021-09-02T15:30:12.836127Z","shell.execute_reply":"2021-09-02T15:43:24.222432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bbox_df[\"img_height\"] = bbox_df[\"image_id\"].map(lambda x: sizes_of_images_w_bboxes[x][0])\nbbox_df[\"img_width\"] = bbox_df[\"image_id\"].map(lambda x: sizes_of_images_w_bboxes[x][1])\n\nbbox_df[\"frac_x_min\"], bbox_df[\"frac_x_max\"], bbox_df[\"frac_y_min\"], bbox_df[\"frac_y_max\"] = zip(*bbox_df.apply(create_fractional_bbox_coordinates, axis=1))\n\nave_src_img_height = np.mean([size[0] for size in sizes_of_images_w_bboxes.values()], dtype = np.uint32)\nave_src_img_width = np.mean([size[1] for size in sizes_of_images_w_bboxes.values()], dtype = np.uint32)\n\nbbox_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:43:26.945788Z","iopub.execute_input":"2021-09-02T15:43:26.946605Z","iopub.status.idle":"2021-09-02T15:43:29.242153Z","shell.execute_reply.started":"2021-09-02T15:43:26.94654Z","shell.execute_reply":"2021-09-02T15:43:29.24106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# HEAT MAP","metadata":{}},{"cell_type":"code","source":"HEATMAP_SIZE = (ave_src_img_width, ave_src_img_height, 14)\n\nheatmap = np.zeros((HEATMAP_SIZE), dtype = np.int16)\n\nbbox_np = bbox_df[[\"class_id\", \"frac_x_min\", \"frac_x_max\", \"frac_y_min\", \"frac_y_max\"]].to_numpy()\nbbox_np[:, 1:3] *= ave_src_img_width\nbbox_np[:, 3:5] *= ave_src_img_height\nbbox_np = np.floor(bbox_np).astype(np.int16)\ncustom_cmaps = [matplotlib.colors.LinearSegmentedColormap.from_list(colors=[(0.,0.,0.), c, (0.95,0.95,0.95)], \n        name=f\"custom_{i}\") for i,c in enumerate(sns.color_palette(\"Spectral\", 15))]\ncustom_cmaps.pop(8)\n\nfor row in tqdm(bbox_np, total=bbox_np.shape[0]):\n    heatmap[row[3]:row[4]+1, row[1]:row[2]+1, row[0]] += 1\n    \nfig = plt.figure(figsize=(20,25))\nplt.suptitle(\"Heatmaps Showing Bounding Box Placement\\n \", fontweight=\"bold\", fontsize=16)\nfor i in range(15):\n    plt.subplot(4, 4, i+1)\n    if i==0:\n        plt.imshow(heatmap.mean(axis=-1), cmap=\"bone\")\n        plt.title(f\"Average of All Classes\", fontweight=\"bold\")\n    else:\n        plt.imshow(heatmap[:, :, i-1], cmap=custom_cmaps[i-1])\n        plt.title(f\"{int_2_str[i-1]} – ({i})\", fontweight=\"bold\")\n        \n    plt.axis(False)\nfig.tight_layout(rect=[0, 0.03, 1, 0.97])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:43:32.546915Z","iopub.execute_input":"2021-09-02T15:43:32.547282Z","iopub.status.idle":"2021-09-02T15:44:16.718696Z","shell.execute_reply.started":"2021-09-02T15:43:32.547252Z","shell.execute_reply":"2021-09-02T15:44:16.717666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ĐIỀU TRA KÍCH THƯỚC CỦA HỘP TRÒN VÀ TÁC ĐỘNG CỦA LỚP","metadata":{}},{"cell_type":"code","source":"bbox_df[\"frac_bbox_area\"] = (bbox_df[\"frac_x_max\"]-bbox_df[\"frac_x_min\"])*(bbox_df[\"frac_y_max\"]-bbox_df[\"frac_y_min\"])\nbbox_df[\"class_id_as_str\"] = bbox_df[\"class_id\"].map(int_2_str)\ndisplay(bbox_df.head(5))\n\nfig = px.box(bbox_df.sort_values(by = \"class_id_as_str\"), x=\"class_id_as_str\", y=\"frac_bbox_area\",\n            color = \"class_id_as_str\", color_discrete_sequence = LABEL_COLORS_WOUT_NO_FINDING, notched = True,\n            labels = {\"x\": \"Class name\", \"y\": \"BBox Area %\"}, title = \"<b>DISTRIBUTION OF BBOX AREAS AS % OF SOURCE IMAGE AREA\")\nfig.update_layout(showlegend = True, xaxis_title = \"\", yaxis_title = \"Bounding Box Area %\",\n                 yaxis_range = [-0.025, 0.4], font = FIG_FONT)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:44:38.890754Z","iopub.execute_input":"2021-09-02T15:44:38.891126Z","iopub.status.idle":"2021-09-02T15:44:39.722016Z","shell.execute_reply.started":"2021-09-02T15:44:38.891093Z","shell.execute_reply":"2021-09-02T15:44:39.721095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ĐIỀU TRA TỶ LỆ TỶ SỐ HOÀN HẢO CỦA CÁC HỘP TRÒN VÀ TÁC ĐỘNG CỦA LỚP","metadata":{}},{"cell_type":"code","source":"bbox_df[\"aspect_ratio\"] = (bbox_df[\"frac_x_max\"]-bbox_df[\"frac_x_min\"]) / (bbox_df[\"frac_y_max\"]-bbox_df[\"frac_y_min\"])\ndisplay(bbox_df.groupby(\"class_id\").mean())\n\nfig = px.bar(x = [int_2_str[x] for x in range(14)], y = bbox_df.groupby(\"class_id\").mean()[\"aspect_ratio\"],\n            color = [int_2_str[x] for x in range(14)], color_discrete_sequence = LABEL_COLORS_WOUT_NO_FINDING,\n            labels = {\"x\":\"Class_name\", \"y\":\"Aspect Ratio\"}, title = \"<b> Aspect ratios for bounding boxes by class</b>\")\nfig.update_layout(showlegend = True, font = FIG_FONT,\n                 xaxis_title = \"\", yaxis_title = \"<b> Aspect ratio (W/H)</b>\", legend_title_text = None)\nfig.add_hline(y = 1, line_width = 2, line_dash = \"dot\", annotation_font_size = 10, annotation_text = \"<b> Square aspect ratio</b>\",\n             annotation_position = \"bottom left\", annotation_font_color = \"black\")\nfig.add_hrect(y0= 0, y1= 0.5, line_width = 0, fillcolor = \"red\", opacity = 0.2, annotation_text = \"<b> Vertical rectangle region</b>\",\n             annotation_position = \"bottom right\", annotation_font_size = 20, annotation_font_color = \"black\")\nfig.add_hrect(y0= 2, y1= 3.5, line_width = 0, fillcolor = \"blue\", opacity = 0.2, annotation_text = \"<b> Horizontal rectangle region</b>\",\n             annotation_position = \"top right\", annotation_font_size = 20, annotation_font_color = \"black\")\n\n","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:45:00.434467Z","iopub.execute_input":"2021-09-02T15:45:00.434841Z","iopub.status.idle":"2021-09-02T15:45:00.671498Z","shell.execute_reply.started":"2021-09-02T15:45:00.434809Z","shell.execute_reply":"2021-09-02T15:45:00.670457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  IMAGE DATA","metadata":{}},{"cell_type":"markdown","source":"LỚP ĐỂ GIÚP KHAI THÁC DỮ LIỆU","metadata":{}},{"cell_type":"code","source":"class TrainData():\n    def __init__(self, df, train_dir, cmap=\"Spectral\"):\n        # Initialize\n        self.df = df\n        self.train_dir = train_dir\n        \n        # Visualization\n        self.cmap = cmap\n        self.pal = [tuple([int(x) for x in np.array(c)*(255,255,255)]) for c in sns.color_palette(cmap, 15)]\n        self.pal.pop(8)\n        \n        # Store df components in individual numpy arrays for easy access based on index\n        tmp_numpy = self.df.to_numpy()\n        image_ids = tmp_numpy[0]\n        class_ids = tmp_numpy[1]\n        rad_ids = tmp_numpy[2]\n        bboxes = tmp_numpy[3:]\n        \n        self.img_annotations = self.get_annotations(get_all=True)\n        \n        # Clean-Up\n        del tmp_numpy; gc.collect();\n        \n        \n    def get_annotations(self, get_all=False, image_ids=None, class_ids=None, rad_ids=None, index=None):\n        \"\"\" TBD \n        \n        Args:\n            get_all (bool, optional): TBD\n            image_ids (list of strs, optional): TBD\n            class_ids (list of ints, optional): TBD\n            rad_ids (list of strs, optional): TBD\n            index (int, optional):\n        \n        Returns:\n        \n        \n        \"\"\"\n        if not get_all and image_ids is None and class_ids is None and rad_ids is None and index is None:\n            raise ValueError(\"Expected one of the following arguments to be passed:\" \\\n                             \"\\n\\t\\t– `get_all`, `image_id`, `class_id`, `rad_id`, or `index`\")\n        # Initialize\n        tmp_df = self.df.copy()\n        \n        if not get_all:\n            if image_ids is not None:\n                tmp_df = tmp_df[tmp_df.image_id.isin(image_ids)]\n            if class_ids is not None:\n                tmp_df = tmp_df[tmp_df.class_id.isin(class_ids)]\n            if rad_ids is not None:\n                tmp_df = tmp_df[tmp_df.rad_id.isin(rad_ids)]\n            if index is not None:\n                tmp_df = tmp_df.iloc[index]\n            \n        annotations = {image_id:[] for image_id in tmp_df.image_id.to_list()}\n        for row in tmp_df.to_numpy():\n            \n            # Update annotations dictionary\n            annotations[row[0]].append(dict(\n                img_path=os.path.join(self.train_dir, row[0]+\".dicom\"),\n                image_id=row[0],\n                class_id=int(row[1]),\n                rad_id=int(row[2][1:]),\n            ))\n            \n            # Catch to convert float array to integer array\n            if row[1]==14:\n                annotations[row[0]][-1][\"bbox\"]=row[3:]\n            else:\n                annotations[row[0]][-1][\"bbox\"]=row[3:].astype(np.int32)\n        return annotations\n    \n    def get_annotated_image(self, image_id, annots=None, plot=False, plot_size=(18,25), plot_title=\"\"):\n        if annots is None:\n            annots = self.img_annotations.copy()\n        \n        if type(annots) != list:\n            image_annots = annots[image_id]\n        else:\n            image_annots = annots\n            \n        img = cv2.cvtColor(dicom2array(image_annots[0][\"img_path\"]),cv2.COLOR_GRAY2RGB)\n        for ann in image_annots:\n            if ann[\"class_id\"] != 14:\n                img = draw_bboxes(img, \n                                ann[\"bbox\"][:2], ann[\"bbox\"][-2:], \n                                rgb=self.pal[ann[\"class_id\"]], \n                                label=int_2_str[ann[\"class_id\"]], \n                                opacity=0.08, line_thickness=4)\n        if plot:\n            plot_image(img, title=plot_title, figsize=plot_size)\n        \n        return img\n    \n    def plot_image_ids(self, image_id_list, height_multiplier=6, verbose=True):\n        annotations = self.get_annotations(image_ids=image_id_list)\n        annotated_imgs = []\n        n = len(image_id_list)\n        \n        plt.figure(figsize=(20, height_multiplier*n))\n        for i, (image_id, annots) in enumerate(annotations.items()):\n            if i >= n:\n                break\n            if verbose:\n                print(f\".\", end=\"\")\n            plt.subplot(n//2,2,i+1)\n            plt.imshow(self.get_annotated_image(image_id, annots))\n            plt.axis(False)\n            plt.title(f\"Image ID – {image_id}\")\n        plt.tight_layout(rect=[0, 0.03, 1, 0.97])\n        plt.show()\n        \n    def plot_classes(self, class_list, n=4, height_multiplier=6, verbose=True):\n        annotations = self.get_annotations(class_ids=class_list)\n        annotated_imgs = []\n\n        plt.figure(figsize=(20, height_multiplier*n))\n        for i, (image_id, annots) in enumerate(annotations.items()):\n            if i >= n:\n                break\n            if verbose:\n                print(f\".\", end=\"\")\n            plt.subplot(n//2,2,i+1)\n            plt.imshow(self.get_annotated_image(image_id, annots))\n            plt.axis(False)\n            plt.title(f\"Image ID – {image_id}\")\n        plt.tight_layout(rect=[0, 0.03, 1, 0.97])\n        plt.show()\n\n    def plot_radiologists(self, rad_id_list, n=4, height_multiplier=6, verbose=True):\n        annotations = self.get_annotations(rad_ids=rad_id_list)\n        annotated_imgs = []\n\n        plt.figure(figsize=(20, height_multiplier*n))\n        for i, (image_id, annots) in enumerate(annotations.items()):\n            if i >= n:\n                break\n            if verbose:\n                print(f\".\", end=\"\")\n            plt.subplot(n//2,2,i+1)\n            plt.imshow(self.get_annotated_image(image_id, annots))\n            plt.axis(False)\n            plt.title(f\"Image ID – {image_id}\")\n        plt.tight_layout(rect=[0, 0.03, 1, 0.97])\n        plt.show()\n\ntrain_data = TrainData(train_df, train_dir)\n        \n        \n        ","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:45:06.869998Z","iopub.execute_input":"2021-09-02T15:45:06.870373Z","iopub.status.idle":"2021-09-02T15:45:07.63253Z","shell.execute_reply.started":"2021-09-02T15:45:06.870342Z","shell.execute_reply":"2021-09-02T15:45:07.631549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XEM HÌNH ẢNH TỪ ID ẢNH KHẮC PHỤC","metadata":{}},{"cell_type":"code","source":"IMG_ID_LIST = train_df[train_df.class_id !=14].image_id[25:29].to_list()\ntrain_data.plot_image_ids(image_id_list = IMG_ID_LIST, verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:45:17.657405Z","iopub.execute_input":"2021-09-02T15:45:17.657965Z","iopub.status.idle":"2021-09-02T15:45:26.091498Z","shell.execute_reply.started":"2021-09-02T15:45:17.657913Z","shell.execute_reply":"2021-09-02T15:45:26.090321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# HÌNH ẢNH CHỨA 1 LỚP","metadata":{}},{"cell_type":"code","source":"train_data.plot_classes(class_list=[0,], n=4, verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:23:22.615533Z","iopub.execute_input":"2021-09-02T15:23:22.615862Z","iopub.status.idle":"2021-09-02T15:23:30.01909Z","shell.execute_reply.started":"2021-09-02T15:23:22.615825Z","shell.execute_reply":"2021-09-02T15:23:30.018074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[1,], n=4, verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:23:45.50958Z","iopub.execute_input":"2021-09-02T15:23:45.509948Z","iopub.status.idle":"2021-09-02T15:23:53.16939Z","shell.execute_reply.started":"2021-09-02T15:23:45.50991Z","shell.execute_reply":"2021-09-02T15:23:53.163008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[2,], n=4, verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:24:15.361667Z","iopub.execute_input":"2021-09-02T15:24:15.362024Z","iopub.status.idle":"2021-09-02T15:24:24.259018Z","shell.execute_reply.started":"2021-09-02T15:24:15.361988Z","shell.execute_reply":"2021-09-02T15:24:24.258047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[3,], n=4, verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:24:31.150077Z","iopub.execute_input":"2021-09-02T15:24:31.150402Z","iopub.status.idle":"2021-09-02T15:24:37.11677Z","shell.execute_reply.started":"2021-09-02T15:24:31.150372Z","shell.execute_reply":"2021-09-02T15:24:37.1158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[4,], n=4, verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:24:44.200338Z","iopub.execute_input":"2021-09-02T15:24:44.200682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[5,], n=4, verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:01:55.806372Z","iopub.execute_input":"2021-09-02T15:01:55.806907Z","iopub.status.idle":"2021-09-02T15:02:04.533279Z","shell.execute_reply.started":"2021-09-02T15:01:55.806869Z","shell.execute_reply":"2021-09-02T15:02:04.532127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[6,], n=4, verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:02:28.857431Z","iopub.execute_input":"2021-09-02T15:02:28.857955Z","iopub.status.idle":"2021-09-02T15:02:36.733078Z","shell.execute_reply.started":"2021-09-02T15:02:28.857922Z","shell.execute_reply":"2021-09-02T15:02:36.732293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[7,], n=4, verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:02:54.763659Z","iopub.execute_input":"2021-09-02T15:02:54.764194Z","iopub.status.idle":"2021-09-02T15:03:01.435163Z","shell.execute_reply.started":"2021-09-02T15:02:54.764145Z","shell.execute_reply":"2021-09-02T15:03:01.433317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[8,], n=4, verbose = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[9,], n=4, verbose = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[10,], n=4, verbose = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[11,], n=4, verbose = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[12,], n=4, verbose = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[13,], n=4, verbose = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.plot_classes(class_list=[14,], n=4, verbose = False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KẾT HỢP / NHẬP CÁC THÔNG BÁO LỚP LÊN LỚP (WIP)","metadata":{}},{"cell_type":"code","source":"def calc_iou(bbox_1, bbox_2):\n    # determine the coordinates of the intersection rectangle\n    x_left = max(bbox_1[0], bbox_2[0])\n    y_top = max(bbox_1[1], bbox_2[1])\n    x_right = min(bbox_1[2], bbox_2[2])\n    y_bottom = min(bbox_1[3], bbox_2[3])\n\n    # Check if bboxes overlap at all (if not return 0)\n    if x_right < x_left or y_bottom < y_top:\n        return 0.0\n    \n    # The intersection of two axis-aligned bounding boxes is always an\n    # axis-aligned bounding box\n    else:\n        intersection_area = (x_right - x_left) * (y_bottom - y_top)\n        \n        # compute the area of both AABBs\n        bbox_1_area = (bbox_1[2] - bbox_1[0]) * (bbox_1[3] - bbox_1[1])\n        bbox_2_area = (bbox_2[2] - bbox_2[0]) * (bbox_2[3] - bbox_2[1])\n\n        # compute the intersection over union by taking the intersection\n        # area and dividing it by the sum of prediction + ground-truth\n        # areas - the interesection area\n        iou = intersection_area / float(bbox_1_area + bbox_2_area - intersection_area)\n        return iou\n\ndef redux_bboxes(annots):\n    def _get_inner_box(bboxes):\n        xmin = max([box[0] for box in bboxes])\n        ymin = max([box[1] for box in bboxes])\n        xmax = min([box[2] for box in bboxes])\n        ymax = min([box[3] for box in bboxes])\n        if (xmax<=xmin) or (ymax<=ymin):\n            return None\n        else:\n            return [xmin, ymin, xmax, ymax]\n        \n    valid_list_indices = [] \n    new_bboxes = []\n    new_class_ids = []\n    new_rad_ids = []\n    \n    for i, (class_id, rad_id, bbox) in enumerate(zip(annots[\"class_id\"], annots[\"rad_id\"], annots[\"bbox\"])):\n        intersecting_boxes = [bbox,]\n        other_bboxes = [x for j,x in enumerate(annots[\"bbox\"]) if j!=i]\n        other_classes = [x for j,x in enumerate(annots[\"class_id\"]) if j!=i]\n        for j, (other_class_id, other_bbox) in enumerate(zip(other_classes, other_bboxes)):\n            if class_id==other_class_id:\n                iou = calc_iou(bbox, other_bbox)\n                if iou>0.:\n                    intersecting_boxes.append(other_bbox)\n\n        if len(intersecting_boxes)>1:\n            inner_box = _get_inner_box(intersecting_boxes)\n            if inner_box and inner_box not in new_bboxes:\n                new_bboxes.append(inner_box)\n                new_class_ids.append(class_id)\n                new_rad_ids.append(rad_id) \n\n    annots[\"bbox\"] = new_bboxes\n    annots[\"rad_id\"] = new_rad_ids\n    annots[\"class_id\"] = new_class_ids\n    \n    return annots\n\n# Make GT Dataframe\ngt_df = train_df[train_df.class_id!=14]\n\n# Apply Manipulations and Merger Functions\ngt_df[\"bbox\"] = gt_df.loc[:, [\"x_min\",\"y_min\",\"x_max\",\"y_max\"]].values.tolist()\ngt_df.drop(columns=[\"x_min\",\"y_min\",\"x_max\",\"y_max\"], inplace=True)\ngt_df = gt_df.groupby([\"image_id\"]).agg({k:list for k in gt_df.columns if k !=\"image_id\"}).reset_index()\ngt_df = gt_df.apply(redux_bboxes, axis=1)\n\n# Recreate the Original Dataframe Style\ngt_df = gt_df.apply(pd.Series.explode).reset_index(drop=True).dropna()\ngt_df[\"x_min\"] = gt_df[\"bbox\"].apply(lambda x: x[0])\ngt_df[\"y_min\"] = gt_df[\"bbox\"].apply(lambda x: x[1])\ngt_df[\"x_max\"] = gt_df[\"bbox\"].apply(lambda x: x[2])\ngt_df[\"y_max\"] = gt_df[\"bbox\"].apply(lambda x: x[3])\ngt_df.drop(columns=[\"bbox\"], inplace=True)\n\n# Add back in NaN Rows As A Single Annotation\ngt_df = pd.concat([\n    gt_df, train_df.loc[train_df['class_id'] == 14].drop_duplicates(subset=[\"image_id\"])\n]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T15:45:34.433309Z","iopub.execute_input":"2021-09-02T15:45:34.433878Z","iopub.status.idle":"2021-09-02T15:45:36.660616Z","shell.execute_reply.started":"2021-09-02T15:45:34.43383Z","shell.execute_reply":"2021-09-02T15:45:36.659611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gt_data = TrainData(gt_df, TRAIN_DIR)\nIMAGE_ID_LIST = gt_df[gt_df.class_id!=14].groupby(\"image_id\") \\\n                                         .count() \\\n                                         .sort_values(by=\"class_id\", ascending=False) \\\n                                         .index[0:100:20]\n\nfor i, IMAGE_ID in enumerate(IMAGE_ID_LIST):\n    train_data.get_annotated_image(IMAGE_ID, annots=None, plot=True, plot_size=(18,22), plot_title=f\"ORIGINAL – IMG #{i+1}\")\n    gt_data.get_annotated_image(IMAGE_ID, annots=None, plot=True, plot_size=(18,22), plot_title=f\"REDUX VERSION – IMG #{i+1}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# HÌNH ẢNH CHỨA MỘT HOẶC NHIỀU LỚP TỪ MỘT BÁC SĨ","metadata":{}},{"cell_type":"code","source":"train_data.plot_radiologists(rad_id_list = [\"R8\"], verbose = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T14:56:38.09924Z","iopub.execute_input":"2021-09-02T14:56:38.099781Z","iopub.status.idle":"2021-09-02T14:56:46.196385Z","shell.execute_reply.started":"2021-09-02T14:56:38.099725Z","shell.execute_reply":"2021-09-02T14:56:46.195476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}