{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":9355809,"datasetId":5671733,"databundleVersionId":9554368},{"sourceType":"datasetVersion","sourceId":9358252,"datasetId":5673625,"databundleVersionId":9557032},{"sourceType":"datasetVersion","sourceId":9363154,"datasetId":5677381,"databundleVersionId":9562337},{"sourceType":"datasetVersion","sourceId":9365277,"datasetId":5679044,"databundleVersionId":9564643},{"sourceType":"datasetVersion","sourceId":9388364,"datasetId":5696638,"databundleVersionId":9589649},{"sourceType":"datasetVersion","sourceId":9395486,"datasetId":5702280,"databundleVersionId":9597289},{"sourceType":"datasetVersion","sourceId":9445869,"datasetId":5740774,"databundleVersionId":9652120},{"sourceType":"datasetVersion","sourceId":9447841,"datasetId":5742248,"databundleVersionId":9654266},{"sourceType":"datasetVersion","sourceId":1810938,"datasetId":1075803,"databundleVersionId":1848422}],"dockerImageVersionId":30683,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\ndDDdEdde# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-19T07:24:27.470566Z","iopub.execute_input":"2024-09-19T07:24:27.471005Z","iopub.status.idle":"2024-09-19T07:24:28.599836Z","shell.execute_reply.started":"2024-09-19T07:24:27.470971Z","shell.execute_reply":"2024-09-19T07:24:28.598947Z"}}},{"cell_type":"code","source":"# Display generic output messages\n!pip install colorama\n\n# Library for visualizing bounding boxes\n!pip install bbox-visualizer\n\n# Install ONNX library, will be used to convert from pytorch model to a tf model\n!pip install onnx onnxruntime onnxsim onnx-tf","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:17:56.090748Z","iopub.execute_input":"2024-09-26T01:17:56.091045Z","iopub.status.idle":"2024-09-26T01:18:36.587069Z","shell.execute_reply.started":"2024-09-26T01:17:56.091018Z","shell.execute_reply":"2024-09-26T01:18:36.585969Z"},"trusted":true},"execution_count":1,"outputs":[{"name":"stdout","text":"Requirement already satisfied: colorama in /opt/conda/lib/python3.10/site-packages (0.4.6)\nCollecting bbox-visualizer\n  Downloading bbox_visualizer-0.1.0-py2.py3-none-any.whl.metadata (4.2 kB)\nDownloading bbox_visualizer-0.1.0-py2.py3-none-any.whl (6.2 kB)\nInstalling collected packages: bbox-visualizer\nSuccessfully installed bbox-visualizer-0.1.0\nRequirement already satisfied: onnx in /opt/conda/lib/python3.10/site-packages (1.16.0)\nCollecting onnxruntime\n  Downloading onnxruntime-1.19.2-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl.metadata (4.5 kB)\nCollecting onnxsim\n  Downloading onnxsim-0.4.36-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (4.3 kB)\nCollecting onnx-tf\n  Downloading onnx_tf-1.10.0-py3-none-any.whl.metadata (510 bytes)\nRequirement already satisfied: numpy>=1.20 in /opt/conda/lib/python3.10/site-packages (from onnx) (1.26.4)\nRequirement already satisfied: protobuf>=3.20.2 in /opt/conda/lib/python3.10/site-packages (from onnx) (3.20.3)\nCollecting coloredlogs (from onnxruntime)\n  Downloading coloredlogs-15.0.1-py2.py3-none-any.whl.metadata (12 kB)\nRequirement already satisfied: flatbuffers in /opt/conda/lib/python3.10/site-packages (from onnxruntime) (23.5.26)\nRequirement already satisfied: packaging in /opt/conda/lib/python3.10/site-packages (from onnxruntime) (21.3)\nRequirement already satisfied: sympy in /opt/conda/lib/python3.10/site-packages (from onnxruntime) (1.12)\nRequirement already satisfied: rich in /opt/conda/lib/python3.10/site-packages (from onnxsim) (13.7.0)\nRequirement already satisfied: PyYAML in /opt/conda/lib/python3.10/site-packages (from onnx-tf) (6.0.1)\nCollecting tensorflow-addons (from onnx-tf)\n  Downloading tensorflow_addons-0.23.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (1.8 kB)\nCollecting humanfriendly>=9.1 (from coloredlogs->onnxruntime)\n  Downloading humanfriendly-10.0-py2.py3-none-any.whl.metadata (9.2 kB)\nRequirement already satisfied: pyparsing!=3.0.5,>=2.0.2 in /opt/conda/lib/python3.10/site-packages (from packaging->onnxruntime) (3.1.1)\nRequirement already satisfied: markdown-it-py>=2.2.0 in /opt/conda/lib/python3.10/site-packages (from rich->onnxsim) (3.0.0)\nRequirement already satisfied: pygments<3.0.0,>=2.13.0 in /opt/conda/lib/python3.10/site-packages (from rich->onnxsim) (2.17.2)\nRequirement already satisfied: mpmath>=0.19 in /opt/conda/lib/python3.10/site-packages (from sympy->onnxruntime) (1.3.0)\nCollecting typeguard<3.0.0,>=2.7 (from tensorflow-addons->onnx-tf)\n  Downloading typeguard-2.13.3-py3-none-any.whl.metadata (3.6 kB)\nRequirement already satisfied: mdurl~=0.1 in /opt/conda/lib/python3.10/site-packages (from markdown-it-py>=2.2.0->rich->onnxsim) (0.1.2)\nDownloading onnxruntime-1.19.2-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl (13.2 MB)\n\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m13.2/13.2 MB\u001b[0m \u001b[31m81.7 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m:00:01\u001b[0m00:01\u001b[0m\n\u001b[?25hDownloading onnxsim-0.4.36-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl (2.3 MB)\n\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m2.3/2.3 MB\u001b[0m \u001b[31m63.0 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m:00:01\u001b[0m\n\u001b[?25hDownloading onnx_tf-1.10.0-py3-none-any.whl (226 kB)\n\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m226.1/226.1 kB\u001b[0m \u001b[31m13.2 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n\u001b[?25hDownloading coloredlogs-15.0.1-py2.py3-none-any.whl (46 kB)\n\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m46.0/46.0 kB\u001b[0m \u001b[31m2.8 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n\u001b[?25hDownloading tensorflow_addons-0.23.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl (611 kB)\n\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m611.8/611.8 kB\u001b[0m \u001b[31m32.5 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n\u001b[?25hDownloading humanfriendly-10.0-py2.py3-none-any.whl (86 kB)\n\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m86.8/86.8 kB\u001b[0m \u001b[31m6.5 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m\n\u001b[?25hDownloading typeguard-2.13.3-py3-none-any.whl (17 kB)\nInstalling collected packages: typeguard, humanfriendly, tensorflow-addons, coloredlogs, onnxsim, onnxruntime, onnx-tf\n  Attempting uninstall: typeguard\n    Found existing installation: typeguard 4.1.5\n    Uninstalling typeguard-4.1.5:\n      Successfully uninstalled typeguard-4.1.5\n\u001b[31mERROR: pip's dependency resolver does not currently take into account all the packages that are installed. This behaviour is the source of the following dependency conflicts.\nydata-profiling 4.6.4 requires numpy<1.26,>=1.16.0, but you have numpy 1.26.4 which is incompatible.\nydata-profiling 4.6.4 requires typeguard<5,>=4.1.2, but you have typeguard 2.13.3 which is incompatible.\u001b[0m\u001b[31m\n\u001b[0mSuccessfully installed coloredlogs-15.0.1 humanfriendly-10.0 onnx-tf-1.10.0 onnxruntime-1.19.2 onnxsim-0.4.36 tensorflow-addons-0.23.0 typeguard-2.13.3\n","output_type":"stream"}]},{"cell_type":"code","source":"import bbox_visualizer as bbv\nimport cv2\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport shutil, os\nimport tensorflow as tf\nimport yaml\n\nfrom colorama import Fore, Back, Style\nfrom IPython.display import Image, display, clear_output\nfrom sklearn.model_selection import GroupShuffleSplit \nfrom tqdm.notebook import tqdm\nfrom typing import List","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:18:36.588972Z","iopub.execute_input":"2024-09-26T01:18:36.589268Z","iopub.status.idle":"2024-09-26T01:18:50.479286Z","shell.execute_reply.started":"2024-09-26T01:18:36.589238Z","shell.execute_reply":"2024-09-26T01:18:50.478277Z"},"trusted":true},"execution_count":2,"outputs":[{"name":"stderr","text":"2024-09-26 01:18:40.427745: E external/local_xla/xla/stream_executor/cuda/cuda_dnn.cc:9261] Unable to register cuDNN factory: Attempting to register factory for plugin cuDNN when one has already been registered\n2024-09-26 01:18:40.427847: E external/local_xla/xla/stream_executor/cuda/cuda_fft.cc:607] Unable to register cuFFT factory: Attempting to register factory for plugin cuFFT when one has already been registered\n2024-09-26 01:18:40.562762: E external/local_xla/xla/stream_executor/cuda/cuda_blas.cc:1515] Unable to register cuBLAS factory: Attempting to register factory for plugin cuBLAS when one has already been registered\n","output_type":"stream"}]},{"cell_type":"code","source":"# !rm -rf /kaggle/working/data\n!mkdir -p /kaggle/working/data","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:18:50.480591Z","iopub.execute_input":"2024-09-26T01:18:50.481261Z","iopub.status.idle":"2024-09-26T01:18:51.496884Z","shell.execute_reply.started":"2024-09-26T01:18:50.481227Z","shell.execute_reply":"2024-09-26T01:18:51.495452Z"},"trusted":true},"execution_count":3,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/* /kaggle/working/data/","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:18:51.499643Z","iopub.execute_input":"2024-09-26T01:18:51.500008Z","iopub.status.idle":"2024-09-26T01:23:11.160602Z","shell.execute_reply.started":"2024-09-26T01:18:51.499976Z","shell.execute_reply":"2024-09-26T01:23:11.159315Z"},"trusted":true},"execution_count":4,"outputs":[]},{"cell_type":"code","source":"!cp /kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv /kaggle/working/data/","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:11.162042Z","iopub.execute_input":"2024-09-26T01:23:11.16236Z","iopub.status.idle":"2024-09-26T01:23:12.239601Z","shell.execute_reply.started":"2024-09-26T01:23:11.16233Z","shell.execute_reply":"2024-09-26T01:23:12.238114Z"},"trusted":true},"execution_count":5,"outputs":[]},{"cell_type":"code","source":"!cp /kaggle/input/lables/output.csv /kaggle/working/data/","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:12.241347Z","iopub.execute_input":"2024-09-26T01:23:12.242184Z","iopub.status.idle":"2024-09-26T01:23:13.292498Z","shell.execute_reply.started":"2024-09-26T01:23:12.242142Z","shell.execute_reply":"2024-09-26T01:23:13.291233Z"},"trusted":true},"execution_count":6,"outputs":[]},{"cell_type":"code","source":"# !cp /kaggle/input/lan2pt/lan2.pt /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.294494Z","iopub.execute_input":"2024-09-26T01:23:13.294864Z","iopub.status.idle":"2024-09-26T01:23:13.300392Z","shell.execute_reply.started":"2024-09-26T01:23:13.294822Z","shell.execute_reply":"2024-09-26T01:23:13.299197Z"},"trusted":true},"execution_count":7,"outputs":[]},{"cell_type":"code","source":"# !git clone https://github.com/WongKinYiu/yolov7.git","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.301957Z","iopub.execute_input":"2024-09-26T01:23:13.302375Z","iopub.status.idle":"2024-09-26T01:23:13.311761Z","shell.execute_reply.started":"2024-09-26T01:23:13.302334Z","shell.execute_reply":"2024-09-26T01:23:13.310605Z"},"trusted":true},"execution_count":8,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Clone the YoloV9 repository, install the libray, and obtain the model\n# !git clone https://github.com/WongKinYiu/yolov9.git\n# !wget https://github.com/WongKinYiu/yolov9/releases/download/v0.1/yolov9-e.pt","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.313032Z","iopub.execute_input":"2024-09-26T01:23:13.313353Z","iopub.status.idle":"2024-09-26T01:23:13.323405Z","shell.execute_reply.started":"2024-09-26T01:23:13.313318Z","shell.execute_reply":"2024-09-26T01:23:13.322488Z"},"trusted":true},"execution_count":9,"outputs":[]},{"cell_type":"code","source":"# !wget https://github.com/WongKinYiu/yolov9/releases/download/v0.1/yolov9-c.pt","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.324708Z","iopub.execute_input":"2024-09-26T01:23:13.325038Z","iopub.status.idle":"2024-09-26T01:23:13.334391Z","shell.execute_reply.started":"2024-09-26T01:23:13.325015Z","shell.execute_reply":"2024-09-26T01:23:13.333378Z"},"trusted":true},"execution_count":10,"outputs":[]},{"cell_type":"code","source":"# !wget https://github.com/WongKinYiu/yolov9/releases/download/v0.1/gelan-c.pt","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.3411Z","iopub.execute_input":"2024-09-26T01:23:13.341594Z","iopub.status.idle":"2024-09-26T01:23:13.347348Z","shell.execute_reply.started":"2024-09-26T01:23:13.341554Z","shell.execute_reply":"2024-09-26T01:23:13.346391Z"},"trusted":true},"execution_count":11,"outputs":[]},{"cell_type":"code","source":"# !pip install -r /kaggle/working/yolov9/requirements.txt","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.348455Z","iopub.execute_input":"2024-09-26T01:23:13.348859Z","iopub.status.idle":"2024-09-26T01:23:13.357532Z","shell.execute_reply.started":"2024-09-26T01:23:13.348834Z","shell.execute_reply":"2024-09-26T01:23:13.356535Z"},"trusted":true},"execution_count":12,"outputs":[]},{"cell_type":"code","source":"# !pip install -r yolov7/requirements.txt\n# !wget https://github.com/WongKinYiu/yolov7/releases/download/v0.1/yolov7.pt","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.35878Z","iopub.execute_input":"2024-09-26T01:23:13.359076Z","iopub.status.idle":"2024-09-26T01:23:13.366894Z","shell.execute_reply.started":"2024-09-26T01:23:13.359053Z","shell.execute_reply":"2024-09-26T01:23:13.366001Z"},"trusted":true},"execution_count":13,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = 1024\nTRAIN_IMAGES_DIRECTORY = '/kaggle/working/data/train/'\nTEST_IMAGES_DIRECTORY = '/kaggle/working/data/test/'\n\nTRAIN_IMAGES_PATH_REFACTORED = '/kaggle/working/refactored_data/images/train/'\nTRAIN_LABELS_PATH_REFACTORED = '/kaggle/working/refactored_data/labels/train/'\nVALID_IMAGES_PATH_REFACTORED = '/kaggle/working/refactored_data/images/val/'\nVALID_LABELS_PATH_REFACTORED = '/kaggle/working/refactored_data/labels/val/'","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.368346Z","iopub.execute_input":"2024-09-26T01:23:13.368611Z","iopub.status.idle":"2024-09-26T01:23:13.379072Z","shell.execute_reply.started":"2024-09-26T01:23:13.368589Z","shell.execute_reply":"2024-09-26T01:23:13.378119Z"},"trusted":true},"execution_count":14,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('data/output.csv')\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.380505Z","iopub.execute_input":"2024-09-26T01:23:13.381412Z","iopub.status.idle":"2024-09-26T01:23:13.464651Z","shell.execute_reply.started":"2024-09-26T01:23:13.381381Z","shell.execute_reply":"2024-09-26T01:23:13.463717Z"},"trusted":true},"execution_count":15,"outputs":[{"execution_count":15,"output_type":"execute_result","data":{"text/plain":"                               image_id  class_id  x_min  y_min  x_max  y_max\n0      9a5094b2563a1ef3ff50dc5c7ff71345         3  339.7  593.0  816.0  787.3\n1      9a5094b2563a1ef3ff50dc5c7ff71345        10  880.0  757.0  923.0  873.0\n2      9a5094b2563a1ef3ff50dc5c7ff71345        11  880.0  757.0  923.0  873.0\n3      9a5094b2563a1ef3ff50dc5c7ff71345         0  517.0  313.0  639.0  423.0\n4      051132a778e61a86eb147c7c6f564dfe         3  423.7  463.7  907.7  594.3\n...                                 ...       ...    ...    ...    ...    ...\n22713  be53fe5a49231f1c1be020b0bdd8561f         7  216.0  388.0  282.0  451.0\n22714  380d07a94cc4b012812119370de47192         0  576.0  284.7  709.0  398.7\n22715  52951d7de2485aba8ed62629eee4d254         3  321.0  573.0  718.0  680.0\n22716  52951d7de2485aba8ed62629eee4d254         9  134.0  512.0  170.0  536.0\n22717  1224f07d895107573588225f692e94f9         0  515.3  314.0  638.3  441.3\n\n[22718 rows x 6 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image_id</th>\n      <th>class_id</th>\n      <th>x_min</th>\n      <th>y_min</th>\n      <th>x_max</th>\n      <th>y_max</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>3</td>\n      <td>339.7</td>\n      <td>593.0</td>\n      <td>816.0</td>\n      <td>787.3</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>10</td>\n      <td>880.0</td>\n      <td>757.0</td>\n      <td>923.0</td>\n      <td>873.0</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>11</td>\n      <td>880.0</td>\n      <td>757.0</td>\n      <td>923.0</td>\n      <td>873.0</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>0</td>\n      <td>517.0</td>\n      <td>313.0</td>\n      <td>639.0</td>\n      <td>423.0</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>051132a778e61a86eb147c7c6f564dfe</td>\n      <td>3</td>\n      <td>423.7</td>\n      <td>463.7</td>\n      <td>907.7</td>\n      <td>594.3</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>22713</th>\n      <td>be53fe5a49231f1c1be020b0bdd8561f</td>\n      <td>7</td>\n      <td>216.0</td>\n      <td>388.0</td>\n      <td>282.0</td>\n      <td>451.0</td>\n    </tr>\n    <tr>\n      <th>22714</th>\n      <td>380d07a94cc4b012812119370de47192</td>\n      <td>0</td>\n      <td>576.0</td>\n      <td>284.7</td>\n      <td>709.0</td>\n      <td>398.7</td>\n    </tr>\n    <tr>\n      <th>22715</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>3</td>\n      <td>321.0</td>\n      <td>573.0</td>\n      <td>718.0</td>\n      <td>680.0</td>\n    </tr>\n    <tr>\n      <th>22716</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>9</td>\n      <td>134.0</td>\n      <td>512.0</td>\n      <td>170.0</td>\n      <td>536.0</td>\n    </tr>\n    <tr>\n      <th>22717</th>\n      <td>1224f07d895107573588225f692e94f9</td>\n      <td>0</td>\n      <td>515.3</td>\n      <td>314.0</td>\n      <td>638.3</td>\n      <td>441.3</td>\n    </tr>\n  </tbody>\n</table>\n<p>22718 rows × 6 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"train_base_df = pd.read_csv('data/train.csv')\ntrain_base_df #image_id,class_name,class_id,rad_id,x_min,y_min,x_max,y_max,width,height","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.466087Z","iopub.execute_input":"2024-09-26T01:23:13.466518Z","iopub.status.idle":"2024-09-26T01:23:13.58309Z","shell.execute_reply.started":"2024-09-26T01:23:13.466485Z","shell.execute_reply":"2024-09-26T01:23:13.582166Z"},"trusted":true},"execution_count":16,"outputs":[{"execution_count":16,"output_type":"execute_result","data":{"text/plain":"                               image_id          class_name  class_id rad_id  \\\n0      50a418190bc3fb1ef1633bf9678929b3          No finding        14    R11   \n1      21a10246a5ec7af151081d0cd6d65dc9          No finding        14     R7   \n2      9a5094b2563a1ef3ff50dc5c7ff71345        Cardiomegaly         3    R10   \n3      051132a778e61a86eb147c7c6f564dfe  Aortic enlargement         0    R10   \n4      063319de25ce7edb9b1c6b8881290140          No finding        14    R10   \n...                                 ...                 ...       ...    ...   \n67909  936fd5cff1c058d39817a08f58b72cae          No finding        14     R1   \n67910  ca7e72954550eeb610fe22bf0244b7fa          No finding        14     R1   \n67911  aa17d5312a0fb4a2939436abca7f9579          No finding        14     R8   \n67912  4b56bc6d22b192f075f13231419dfcc8        Cardiomegaly         3     R8   \n67913  5e272e3adbdaafb07a7e84a9e62b1a4c          No finding        14    R16   \n\n        x_min   y_min   x_max   y_max  \n0         NaN     NaN     NaN     NaN  \n1         NaN     NaN     NaN     NaN  \n2       691.0  1375.0  1653.0  1831.0  \n3      1264.0   743.0  1611.0  1019.0  \n4         NaN     NaN     NaN     NaN  \n...       ...     ...     ...     ...  \n67909     NaN     NaN     NaN     NaN  \n67910     NaN     NaN     NaN     NaN  \n67911     NaN     NaN     NaN     NaN  \n67912   771.0   979.0  1680.0  1311.0  \n67913     NaN     NaN     NaN     NaN  \n\n[67914 rows x 8 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image_id</th>\n      <th>class_name</th>\n      <th>class_id</th>\n      <th>rad_id</th>\n      <th>x_min</th>\n      <th>y_min</th>\n      <th>x_max</th>\n      <th>y_max</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>50a418190bc3fb1ef1633bf9678929b3</td>\n      <td>No finding</td>\n      <td>14</td>\n      <td>R11</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>21a10246a5ec7af151081d0cd6d65dc9</td>\n      <td>No finding</td>\n      <td>14</td>\n      <td>R7</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>Cardiomegaly</td>\n      <td>3</td>\n      <td>R10</td>\n      <td>691.0</td>\n      <td>1375.0</td>\n      <td>1653.0</td>\n      <td>1831.0</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>051132a778e61a86eb147c7c6f564dfe</td>\n      <td>Aortic enlargement</td>\n      <td>0</td>\n      <td>R10</td>\n      <td>1264.0</td>\n      <td>743.0</td>\n      <td>1611.0</td>\n      <td>1019.0</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>063319de25ce7edb9b1c6b8881290140</td>\n      <td>No finding</td>\n      <td>14</td>\n      <td>R10</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>67909</th>\n      <td>936fd5cff1c058d39817a08f58b72cae</td>\n      <td>No finding</td>\n      <td>14</td>\n      <td>R1</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n    </tr>\n    <tr>\n      <th>67910</th>\n      <td>ca7e72954550eeb610fe22bf0244b7fa</td>\n      <td>No finding</td>\n      <td>14</td>\n      <td>R1</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n    </tr>\n    <tr>\n      <th>67911</th>\n      <td>aa17d5312a0fb4a2939436abca7f9579</td>\n      <td>No finding</td>\n      <td>14</td>\n      <td>R8</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n    </tr>\n    <tr>\n      <th>67912</th>\n      <td>4b56bc6d22b192f075f13231419dfcc8</td>\n      <td>Cardiomegaly</td>\n      <td>3</td>\n      <td>R8</td>\n      <td>771.0</td>\n      <td>979.0</td>\n      <td>1680.0</td>\n      <td>1311.0</td>\n    </tr>\n    <tr>\n      <th>67913</th>\n      <td>5e272e3adbdaafb07a7e84a9e62b1a4c</td>\n      <td>No finding</td>\n      <td>14</td>\n      <td>R16</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n      <td>NaN</td>\n    </tr>\n  </tbody>\n</table>\n<p>67914 rows × 8 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"# Tạo từ điển ánh xạ từ class_id đến class_name\nclass_id_to_name = dict(zip(train_base_df['class_id'], train_base_df['class_name']))\n\n# Thêm cột class_name vào new_df dựa trên class_id\ntrain_df['class_name'] = train_df['class_id'].map(class_id_to_name)\n\n# Kiểm tra new_df sau khi thêm cột\ntrain_df\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.584685Z","iopub.execute_input":"2024-09-26T01:23:13.585085Z","iopub.status.idle":"2024-09-26T01:23:13.624628Z","shell.execute_reply.started":"2024-09-26T01:23:13.585053Z","shell.execute_reply":"2024-09-26T01:23:13.62383Z"},"trusted":true},"execution_count":17,"outputs":[{"execution_count":17,"output_type":"execute_result","data":{"text/plain":"                               image_id  class_id  x_min  y_min  x_max  y_max  \\\n0      9a5094b2563a1ef3ff50dc5c7ff71345         3  339.7  593.0  816.0  787.3   \n1      9a5094b2563a1ef3ff50dc5c7ff71345        10  880.0  757.0  923.0  873.0   \n2      9a5094b2563a1ef3ff50dc5c7ff71345        11  880.0  757.0  923.0  873.0   \n3      9a5094b2563a1ef3ff50dc5c7ff71345         0  517.0  313.0  639.0  423.0   \n4      051132a778e61a86eb147c7c6f564dfe         3  423.7  463.7  907.7  594.3   \n...                                 ...       ...    ...    ...    ...    ...   \n22713  be53fe5a49231f1c1be020b0bdd8561f         7  216.0  388.0  282.0  451.0   \n22714  380d07a94cc4b012812119370de47192         0  576.0  284.7  709.0  398.7   \n22715  52951d7de2485aba8ed62629eee4d254         3  321.0  573.0  718.0  680.0   \n22716  52951d7de2485aba8ed62629eee4d254         9  134.0  512.0  170.0  536.0   \n22717  1224f07d895107573588225f692e94f9         0  515.3  314.0  638.3  441.3   \n\n               class_name  \n0            Cardiomegaly  \n1        Pleural effusion  \n2      Pleural thickening  \n3      Aortic enlargement  \n4            Cardiomegaly  \n...                   ...  \n22713        Lung Opacity  \n22714  Aortic enlargement  \n22715        Cardiomegaly  \n22716        Other lesion  \n22717  Aortic enlargement  \n\n[22718 rows x 7 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image_id</th>\n      <th>class_id</th>\n      <th>x_min</th>\n      <th>y_min</th>\n      <th>x_max</th>\n      <th>y_max</th>\n      <th>class_name</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>3</td>\n      <td>339.7</td>\n      <td>593.0</td>\n      <td>816.0</td>\n      <td>787.3</td>\n      <td>Cardiomegaly</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>10</td>\n      <td>880.0</td>\n      <td>757.0</td>\n      <td>923.0</td>\n      <td>873.0</td>\n      <td>Pleural effusion</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>11</td>\n      <td>880.0</td>\n      <td>757.0</td>\n      <td>923.0</td>\n      <td>873.0</td>\n      <td>Pleural thickening</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>0</td>\n      <td>517.0</td>\n      <td>313.0</td>\n      <td>639.0</td>\n      <td>423.0</td>\n      <td>Aortic enlargement</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>051132a778e61a86eb147c7c6f564dfe</td>\n      <td>3</td>\n      <td>423.7</td>\n      <td>463.7</td>\n      <td>907.7</td>\n      <td>594.3</td>\n      <td>Cardiomegaly</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>22713</th>\n      <td>be53fe5a49231f1c1be020b0bdd8561f</td>\n      <td>7</td>\n      <td>216.0</td>\n      <td>388.0</td>\n      <td>282.0</td>\n      <td>451.0</td>\n      <td>Lung Opacity</td>\n    </tr>\n    <tr>\n      <th>22714</th>\n      <td>380d07a94cc4b012812119370de47192</td>\n      <td>0</td>\n      <td>576.0</td>\n      <td>284.7</td>\n      <td>709.0</td>\n      <td>398.7</td>\n      <td>Aortic enlargement</td>\n    </tr>\n    <tr>\n      <th>22715</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>3</td>\n      <td>321.0</td>\n      <td>573.0</td>\n      <td>718.0</td>\n      <td>680.0</td>\n      <td>Cardiomegaly</td>\n    </tr>\n    <tr>\n      <th>22716</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>9</td>\n      <td>134.0</td>\n      <td>512.0</td>\n      <td>170.0</td>\n      <td>536.0</td>\n      <td>Other lesion</td>\n    </tr>\n    <tr>\n      <th>22717</th>\n      <td>1224f07d895107573588225f692e94f9</td>\n      <td>0</td>\n      <td>515.3</td>\n      <td>314.0</td>\n      <td>638.3</td>\n      <td>441.3</td>\n      <td>Aortic enlargement</td>\n    </tr>\n  </tbody>\n</table>\n<p>22718 rows × 7 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"path = 'data/train/'\nimg_path = []\nfor i in train_df['image_id']:\n  img_path.append(path+i+'.png')\n\ntrain_df['img_path'] = img_path\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.62562Z","iopub.execute_input":"2024-09-26T01:23:13.625907Z","iopub.status.idle":"2024-09-26T01:23:13.655662Z","shell.execute_reply.started":"2024-09-26T01:23:13.625885Z","shell.execute_reply":"2024-09-26T01:23:13.654838Z"},"trusted":true},"execution_count":18,"outputs":[{"execution_count":18,"output_type":"execute_result","data":{"text/plain":"                               image_id  class_id  x_min  y_min  x_max  y_max  \\\n0      9a5094b2563a1ef3ff50dc5c7ff71345         3  339.7  593.0  816.0  787.3   \n1      9a5094b2563a1ef3ff50dc5c7ff71345        10  880.0  757.0  923.0  873.0   \n2      9a5094b2563a1ef3ff50dc5c7ff71345        11  880.0  757.0  923.0  873.0   \n3      9a5094b2563a1ef3ff50dc5c7ff71345         0  517.0  313.0  639.0  423.0   \n4      051132a778e61a86eb147c7c6f564dfe         3  423.7  463.7  907.7  594.3   \n...                                 ...       ...    ...    ...    ...    ...   \n22713  be53fe5a49231f1c1be020b0bdd8561f         7  216.0  388.0  282.0  451.0   \n22714  380d07a94cc4b012812119370de47192         0  576.0  284.7  709.0  398.7   \n22715  52951d7de2485aba8ed62629eee4d254         3  321.0  573.0  718.0  680.0   \n22716  52951d7de2485aba8ed62629eee4d254         9  134.0  512.0  170.0  536.0   \n22717  1224f07d895107573588225f692e94f9         0  515.3  314.0  638.3  441.3   \n\n               class_name                                         img_path  \n0            Cardiomegaly  data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  \n1        Pleural effusion  data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  \n2      Pleural thickening  data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  \n3      Aortic enlargement  data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  \n4            Cardiomegaly  data/train/051132a778e61a86eb147c7c6f564dfe.png  \n...                   ...                                              ...  \n22713        Lung Opacity  data/train/be53fe5a49231f1c1be020b0bdd8561f.png  \n22714  Aortic enlargement  data/train/380d07a94cc4b012812119370de47192.png  \n22715        Cardiomegaly  data/train/52951d7de2485aba8ed62629eee4d254.png  \n22716        Other lesion  data/train/52951d7de2485aba8ed62629eee4d254.png  \n22717  Aortic enlargement  data/train/1224f07d895107573588225f692e94f9.png  \n\n[22718 rows x 8 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image_id</th>\n      <th>class_id</th>\n      <th>x_min</th>\n      <th>y_min</th>\n      <th>x_max</th>\n      <th>y_max</th>\n      <th>class_name</th>\n      <th>img_path</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>3</td>\n      <td>339.7</td>\n      <td>593.0</td>\n      <td>816.0</td>\n      <td>787.3</td>\n      <td>Cardiomegaly</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>10</td>\n      <td>880.0</td>\n      <td>757.0</td>\n      <td>923.0</td>\n      <td>873.0</td>\n      <td>Pleural effusion</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>11</td>\n      <td>880.0</td>\n      <td>757.0</td>\n      <td>923.0</td>\n      <td>873.0</td>\n      <td>Pleural thickening</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>0</td>\n      <td>517.0</td>\n      <td>313.0</td>\n      <td>639.0</td>\n      <td>423.0</td>\n      <td>Aortic enlargement</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>051132a778e61a86eb147c7c6f564dfe</td>\n      <td>3</td>\n      <td>423.7</td>\n      <td>463.7</td>\n      <td>907.7</td>\n      <td>594.3</td>\n      <td>Cardiomegaly</td>\n      <td>data/train/051132a778e61a86eb147c7c6f564dfe.png</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>22713</th>\n      <td>be53fe5a49231f1c1be020b0bdd8561f</td>\n      <td>7</td>\n      <td>216.0</td>\n      <td>388.0</td>\n      <td>282.0</td>\n      <td>451.0</td>\n      <td>Lung Opacity</td>\n      <td>data/train/be53fe5a49231f1c1be020b0bdd8561f.png</td>\n    </tr>\n    <tr>\n      <th>22714</th>\n      <td>380d07a94cc4b012812119370de47192</td>\n      <td>0</td>\n      <td>576.0</td>\n      <td>284.7</td>\n      <td>709.0</td>\n      <td>398.7</td>\n      <td>Aortic enlargement</td>\n      <td>data/train/380d07a94cc4b012812119370de47192.png</td>\n    </tr>\n    <tr>\n      <th>22715</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>3</td>\n      <td>321.0</td>\n      <td>573.0</td>\n      <td>718.0</td>\n      <td>680.0</td>\n      <td>Cardiomegaly</td>\n      <td>data/train/52951d7de2485aba8ed62629eee4d254.png</td>\n    </tr>\n    <tr>\n      <th>22716</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>9</td>\n      <td>134.0</td>\n      <td>512.0</td>\n      <td>170.0</td>\n      <td>536.0</td>\n      <td>Other lesion</td>\n      <td>data/train/52951d7de2485aba8ed62629eee4d254.png</td>\n    </tr>\n    <tr>\n      <th>22717</th>\n      <td>1224f07d895107573588225f692e94f9</td>\n      <td>0</td>\n      <td>515.3</td>\n      <td>314.0</td>\n      <td>638.3</td>\n      <td>441.3</td>\n      <td>Aortic enlargement</td>\n      <td>data/train/1224f07d895107573588225f692e94f9.png</td>\n    </tr>\n  </tbody>\n</table>\n<p>22718 rows × 8 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"#x_min, y_min, x_max, y_max normalization값으로 update\ntrain_df['x_min'] = train_df.apply(lambda row: (row.x_min) /1024, axis =1)\ntrain_df['y_min'] = train_df.apply(lambda row: (row.y_min) /1024, axis =1)\n\ntrain_df['x_max'] = train_df.apply(lambda row: (row.x_max) /1024, axis =1)\ntrain_df['y_max'] = train_df.apply(lambda row: (row.y_max) /1024, axis =1)\n#x_mid, y_mid가 추가\ntrain_df['x_mid'] = train_df.apply(lambda row: (row.x_min + row.x_max)/2,axis =1)\ntrain_df['y_mid'] = train_df.apply(lambda row: (row.y_min + row.y_max)/2, axis =1)\n#normalization된 width & height 추가\ntrain_df['w'] = train_df.apply(lambda row: (row.x_max - row.x_min), axis = 1)\ntrain_df['h'] = train_df.apply(lambda row: (row.y_max - row.y_min), axis = 1)\n#area 추가\ntrain_df['area'] = train_df['w']*train_df['h']\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:13.65681Z","iopub.execute_input":"2024-09-26T01:23:13.657135Z","iopub.status.idle":"2024-09-26T01:23:17.392676Z","shell.execute_reply.started":"2024-09-26T01:23:13.657103Z","shell.execute_reply":"2024-09-26T01:23:17.391757Z"},"trusted":true},"execution_count":19,"outputs":[{"execution_count":19,"output_type":"execute_result","data":{"text/plain":"                               image_id  class_id     x_min     y_min  \\\n0      9a5094b2563a1ef3ff50dc5c7ff71345         3  0.331738  0.579102   \n1      9a5094b2563a1ef3ff50dc5c7ff71345        10  0.859375  0.739258   \n2      9a5094b2563a1ef3ff50dc5c7ff71345        11  0.859375  0.739258   \n3      9a5094b2563a1ef3ff50dc5c7ff71345         0  0.504883  0.305664   \n4      051132a778e61a86eb147c7c6f564dfe         3  0.413770  0.452832   \n...                                 ...       ...       ...       ...   \n22713  be53fe5a49231f1c1be020b0bdd8561f         7  0.210938  0.378906   \n22714  380d07a94cc4b012812119370de47192         0  0.562500  0.278027   \n22715  52951d7de2485aba8ed62629eee4d254         3  0.313477  0.559570   \n22716  52951d7de2485aba8ed62629eee4d254         9  0.130859  0.500000   \n22717  1224f07d895107573588225f692e94f9         0  0.503223  0.306641   \n\n          x_max     y_max          class_name  \\\n0      0.796875  0.768848        Cardiomegaly   \n1      0.901367  0.852539    Pleural effusion   \n2      0.901367  0.852539  Pleural thickening   \n3      0.624023  0.413086  Aortic enlargement   \n4      0.886426  0.580371        Cardiomegaly   \n...         ...       ...                 ...   \n22713  0.275391  0.440430        Lung Opacity   \n22714  0.692383  0.389355  Aortic enlargement   \n22715  0.701172  0.664062        Cardiomegaly   \n22716  0.166016  0.523438        Other lesion   \n22717  0.623340  0.430957  Aortic enlargement   \n\n                                              img_path     x_mid     y_mid  \\\n0      data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  0.564307  0.673975   \n1      data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  0.880371  0.795898   \n2      data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  0.880371  0.795898   \n3      data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  0.564453  0.359375   \n4      data/train/051132a778e61a86eb147c7c6f564dfe.png  0.650098  0.516602   \n...                                                ...       ...       ...   \n22713  data/train/be53fe5a49231f1c1be020b0bdd8561f.png  0.243164  0.409668   \n22714  data/train/380d07a94cc4b012812119370de47192.png  0.627441  0.333691   \n22715  data/train/52951d7de2485aba8ed62629eee4d254.png  0.507324  0.611816   \n22716  data/train/52951d7de2485aba8ed62629eee4d254.png  0.148438  0.511719   \n22717  data/train/1224f07d895107573588225f692e94f9.png  0.563281  0.368799   \n\n              w         h      area  \n0      0.465137  0.189746  0.088258  \n1      0.041992  0.113281  0.004757  \n2      0.041992  0.113281  0.004757  \n3      0.119141  0.107422  0.012798  \n4      0.472656  0.127539  0.060282  \n...         ...       ...       ...  \n22713  0.064453  0.061523  0.003965  \n22714  0.129883  0.111328  0.014460  \n22715  0.387695  0.104492  0.040511  \n22716  0.035156  0.023438  0.000824  \n22717  0.120117  0.124316  0.014933  \n\n[22718 rows x 13 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image_id</th>\n      <th>class_id</th>\n      <th>x_min</th>\n      <th>y_min</th>\n      <th>x_max</th>\n      <th>y_max</th>\n      <th>class_name</th>\n      <th>img_path</th>\n      <th>x_mid</th>\n      <th>y_mid</th>\n      <th>w</th>\n      <th>h</th>\n      <th>area</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>3</td>\n      <td>0.331738</td>\n      <td>0.579102</td>\n      <td>0.796875</td>\n      <td>0.768848</td>\n      <td>Cardiomegaly</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n      <td>0.564307</td>\n      <td>0.673975</td>\n      <td>0.465137</td>\n      <td>0.189746</td>\n      <td>0.088258</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>10</td>\n      <td>0.859375</td>\n      <td>0.739258</td>\n      <td>0.901367</td>\n      <td>0.852539</td>\n      <td>Pleural effusion</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n      <td>0.880371</td>\n      <td>0.795898</td>\n      <td>0.041992</td>\n      <td>0.113281</td>\n      <td>0.004757</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>11</td>\n      <td>0.859375</td>\n      <td>0.739258</td>\n      <td>0.901367</td>\n      <td>0.852539</td>\n      <td>Pleural thickening</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n      <td>0.880371</td>\n      <td>0.795898</td>\n      <td>0.041992</td>\n      <td>0.113281</td>\n      <td>0.004757</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>0</td>\n      <td>0.504883</td>\n      <td>0.305664</td>\n      <td>0.624023</td>\n      <td>0.413086</td>\n      <td>Aortic enlargement</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n      <td>0.564453</td>\n      <td>0.359375</td>\n      <td>0.119141</td>\n      <td>0.107422</td>\n      <td>0.012798</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>051132a778e61a86eb147c7c6f564dfe</td>\n      <td>3</td>\n      <td>0.413770</td>\n      <td>0.452832</td>\n      <td>0.886426</td>\n      <td>0.580371</td>\n      <td>Cardiomegaly</td>\n      <td>data/train/051132a778e61a86eb147c7c6f564dfe.png</td>\n      <td>0.650098</td>\n      <td>0.516602</td>\n      <td>0.472656</td>\n      <td>0.127539</td>\n      <td>0.060282</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>22713</th>\n      <td>be53fe5a49231f1c1be020b0bdd8561f</td>\n      <td>7</td>\n      <td>0.210938</td>\n      <td>0.378906</td>\n      <td>0.275391</td>\n      <td>0.440430</td>\n      <td>Lung Opacity</td>\n      <td>data/train/be53fe5a49231f1c1be020b0bdd8561f.png</td>\n      <td>0.243164</td>\n      <td>0.409668</td>\n      <td>0.064453</td>\n      <td>0.061523</td>\n      <td>0.003965</td>\n    </tr>\n    <tr>\n      <th>22714</th>\n      <td>380d07a94cc4b012812119370de47192</td>\n      <td>0</td>\n      <td>0.562500</td>\n      <td>0.278027</td>\n      <td>0.692383</td>\n      <td>0.389355</td>\n      <td>Aortic enlargement</td>\n      <td>data/train/380d07a94cc4b012812119370de47192.png</td>\n      <td>0.627441</td>\n      <td>0.333691</td>\n      <td>0.129883</td>\n      <td>0.111328</td>\n      <td>0.014460</td>\n    </tr>\n    <tr>\n      <th>22715</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>3</td>\n      <td>0.313477</td>\n      <td>0.559570</td>\n      <td>0.701172</td>\n      <td>0.664062</td>\n      <td>Cardiomegaly</td>\n      <td>data/train/52951d7de2485aba8ed62629eee4d254.png</td>\n      <td>0.507324</td>\n      <td>0.611816</td>\n      <td>0.387695</td>\n      <td>0.104492</td>\n      <td>0.040511</td>\n    </tr>\n    <tr>\n      <th>22716</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>9</td>\n      <td>0.130859</td>\n      <td>0.500000</td>\n      <td>0.166016</td>\n      <td>0.523438</td>\n      <td>Other lesion</td>\n      <td>data/train/52951d7de2485aba8ed62629eee4d254.png</td>\n      <td>0.148438</td>\n      <td>0.511719</td>\n      <td>0.035156</td>\n      <td>0.023438</td>\n      <td>0.000824</td>\n    </tr>\n    <tr>\n      <th>22717</th>\n      <td>1224f07d895107573588225f692e94f9</td>\n      <td>0</td>\n      <td>0.503223</td>\n      <td>0.306641</td>\n      <td>0.623340</td>\n      <td>0.430957</td>\n      <td>Aortic enlargement</td>\n      <td>data/train/1224f07d895107573588225f692e94f9.png</td>\n      <td>0.563281</td>\n      <td>0.368799</td>\n      <td>0.120117</td>\n      <td>0.124316</td>\n      <td>0.014933</td>\n    </tr>\n  </tbody>\n</table>\n<p>22718 rows × 13 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"train_df['ori_x_min'] = (train_df['x_min']*1024).astype('int')\ntrain_df['ori_y_min'] = (train_df['y_min']*1024).astype('int')\ntrain_df['ori_x_max'] = (train_df['x_max']*1024).astype('int')\ntrain_df['ori_y_max'] = (train_df['y_max']*1024).astype('int')\n\ntrain_df['ori_x_mid'] = (train_df['x_mid']*1024).astype('int')\ntrain_df['ori_y_mid'] = (train_df['y_mid']*1024).astype('int')\ntrain_df['ori_w'] = (train_df['w']*1024).astype('int')\ntrain_df['ori_h'] = (train_df['h']*1024).astype('int')","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:17.393942Z","iopub.execute_input":"2024-09-26T01:23:17.394307Z","iopub.status.idle":"2024-09-26T01:23:17.408322Z","shell.execute_reply.started":"2024-09-26T01:23:17.39428Z","shell.execute_reply":"2024-09-26T01:23:17.407596Z"},"trusted":true},"execution_count":20,"outputs":[]},{"cell_type":"code","source":"final_df = train_df.copy()\nfinal_df = final_df[['image_id', 'class_name', 'class_id', 'ori_x_min', 'ori_y_min', 'ori_x_max', 'ori_y_max', 'ori_x_mid', 'ori_y_mid', 'ori_w', 'ori_h','img_path']]\nfinal_df","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:17.409583Z","iopub.execute_input":"2024-09-26T01:23:17.410965Z","iopub.status.idle":"2024-09-26T01:23:17.435405Z","shell.execute_reply.started":"2024-09-26T01:23:17.410931Z","shell.execute_reply":"2024-09-26T01:23:17.434572Z"},"trusted":true},"execution_count":21,"outputs":[{"execution_count":21,"output_type":"execute_result","data":{"text/plain":"                               image_id          class_name  class_id  \\\n0      9a5094b2563a1ef3ff50dc5c7ff71345        Cardiomegaly         3   \n1      9a5094b2563a1ef3ff50dc5c7ff71345    Pleural effusion        10   \n2      9a5094b2563a1ef3ff50dc5c7ff71345  Pleural thickening        11   \n3      9a5094b2563a1ef3ff50dc5c7ff71345  Aortic enlargement         0   \n4      051132a778e61a86eb147c7c6f564dfe        Cardiomegaly         3   \n...                                 ...                 ...       ...   \n22713  be53fe5a49231f1c1be020b0bdd8561f        Lung Opacity         7   \n22714  380d07a94cc4b012812119370de47192  Aortic enlargement         0   \n22715  52951d7de2485aba8ed62629eee4d254        Cardiomegaly         3   \n22716  52951d7de2485aba8ed62629eee4d254        Other lesion         9   \n22717  1224f07d895107573588225f692e94f9  Aortic enlargement         0   \n\n       ori_x_min  ori_y_min  ori_x_max  ori_y_max  ori_x_mid  ori_y_mid  \\\n0            339        593        816        787        577        690   \n1            880        757        923        873        901        815   \n2            880        757        923        873        901        815   \n3            517        313        639        423        578        368   \n4            423        463        907        594        665        529   \n...          ...        ...        ...        ...        ...        ...   \n22713        216        388        282        451        249        419   \n22714        576        284        709        398        642        341   \n22715        321        573        718        680        519        626   \n22716        134        512        170        536        152        524   \n22717        515        314        638        441        576        377   \n\n       ori_w  ori_h                                         img_path  \n0        476    194  data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  \n1         43    116  data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  \n2         43    116  data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  \n3        122    110  data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png  \n4        484    130  data/train/051132a778e61a86eb147c7c6f564dfe.png  \n...      ...    ...                                              ...  \n22713     66     63  data/train/be53fe5a49231f1c1be020b0bdd8561f.png  \n22714    133    114  data/train/380d07a94cc4b012812119370de47192.png  \n22715    397    107  data/train/52951d7de2485aba8ed62629eee4d254.png  \n22716     36     24  data/train/52951d7de2485aba8ed62629eee4d254.png  \n22717    123    127  data/train/1224f07d895107573588225f692e94f9.png  \n\n[22718 rows x 12 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image_id</th>\n      <th>class_name</th>\n      <th>class_id</th>\n      <th>ori_x_min</th>\n      <th>ori_y_min</th>\n      <th>ori_x_max</th>\n      <th>ori_y_max</th>\n      <th>ori_x_mid</th>\n      <th>ori_y_mid</th>\n      <th>ori_w</th>\n      <th>ori_h</th>\n      <th>img_path</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>Cardiomegaly</td>\n      <td>3</td>\n      <td>339</td>\n      <td>593</td>\n      <td>816</td>\n      <td>787</td>\n      <td>577</td>\n      <td>690</td>\n      <td>476</td>\n      <td>194</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>Pleural effusion</td>\n      <td>10</td>\n      <td>880</td>\n      <td>757</td>\n      <td>923</td>\n      <td>873</td>\n      <td>901</td>\n      <td>815</td>\n      <td>43</td>\n      <td>116</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>Pleural thickening</td>\n      <td>11</td>\n      <td>880</td>\n      <td>757</td>\n      <td>923</td>\n      <td>873</td>\n      <td>901</td>\n      <td>815</td>\n      <td>43</td>\n      <td>116</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>9a5094b2563a1ef3ff50dc5c7ff71345</td>\n      <td>Aortic enlargement</td>\n      <td>0</td>\n      <td>517</td>\n      <td>313</td>\n      <td>639</td>\n      <td>423</td>\n      <td>578</td>\n      <td>368</td>\n      <td>122</td>\n      <td>110</td>\n      <td>data/train/9a5094b2563a1ef3ff50dc5c7ff71345.png</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>051132a778e61a86eb147c7c6f564dfe</td>\n      <td>Cardiomegaly</td>\n      <td>3</td>\n      <td>423</td>\n      <td>463</td>\n      <td>907</td>\n      <td>594</td>\n      <td>665</td>\n      <td>529</td>\n      <td>484</td>\n      <td>130</td>\n      <td>data/train/051132a778e61a86eb147c7c6f564dfe.png</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>22713</th>\n      <td>be53fe5a49231f1c1be020b0bdd8561f</td>\n      <td>Lung Opacity</td>\n      <td>7</td>\n      <td>216</td>\n      <td>388</td>\n      <td>282</td>\n      <td>451</td>\n      <td>249</td>\n      <td>419</td>\n      <td>66</td>\n      <td>63</td>\n      <td>data/train/be53fe5a49231f1c1be020b0bdd8561f.png</td>\n    </tr>\n    <tr>\n      <th>22714</th>\n      <td>380d07a94cc4b012812119370de47192</td>\n      <td>Aortic enlargement</td>\n      <td>0</td>\n      <td>576</td>\n      <td>284</td>\n      <td>709</td>\n      <td>398</td>\n      <td>642</td>\n      <td>341</td>\n      <td>133</td>\n      <td>114</td>\n      <td>data/train/380d07a94cc4b012812119370de47192.png</td>\n    </tr>\n    <tr>\n      <th>22715</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>Cardiomegaly</td>\n      <td>3</td>\n      <td>321</td>\n      <td>573</td>\n      <td>718</td>\n      <td>680</td>\n      <td>519</td>\n      <td>626</td>\n      <td>397</td>\n      <td>107</td>\n      <td>data/train/52951d7de2485aba8ed62629eee4d254.png</td>\n    </tr>\n    <tr>\n      <th>22716</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>Other lesion</td>\n      <td>9</td>\n      <td>134</td>\n      <td>512</td>\n      <td>170</td>\n      <td>536</td>\n      <td>152</td>\n      <td>524</td>\n      <td>36</td>\n      <td>24</td>\n      <td>data/train/52951d7de2485aba8ed62629eee4d254.png</td>\n    </tr>\n    <tr>\n      <th>22717</th>\n      <td>1224f07d895107573588225f692e94f9</td>\n      <td>Aortic enlargement</td>\n      <td>0</td>\n      <td>515</td>\n      <td>314</td>\n      <td>638</td>\n      <td>441</td>\n      <td>576</td>\n      <td>377</td>\n      <td>123</td>\n      <td>127</td>\n      <td>data/train/1224f07d895107573588225f692e94f9.png</td>\n    </tr>\n  </tbody>\n</table>\n<p>22718 rows × 12 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"import pandas as pd\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\nimport os\n\n# Đọc dữ liệu từ final_df (giả sử final_df đã được định nghĩa trước đó)\nfiltered_df = final_df[final_df['class_id'].isin([1, 12])].copy()\n\n# Hàm để cập nhật bounding box sau khi tăng cường\ndef update_bounding_box(bbox, transform_matrix):\n    \"\"\"\n    Cập nhật bounding box sau khi áp dụng ma trận biến đổi.\n    \n    Args:\n        bbox (list): Danh sách chứa các tọa độ bounding box [x_min, y_min, x_max, y_max].\n        transform_matrix (np.array): Ma trận biến đổi affine 2x3.\n    \n    Returns:\n        tuple: Tọa độ bounding box mới [x_min, y_min, x_max, y_max].\n    \"\"\"\n    ori_x_min, ori_y_min, ori_x_max, ori_y_max = bbox\n    \n    # Tạo mảng các điểm của bounding box\n    points = np.array([\n        [ori_x_min, ori_y_min],  # Góc trên bên trái\n        [ori_x_max, ori_y_min],  # Góc trên bên phải\n        [ori_x_max, ori_y_max],  # Góc dưới bên phải\n        [ori_x_min, ori_y_max]   # Góc dưới bên trái\n    ])\n    \n    # Chuyển đổi điểm bằng ma trận biến đổi\n    # Thay đổi transform_matrix thành dạng 3x3 để dễ tính toán\n    # Thêm hàng thứ ba [0, 0, 1] vào ma trận biến đổi để phù hợp với tọa độ đồng nhất\n    transform_matrix_3x3 = np.vstack([transform_matrix, [0, 0, 1]])\n    \n    # Thêm hàng thứ ba [1] vào các điểm để phù hợp với tọa độ đồng nhất\n    points_homogeneous = np.hstack([points, np.ones((points.shape[0], 1))])\n    \n    # Áp dụng ma trận biến đổi\n    transformed_points = np.dot(points_homogeneous, transform_matrix_3x3.T)\n    \n    # Chuyển đổi trở lại tọa độ không đồng nhất\n    transformed_points[:, 0] /= transformed_points[:, 2]\n    transformed_points[:, 1] /= transformed_points[:, 2]\n    \n    # Tìm giá trị min và max cho các trục x và y\n    ori_x_min = transformed_points[:, 0].min()\n    ori_y_min = transformed_points[:, 1].min()\n    ori_x_max = transformed_points[:, 0].max()\n    ori_y_max = transformed_points[:, 1].max()\n    \n    return ori_x_min, ori_y_min, ori_x_max, ori_y_max\n\n# Thực hiện tăng cường dữ liệu và lưu thông số mới vào CSV\nnew_rows = []\n\nfor index, row in tqdm(filtered_df.iterrows(), total=filtered_df.shape[0]):\n    image_path = row['img_path']\n    bbox = [row['ori_x_min'], row['ori_y_min'], row['ori_x_max'], row['ori_y_max']]\n    image_id = row['image_id']\n    class_name = row['class_name']\n    class_id = row['class_id']\n    ori_x_mid = row['ori_x_mid']\n    ori_y_mid = row['ori_y_mid']\n    ori_w = row['ori_w']\n    ori_h = row['ori_h']\n    \n    # Đọc ảnh\n    image = cv2.imread(image_path)\n    if image is None:\n        continue\n    \n    # Ví dụ về một biến đổi: xoay ảnh 90 độ\n    height, width = image.shape[:2]\n    M = cv2.getRotationMatrix2D((width / 2, height / 2), 90, 1)\n    rotated_image = cv2.warpAffine(image, M, (width, height))\n    \n    # Cập nhật bounding box\n    new_bbox = update_bounding_box(bbox, M)\n    \n    # Lưu ảnh mới\n    new_image_path = os.path.splitext(image_path)[0] + '_rotated.png'\n    cv2.imwrite(new_image_path, rotated_image)\n    \n    # Tạo một dòng mới với thông số cập nhật\n    new_row = row.copy()\n    new_row['img_path'] = new_image_path\n    new_row['ori_x_min'], new_row['ori_y_min'], new_row['ori_x_max'], new_row['ori_y_max'] = new_bbox\n    new_row['image_id'] = image_id + '_rotated'\n    new_row['class_name'] = class_name\n    new_row['class_id'] = class_id\n    new_row['ori_x_mid'] = ori_x_mid\n    new_row['ori_y_mid'] = ori_y_mid\n    new_row['ori_w'] = ori_w\n    new_row['ori_h'] = ori_h\n    new_rows.append(new_row)\n\n# Tạo DataFrame từ các dòng mới\nnew5_df = pd.DataFrame(new_rows)\n\n# # Gộp dữ liệu mới vào DataFrame gốc\n# augmented_df = pd.concat([final_df, new_df], ignore_index=True)\n\n# # Lưu DataFrame đã cập nhật vào file CSV mới\n# augmented_df.to_csv('/kaggle/working/data/train_augmented.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:17.436516Z","iopub.execute_input":"2024-09-26T01:23:17.436815Z","iopub.status.idle":"2024-09-26T01:23:42.994979Z","shell.execute_reply.started":"2024-09-26T01:23:17.436792Z","shell.execute_reply":"2024-09-26T01:23:42.993857Z"},"trusted":true},"execution_count":22,"outputs":[{"name":"stderr","text":"100%|██████████| 354/354 [00:25<00:00, 13.88it/s]\n","output_type":"stream"}]},{"cell_type":"code","source":"def flip_image(image_path, bbox):\n    \"\"\"\n    Lật ảnh theo chiều ngang và cập nhật bounding box.\n\n    Args:\n        image_path (str): Đường dẫn đến ảnh gốc.\n        bbox (list): Danh sách chứa các tọa độ bounding box [x_min, y_min, x_max, y_max].\n\n    Returns:\n        tuple: Đường dẫn ảnh mới và bounding box đã được cập nhật.\n    \"\"\"\n    # Đọc ảnh gốc\n    image = cv2.imread(image_path)\n    if image is None:\n        return None, None\n\n    # Lật ảnh theo chiều ngang\n    flipped_image = cv2.flip(image, 1)\n\n    # Cập nhật bounding box\n    width = image.shape[1]\n    flipped_bbox = [\n        width - bbox[2], bbox[1],  # x_max -> new_x_min, y_min\n        width - bbox[0], bbox[3]   # x_min -> new_x_max, y_max\n    ]\n\n    # Lưu ảnh mới\n    new_image_path = os.path.splitext(image_path)[0] + '_flip.png'\n    cv2.imwrite(new_image_path, flipped_image)\n    \n    return new_image_path, flipped_bbox\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:42.996647Z","iopub.execute_input":"2024-09-26T01:23:42.997102Z","iopub.status.idle":"2024-09-26T01:23:43.007814Z","shell.execute_reply.started":"2024-09-26T01:23:42.997061Z","shell.execute_reply":"2024-09-26T01:23:43.006454Z"},"trusted":true},"execution_count":23,"outputs":[]},{"cell_type":"code","source":"def flip_images_in_df(df):\n    \"\"\"\n    Lật ảnh cho các lớp cụ thể và cập nhật thông tin bounding box trong DataFrame.\n\n    Args:\n        df (pd.DataFrame): DataFrame chứa thông tin về ảnh và bounding box.\n\n    Returns:\n        pd.DataFrame: DataFrame đã được cập nhật với ảnh lật và bounding box mới.\n    \"\"\"\n    filtered_df = final_df[final_df['class_id'].isin([1, 12,2,4])].copy()\n    new_rows = []\n\n    for index, row in tqdm(df.iterrows(), total=df.shape[0]):\n        image_path = row['img_path']\n        bbox = [row['ori_x_min'], row['ori_y_min'], row['ori_x_max'], row['ori_y_max']]\n        image_id = row['image_id']\n        class_name = row['class_name']\n        class_id = row['class_id']\n        ori_x_mid = row['ori_x_mid']\n        ori_y_mid = row['ori_y_mid']\n        ori_w = row['ori_w']\n        ori_h = row['ori_h']\n        \n        # Lật ảnh và cập nhật bounding box\n        new_image_path, new_bbox = flip_image(image_path, bbox)\n        \n        if new_image_path is None:\n            continue\n        \n        # Tạo một dòng mới với thông số cập nhật\n        new_row = row.copy()\n        new_row['img_path'] = new_image_path\n        new_row['ori_x_min'], new_row['ori_y_min'], new_row['ori_x_max'], new_row['ori_y_max'] = new_bbox\n        new_row['image_id'] = image_id + '_flip'\n        new_row['class_name'] = class_name\n        new_row['class_id'] = class_id\n        new_row['ori_x_mid'] = ori_x_mid\n        new_row['ori_y_mid'] = ori_y_mid\n        new_row['ori_w'] = ori_w\n        new_row['ori_h'] = ori_h\n        new_rows.append(new_row)\n\n    # Tạo DataFrame từ các dòng mới\n    new_df = pd.DataFrame(new_rows)\n\n    return new_df\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:43.009271Z","iopub.execute_input":"2024-09-26T01:23:43.009601Z","iopub.status.idle":"2024-09-26T01:23:43.130881Z","shell.execute_reply.started":"2024-09-26T01:23:43.009572Z","shell.execute_reply":"2024-09-26T01:23:43.129992Z"},"trusted":true},"execution_count":24,"outputs":[]},{"cell_type":"code","source":"new5_df","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:43.131995Z","iopub.execute_input":"2024-09-26T01:23:43.132318Z","iopub.status.idle":"2024-09-26T01:23:43.160392Z","shell.execute_reply.started":"2024-09-26T01:23:43.132288Z","shell.execute_reply":"2024-09-26T01:23:43.159484Z"},"trusted":true},"execution_count":25,"outputs":[{"execution_count":25,"output_type":"execute_result","data":{"text/plain":"                                       image_id    class_name  class_id  \\\n59     80caa435b6ab5edaff4a0a758ffaec6e_rotated   Atelectasis         1   \n60     80caa435b6ab5edaff4a0a758ffaec6e_rotated   Atelectasis         1   \n61     80caa435b6ab5edaff4a0a758ffaec6e_rotated   Atelectasis         1   \n71     80caa435b6ab5edaff4a0a758ffaec6e_rotated   Atelectasis         1   \n103    c394eadea89e5795c8037280492d116d_rotated  Pneumothorax        12   \n...                                         ...           ...       ...   \n22400  2f3264d3c0a52bb2e280855bcfd35733_rotated   Atelectasis         1   \n22557  7253800842122c9e6e95878b46008f54_rotated   Atelectasis         1   \n22577  06efcd37617307118fd48b3a493c133b_rotated   Atelectasis         1   \n22578  06efcd37617307118fd48b3a493c133b_rotated   Atelectasis         1   \n22688  afa4de6570c31504ba4b77978377ccc9_rotated  Pneumothorax        12   \n\n       ori_x_min  ori_y_min  ori_x_max  ori_y_max  ori_x_mid  ori_y_mid  \\\n59          91.0      622.0      288.0      805.0        310        189   \n60         129.0      614.0      461.0      778.0        328        295   \n61         152.0      552.0      779.0      912.0        292        465   \n71         105.0      222.0      267.0      392.0        717        186   \n103        221.0      580.0      694.0      902.0        283        457   \n...          ...        ...        ...        ...        ...        ...   \n22400      519.0      248.0      546.0      303.0        748        532   \n22557      443.0      134.0      510.0      324.0        795        477   \n22577      685.0       70.0      712.0      126.0        926        698   \n22578      639.0       44.0      755.0      162.0        921        697   \n22688      157.0      547.0      616.0      788.0        357        386   \n\n       ori_w  ori_h                                           img_path  \n59       183    197  data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...  \n60       164    332  data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...  \n61       360    627  data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...  \n71       170    162  data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...  \n103      322    473  data/train/c394eadea89e5795c8037280492d116d_ro...  \n...      ...    ...                                                ...  \n22400     55     27  data/train/2f3264d3c0a52bb2e280855bcfd35733_ro...  \n22557    190     67  data/train/7253800842122c9e6e95878b46008f54_ro...  \n22577     56     27  data/train/06efcd37617307118fd48b3a493c133b_ro...  \n22578    118    116  data/train/06efcd37617307118fd48b3a493c133b_ro...  \n22688    241    459  data/train/afa4de6570c31504ba4b77978377ccc9_ro...  \n\n[354 rows x 12 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image_id</th>\n      <th>class_name</th>\n      <th>class_id</th>\n      <th>ori_x_min</th>\n      <th>ori_y_min</th>\n      <th>ori_x_max</th>\n      <th>ori_y_max</th>\n      <th>ori_x_mid</th>\n      <th>ori_y_mid</th>\n      <th>ori_w</th>\n      <th>ori_h</th>\n      <th>img_path</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>59</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>91.0</td>\n      <td>622.0</td>\n      <td>288.0</td>\n      <td>805.0</td>\n      <td>310</td>\n      <td>189</td>\n      <td>183</td>\n      <td>197</td>\n      <td>data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...</td>\n    </tr>\n    <tr>\n      <th>60</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>129.0</td>\n      <td>614.0</td>\n      <td>461.0</td>\n      <td>778.0</td>\n      <td>328</td>\n      <td>295</td>\n      <td>164</td>\n      <td>332</td>\n      <td>data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...</td>\n    </tr>\n    <tr>\n      <th>61</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>152.0</td>\n      <td>552.0</td>\n      <td>779.0</td>\n      <td>912.0</td>\n      <td>292</td>\n      <td>465</td>\n      <td>360</td>\n      <td>627</td>\n      <td>data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...</td>\n    </tr>\n    <tr>\n      <th>71</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>105.0</td>\n      <td>222.0</td>\n      <td>267.0</td>\n      <td>392.0</td>\n      <td>717</td>\n      <td>186</td>\n      <td>170</td>\n      <td>162</td>\n      <td>data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...</td>\n    </tr>\n    <tr>\n      <th>103</th>\n      <td>c394eadea89e5795c8037280492d116d_rotated</td>\n      <td>Pneumothorax</td>\n      <td>12</td>\n      <td>221.0</td>\n      <td>580.0</td>\n      <td>694.0</td>\n      <td>902.0</td>\n      <td>283</td>\n      <td>457</td>\n      <td>322</td>\n      <td>473</td>\n      <td>data/train/c394eadea89e5795c8037280492d116d_ro...</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>22400</th>\n      <td>2f3264d3c0a52bb2e280855bcfd35733_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>519.0</td>\n      <td>248.0</td>\n      <td>546.0</td>\n      <td>303.0</td>\n      <td>748</td>\n      <td>532</td>\n      <td>55</td>\n      <td>27</td>\n      <td>data/train/2f3264d3c0a52bb2e280855bcfd35733_ro...</td>\n    </tr>\n    <tr>\n      <th>22557</th>\n      <td>7253800842122c9e6e95878b46008f54_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>443.0</td>\n      <td>134.0</td>\n      <td>510.0</td>\n      <td>324.0</td>\n      <td>795</td>\n      <td>477</td>\n      <td>190</td>\n      <td>67</td>\n      <td>data/train/7253800842122c9e6e95878b46008f54_ro...</td>\n    </tr>\n    <tr>\n      <th>22577</th>\n      <td>06efcd37617307118fd48b3a493c133b_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>685.0</td>\n      <td>70.0</td>\n      <td>712.0</td>\n      <td>126.0</td>\n      <td>926</td>\n      <td>698</td>\n      <td>56</td>\n      <td>27</td>\n      <td>data/train/06efcd37617307118fd48b3a493c133b_ro...</td>\n    </tr>\n    <tr>\n      <th>22578</th>\n      <td>06efcd37617307118fd48b3a493c133b_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>639.0</td>\n      <td>44.0</td>\n      <td>755.0</td>\n      <td>162.0</td>\n      <td>921</td>\n      <td>697</td>\n      <td>118</td>\n      <td>116</td>\n      <td>data/train/06efcd37617307118fd48b3a493c133b_ro...</td>\n    </tr>\n    <tr>\n      <th>22688</th>\n      <td>afa4de6570c31504ba4b77978377ccc9_rotated</td>\n      <td>Pneumothorax</td>\n      <td>12</td>\n      <td>157.0</td>\n      <td>547.0</td>\n      <td>616.0</td>\n      <td>788.0</td>\n      <td>357</td>\n      <td>386</td>\n      <td>241</td>\n      <td>459</td>\n      <td>data/train/afa4de6570c31504ba4b77978377ccc9_ro...</td>\n    </tr>\n  </tbody>\n</table>\n<p>354 rows × 12 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"filtered_df = final_df[final_df['class_id'].isin([1, 12])].copy()\nnew1_df = flip_images_in_df(filtered_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:23:43.161606Z","iopub.execute_input":"2024-09-26T01:23:43.161991Z","iopub.status.idle":"2024-09-26T01:24:06.020329Z","shell.execute_reply.started":"2024-09-26T01:23:43.161964Z","shell.execute_reply":"2024-09-26T01:24:06.01943Z"},"trusted":true},"execution_count":26,"outputs":[{"name":"stderr","text":"100%|██████████| 354/354 [00:22<00:00, 15.51it/s]\n","output_type":"stream"}]},{"cell_type":"code","source":"import pandas as pd\nimport cv2\nimport numpy as np\nimport imgaug.augmenters as iaa\nfrom tqdm import tqdm\nimport os\n\ndef zoom_image(image_path, bbox, zoom_factor):\n    \"\"\"\n    Zoom ảnh và cập nhật bounding box.\n\n    Args:\n        image_path (str): Đường dẫn đến ảnh gốc.\n        bbox (list): Danh sách chứa các tọa độ bounding box [x_min, y_min, x_max, y_max].\n        zoom_factor (float): Hệ số zoom.\n\n    Returns:\n        tuple: Đường dẫn ảnh mới và bounding box đã được cập nhật.\n    \"\"\"\n    # Đọc ảnh gốc\n    image = cv2.imread(image_path)\n    if image is None:\n        return None, None\n\n    # Tăng cường zoom\n    augment_img_zoom = iaa.Affine(scale=(zoom_factor))\n    zoomed_image = augment_img_zoom.augment_image(image)\n\n    # Cập nhật bounding box\n    height, width = image.shape[:2]\n    new_width, new_height = int(width * zoom_factor), int(height * zoom_factor)\n\n    # Tính toán offset để điều chỉnh bounding box\n    offset_x = (new_width - width) / 2\n    offset_y = (new_height - height) / 2\n\n    new_bbox = [\n        bbox[0] * zoom_factor - offset_x,  # x_min\n        bbox[1] * zoom_factor - offset_y,  # y_min\n        bbox[2] * zoom_factor - offset_x,  # x_max\n        bbox[3] * zoom_factor - offset_y   # y_max\n    ]\n\n    # Lưu ảnh mới\n    new_image_path = os.path.splitext(image_path)[0] + '_zoom.png'\n    cv2.imwrite(new_image_path, zoomed_image)\n    \n    return new_image_path, new_bbox\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:24:06.021849Z","iopub.execute_input":"2024-09-26T01:24:06.022163Z","iopub.status.idle":"2024-09-26T01:24:06.356965Z","shell.execute_reply.started":"2024-09-26T01:24:06.022139Z","shell.execute_reply":"2024-09-26T01:24:06.356018Z"},"trusted":true},"execution_count":27,"outputs":[]},{"cell_type":"code","source":"def zoom_images_in_df(df, zoom_factor):\n    \"\"\"\n    Thực hiện zoom cho các ảnh thuộc lớp cụ thể và cập nhật thông tin bounding box trong DataFrame.\n\n    Args:\n        df (pd.DataFrame): DataFrame chứa thông tin về ảnh và bounding box.\n        zoom_factor (float): Hệ số zoom.\n\n    Returns:\n        pd.DataFrame: DataFrame đã được cập nhật với ảnh zoom và bounding box mới.\n    \"\"\"\n    new_rows = []\n\n    for index, row in tqdm(df.iterrows(), total=df.shape[0]):\n        image_path = row['img_path']\n        bbox = [row['ori_x_min'], row['ori_y_min'], row['ori_x_max'], row['ori_y_max']]\n        image_id = row['image_id']\n        class_name = row['class_name']\n        class_id = row['class_id']\n        ori_x_mid = row['ori_x_mid']\n        ori_y_mid = row['ori_y_mid']\n        ori_w = row['ori_w']\n        ori_h = row['ori_h']\n        \n        # Zoom ảnh và cập nhật bounding box\n        new_image_path, new_bbox = zoom_image(image_path, bbox, zoom_factor)\n        \n        if new_image_path is None:\n            continue\n        \n        # Tạo một dòng mới với thông số cập nhật\n        new_row = row.copy()\n        new_row['img_path'] = new_image_path\n        new_row['ori_x_min'], new_row['ori_y_min'], new_row['ori_x_max'], new_row['ori_y_max'] = new_bbox\n        new_row['image_id'] = image_id + '_zoom'\n        new_row['class_name'] = class_name\n        new_row['class_id'] = class_id\n        new_row['ori_x_mid'] = ori_x_mid\n        new_row['ori_y_mid'] = ori_y_mid\n        new_row['ori_w'] = ori_w\n        new_row['ori_h'] = ori_h\n        new_rows.append(new_row)\n\n    # Tạo DataFrame từ các dòng mới\n    new_df = pd.DataFrame(new_rows)\n\n    return new_df\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:24:06.3582Z","iopub.execute_input":"2024-09-26T01:24:06.358934Z","iopub.status.idle":"2024-09-26T01:24:06.368787Z","shell.execute_reply.started":"2024-09-26T01:24:06.358901Z","shell.execute_reply":"2024-09-26T01:24:06.367609Z"},"trusted":true},"execution_count":28,"outputs":[]},{"cell_type":"code","source":"filtered_df = final_df[final_df['class_id'].isin([1, 12])].copy()\n\n# Áp dụng hàm zoom ảnh cho DataFrame đã lọc với hệ số zoom là 10% (1.1)\nnew3_df = zoom_images_in_df(filtered_df, zoom_factor=1.1)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:24:06.369922Z","iopub.execute_input":"2024-09-26T01:24:06.370241Z","iopub.status.idle":"2024-09-26T01:24:30.910036Z","shell.execute_reply.started":"2024-09-26T01:24:06.370211Z","shell.execute_reply":"2024-09-26T01:24:30.909116Z"},"trusted":true},"execution_count":29,"outputs":[{"name":"stderr","text":"100%|██████████| 354/354 [00:24<00:00, 14.45it/s]\n","output_type":"stream"}]},{"cell_type":"code","source":"import cv2\nimport os\n\ndef clahe_image(image_list):\n    \"\"\"\n    Áp dụng CLAHE cho ảnh và lưu ảnh đã được tăng cường.\n\n    Args:\n        image_list (list): Danh sách các đường dẫn đến các ảnh cần áp dụng CLAHE.\n    \"\"\"\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    \n    for path in image_list:\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        if img is None:\n            continue\n        clahe_img = clahe.apply(img)\n        img_name = os.path.basename(path).split('.')[0]\n        new_image_path = f'{os.path.splitext(path)[0]}_clahe.png'\n        cv2.imwrite(new_image_path, clahe_img)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:24:30.911351Z","iopub.execute_input":"2024-09-26T01:24:30.91173Z","iopub.status.idle":"2024-09-26T01:24:30.918545Z","shell.execute_reply.started":"2024-09-26T01:24:30.911682Z","shell.execute_reply":"2024-09-26T01:24:30.917689Z"},"trusted":true},"execution_count":30,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\n\n# Lọc DataFrame chỉ chứa các lớp có class_id là 1 và 12\nfiltered_df = final_df[final_df['class_id'].isin([1, 12,4])].copy()\n\n# Tạo danh sách các đường dẫn ảnh\nimage_list = filtered_df['img_path'].tolist()\n\n# Thực hiện CLAHE cho các ảnh\nclahe_image(image_list)\n\n# Cập nhật DataFrame với ảnh mới đã áp dụng CLAHE\ndef update_df_with_clahe(df):\n    new_rows = []\n\n    for index, row in tqdm(df.iterrows(), total=df.shape[0]):\n        image_path = row['img_path']\n        img_name = os.path.basename(image_path).split('.')[0]\n        new_image_path = f'{os.path.splitext(image_path)[0]}_clahe.png'\n\n        # Tạo một dòng mới với đường dẫn ảnh CLAHE\n        new_row = row.copy()\n        new_row['img_path'] = new_image_path\n        new_row['image_id'] = row['image_id'] + '_clahe'\n        new_rows.append(new_row)\n\n    # Tạo DataFrame từ các dòng mới\n    new_df = pd.DataFrame(new_rows)\n    \n    return new_df\n\n# Áp dụng hàm cập nhật DataFrame\nnew4_df = update_df_with_clahe(filtered_df)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:24:30.919843Z","iopub.execute_input":"2024-09-26T01:24:30.920106Z","iopub.status.idle":"2024-09-26T01:24:58.346566Z","shell.execute_reply.started":"2024-09-26T01:24:30.920084Z","shell.execute_reply":"2024-09-26T01:24:58.345736Z"},"trusted":true},"execution_count":31,"outputs":[{"name":"stderr","text":"100%|██████████| 782/782 [00:00<00:00, 8052.13it/s]\n","output_type":"stream"}]},{"cell_type":"code","source":"import cv2\nimport os\n\ndef equ_image(image_list):\n    \"\"\"\n    Áp dụng EqualizeHist cho ảnh và lưu ảnh đã được tăng cường.\n\n    Args:\n        image_list (list): Danh sách các đường dẫn đến các ảnh cần áp dụng EqualizeHist.\n    \"\"\"\n    for path in image_list:\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        if img is None:\n            continue\n        equ_img = cv2.equalizeHist(img)\n        img_name = os.path.basename(path).split('.')[0]\n        new_image_path = f'{os.path.splitext(path)[0]}_equ.png'\n        cv2.imwrite(new_image_path, equ_img)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:24:58.35333Z","iopub.execute_input":"2024-09-26T01:24:58.353646Z","iopub.status.idle":"2024-09-26T01:24:58.359575Z","shell.execute_reply.started":"2024-09-26T01:24:58.35362Z","shell.execute_reply":"2024-09-26T01:24:58.358649Z"},"trusted":true},"execution_count":32,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\n\n# Lọc DataFrame chỉ chứa các lớp có class_id là 1 và 12\nfiltered_df = final_df[final_df['class_id'].isin([1, 12,4,5,2,6])].copy()\n\n# Tạo danh sách các đường dẫn ảnh\nimage_list = filtered_df['img_path'].tolist()\n\n# Thực hiện EqualizeHist cho các ảnh\nequ_image(image_list)\n\n# Cập nhật DataFrame với ảnh mới đã áp dụng EqualizeHist\ndef update_df_with_equ(df):\n    new_rows = []\n\n    for index, row in tqdm(df.iterrows(), total=df.shape[0]):\n        image_path = row['img_path']\n        img_name = os.path.basename(image_path).split('.')[0]\n        new_image_path = f'{os.path.splitext(image_path)[0]}_equ.png'\n\n        # Tạo một dòng mới với đường dẫn ảnh EqualizeHist\n        new_row = row.copy()\n        new_row['img_path'] = new_image_path\n        new_row['image_id'] = row['image_id'] + '_equ'\n        new_rows.append(new_row)\n\n    # Tạo DataFrame từ các dòng mới\n    new_df = pd.DataFrame(new_rows)\n    \n    return new_df\n\n# Áp dụng hàm cập nhật DataFrame\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:24:58.360773Z","iopub.execute_input":"2024-09-26T01:24:58.361133Z","iopub.status.idle":"2024-09-26T01:26:42.998456Z","shell.execute_reply.started":"2024-09-26T01:24:58.361109Z","shell.execute_reply":"2024-09-26T01:26:42.996884Z"},"trusted":true},"execution_count":33,"outputs":[]},{"cell_type":"code","source":"new2_df=update_df_with_equ(filtered_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:26:43.000117Z","iopub.execute_input":"2024-09-26T01:26:43.000508Z","iopub.status.idle":"2024-09-26T01:26:43.781281Z","shell.execute_reply.started":"2024-09-26T01:26:43.000475Z","shell.execute_reply":"2024-09-26T01:26:43.780492Z"},"trusted":true},"execution_count":34,"outputs":[{"name":"stderr","text":"100%|██████████| 3114/3114 [00:00<00:00, 7960.19it/s]\n","output_type":"stream"}]},{"cell_type":"code","source":"new_train_df = pd.concat([new5_df, new1_df, new2_df, new4_df,new3_df, final_df], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:26:43.782355Z","iopub.execute_input":"2024-09-26T01:26:43.78265Z","iopub.status.idle":"2024-09-26T01:26:43.789617Z","shell.execute_reply.started":"2024-09-26T01:26:43.782625Z","shell.execute_reply":"2024-09-26T01:26:43.788609Z"},"trusted":true},"execution_count":35,"outputs":[]},{"cell_type":"code","source":"new_train_df","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:26:43.790871Z","iopub.execute_input":"2024-09-26T01:26:43.791143Z","iopub.status.idle":"2024-09-26T01:26:43.812985Z","shell.execute_reply.started":"2024-09-26T01:26:43.791121Z","shell.execute_reply":"2024-09-26T01:26:43.812222Z"},"trusted":true},"execution_count":36,"outputs":[{"execution_count":36,"output_type":"execute_result","data":{"text/plain":"                                       image_id          class_name  class_id  \\\n0      80caa435b6ab5edaff4a0a758ffaec6e_rotated         Atelectasis         1   \n1      80caa435b6ab5edaff4a0a758ffaec6e_rotated         Atelectasis         1   \n2      80caa435b6ab5edaff4a0a758ffaec6e_rotated         Atelectasis         1   \n3      80caa435b6ab5edaff4a0a758ffaec6e_rotated         Atelectasis         1   \n4      c394eadea89e5795c8037280492d116d_rotated        Pneumothorax        12   \n...                                         ...                 ...       ...   \n27671          be53fe5a49231f1c1be020b0bdd8561f        Lung Opacity         7   \n27672          380d07a94cc4b012812119370de47192  Aortic enlargement         0   \n27673          52951d7de2485aba8ed62629eee4d254        Cardiomegaly         3   \n27674          52951d7de2485aba8ed62629eee4d254        Other lesion         9   \n27675          1224f07d895107573588225f692e94f9  Aortic enlargement         0   \n\n       ori_x_min  ori_y_min  ori_x_max  ori_y_max  ori_x_mid  ori_y_mid  \\\n0           91.0      622.0      288.0      805.0        310        189   \n1          129.0      614.0      461.0      778.0        328        295   \n2          152.0      552.0      779.0      912.0        292        465   \n3          105.0      222.0      267.0      392.0        717        186   \n4          221.0      580.0      694.0      902.0        283        457   \n...          ...        ...        ...        ...        ...        ...   \n27671      216.0      388.0      282.0      451.0        249        419   \n27672      576.0      284.0      709.0      398.0        642        341   \n27673      321.0      573.0      718.0      680.0        519        626   \n27674      134.0      512.0      170.0      536.0        152        524   \n27675      515.0      314.0      638.0      441.0        576        377   \n\n       ori_w  ori_h                                           img_path  \n0        183    197  data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...  \n1        164    332  data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...  \n2        360    627  data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...  \n3        170    162  data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...  \n4        322    473  data/train/c394eadea89e5795c8037280492d116d_ro...  \n...      ...    ...                                                ...  \n27671     66     63    data/train/be53fe5a49231f1c1be020b0bdd8561f.png  \n27672    133    114    data/train/380d07a94cc4b012812119370de47192.png  \n27673    397    107    data/train/52951d7de2485aba8ed62629eee4d254.png  \n27674     36     24    data/train/52951d7de2485aba8ed62629eee4d254.png  \n27675    123    127    data/train/1224f07d895107573588225f692e94f9.png  \n\n[27676 rows x 12 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image_id</th>\n      <th>class_name</th>\n      <th>class_id</th>\n      <th>ori_x_min</th>\n      <th>ori_y_min</th>\n      <th>ori_x_max</th>\n      <th>ori_y_max</th>\n      <th>ori_x_mid</th>\n      <th>ori_y_mid</th>\n      <th>ori_w</th>\n      <th>ori_h</th>\n      <th>img_path</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>91.0</td>\n      <td>622.0</td>\n      <td>288.0</td>\n      <td>805.0</td>\n      <td>310</td>\n      <td>189</td>\n      <td>183</td>\n      <td>197</td>\n      <td>data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>129.0</td>\n      <td>614.0</td>\n      <td>461.0</td>\n      <td>778.0</td>\n      <td>328</td>\n      <td>295</td>\n      <td>164</td>\n      <td>332</td>\n      <td>data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>152.0</td>\n      <td>552.0</td>\n      <td>779.0</td>\n      <td>912.0</td>\n      <td>292</td>\n      <td>465</td>\n      <td>360</td>\n      <td>627</td>\n      <td>data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>Atelectasis</td>\n      <td>1</td>\n      <td>105.0</td>\n      <td>222.0</td>\n      <td>267.0</td>\n      <td>392.0</td>\n      <td>717</td>\n      <td>186</td>\n      <td>170</td>\n      <td>162</td>\n      <td>data/train/80caa435b6ab5edaff4a0a758ffaec6e_ro...</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>c394eadea89e5795c8037280492d116d_rotated</td>\n      <td>Pneumothorax</td>\n      <td>12</td>\n      <td>221.0</td>\n      <td>580.0</td>\n      <td>694.0</td>\n      <td>902.0</td>\n      <td>283</td>\n      <td>457</td>\n      <td>322</td>\n      <td>473</td>\n      <td>data/train/c394eadea89e5795c8037280492d116d_ro...</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>27671</th>\n      <td>be53fe5a49231f1c1be020b0bdd8561f</td>\n      <td>Lung Opacity</td>\n      <td>7</td>\n      <td>216.0</td>\n      <td>388.0</td>\n      <td>282.0</td>\n      <td>451.0</td>\n      <td>249</td>\n      <td>419</td>\n      <td>66</td>\n      <td>63</td>\n      <td>data/train/be53fe5a49231f1c1be020b0bdd8561f.png</td>\n    </tr>\n    <tr>\n      <th>27672</th>\n      <td>380d07a94cc4b012812119370de47192</td>\n      <td>Aortic enlargement</td>\n      <td>0</td>\n      <td>576.0</td>\n      <td>284.0</td>\n      <td>709.0</td>\n      <td>398.0</td>\n      <td>642</td>\n      <td>341</td>\n      <td>133</td>\n      <td>114</td>\n      <td>data/train/380d07a94cc4b012812119370de47192.png</td>\n    </tr>\n    <tr>\n      <th>27673</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>Cardiomegaly</td>\n      <td>3</td>\n      <td>321.0</td>\n      <td>573.0</td>\n      <td>718.0</td>\n      <td>680.0</td>\n      <td>519</td>\n      <td>626</td>\n      <td>397</td>\n      <td>107</td>\n      <td>data/train/52951d7de2485aba8ed62629eee4d254.png</td>\n    </tr>\n    <tr>\n      <th>27674</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>Other lesion</td>\n      <td>9</td>\n      <td>134.0</td>\n      <td>512.0</td>\n      <td>170.0</td>\n      <td>536.0</td>\n      <td>152</td>\n      <td>524</td>\n      <td>36</td>\n      <td>24</td>\n      <td>data/train/52951d7de2485aba8ed62629eee4d254.png</td>\n    </tr>\n    <tr>\n      <th>27675</th>\n      <td>1224f07d895107573588225f692e94f9</td>\n      <td>Aortic enlargement</td>\n      <td>0</td>\n      <td>515.0</td>\n      <td>314.0</td>\n      <td>638.0</td>\n      <td>441.0</td>\n      <td>576</td>\n      <td>377</td>\n      <td>123</td>\n      <td>127</td>\n      <td>data/train/1224f07d895107573588225f692e94f9.png</td>\n    </tr>\n  </tbody>\n</table>\n<p>27676 rows × 12 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"import pandas as pd\n\n# Giả sử new_train_df đã tồn tại\n# Ví dụ:\n# new_train_df = pd.DataFrame(...)\n\n# Tạo một bản sao của new_train_df để xử lý\nnew_df = new_train_df.copy()\n\n# Đổi tên các cột liên quan\nnew_df.rename(columns={\n    'ori_x_min': 'x_min',\n    'ori_y_min': 'y_min',\n    'ori_x_max': 'x_max',\n    'ori_y_max': 'y_max'\n}, inplace=True)\n\n# Chuyển đổi giá trị các cột x_min, x_max, y_min, y_max\nnew_df['x_min'] = new_df['x_min'] / 1024\nnew_df['y_min'] = new_df['y_min'] / 1024\nnew_df['x_max'] = new_df['x_max'] / 1024\nnew_df['y_max'] = new_df['y_max'] / 1024\n\n# Chỉ giữ lại các cột x_min, x_max, y_min, y_max\nnew_df = new_df[['image_id','class_id','class_name','x_min', 'x_max', 'y_min', 'y_max']]\nnew_df['x_center'] = (new_df['x_max'] + new_df['x_min']) / 2\nnew_df['y_center'] = (new_df['y_max'] + new_df['y_min']) / 2\n\nnew_df['w'] = (new_df['x_max'] - new_df['x_min'])\nnew_df['h'] = (new_df['y_max'] - new_df['y_min'])\n# Hiển thị kết quả\nnew_df\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:26:43.814132Z","iopub.execute_input":"2024-09-26T01:26:43.814393Z","iopub.status.idle":"2024-09-26T01:26:43.848162Z","shell.execute_reply.started":"2024-09-26T01:26:43.814371Z","shell.execute_reply":"2024-09-26T01:26:43.847289Z"},"trusted":true},"execution_count":37,"outputs":[{"execution_count":37,"output_type":"execute_result","data":{"text/plain":"                                       image_id  class_id          class_name  \\\n0      80caa435b6ab5edaff4a0a758ffaec6e_rotated         1         Atelectasis   \n1      80caa435b6ab5edaff4a0a758ffaec6e_rotated         1         Atelectasis   \n2      80caa435b6ab5edaff4a0a758ffaec6e_rotated         1         Atelectasis   \n3      80caa435b6ab5edaff4a0a758ffaec6e_rotated         1         Atelectasis   \n4      c394eadea89e5795c8037280492d116d_rotated        12        Pneumothorax   \n...                                         ...       ...                 ...   \n27671          be53fe5a49231f1c1be020b0bdd8561f         7        Lung Opacity   \n27672          380d07a94cc4b012812119370de47192         0  Aortic enlargement   \n27673          52951d7de2485aba8ed62629eee4d254         3        Cardiomegaly   \n27674          52951d7de2485aba8ed62629eee4d254         9        Other lesion   \n27675          1224f07d895107573588225f692e94f9         0  Aortic enlargement   \n\n          x_min     x_max     y_min     y_max  x_center  y_center         w  \\\n0      0.088867  0.281250  0.607422  0.786133  0.185059  0.696777  0.192383   \n1      0.125977  0.450195  0.599609  0.759766  0.288086  0.679688  0.324219   \n2      0.148437  0.760742  0.539062  0.890625  0.454590  0.714844  0.612305   \n3      0.102539  0.260742  0.216797  0.382812  0.181641  0.299805  0.158203   \n4      0.215820  0.677734  0.566406  0.880859  0.446777  0.723633  0.461914   \n...         ...       ...       ...       ...       ...       ...       ...   \n27671  0.210938  0.275391  0.378906  0.440430  0.243164  0.409668  0.064453   \n27672  0.562500  0.692383  0.277344  0.388672  0.627441  0.333008  0.129883   \n27673  0.313477  0.701172  0.559570  0.664062  0.507324  0.611816  0.387695   \n27674  0.130859  0.166016  0.500000  0.523438  0.148438  0.511719  0.035156   \n27675  0.502930  0.623047  0.306641  0.430664  0.562988  0.368652  0.120117   \n\n              h  \n0      0.178711  \n1      0.160156  \n2      0.351562  \n3      0.166016  \n4      0.314453  \n...         ...  \n27671  0.061523  \n27672  0.111328  \n27673  0.104492  \n27674  0.023438  \n27675  0.124023  \n\n[27676 rows x 11 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>image_id</th>\n      <th>class_id</th>\n      <th>class_name</th>\n      <th>x_min</th>\n      <th>x_max</th>\n      <th>y_min</th>\n      <th>y_max</th>\n      <th>x_center</th>\n      <th>y_center</th>\n      <th>w</th>\n      <th>h</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>1</td>\n      <td>Atelectasis</td>\n      <td>0.088867</td>\n      <td>0.281250</td>\n      <td>0.607422</td>\n      <td>0.786133</td>\n      <td>0.185059</td>\n      <td>0.696777</td>\n      <td>0.192383</td>\n      <td>0.178711</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>1</td>\n      <td>Atelectasis</td>\n      <td>0.125977</td>\n      <td>0.450195</td>\n      <td>0.599609</td>\n      <td>0.759766</td>\n      <td>0.288086</td>\n      <td>0.679688</td>\n      <td>0.324219</td>\n      <td>0.160156</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>1</td>\n      <td>Atelectasis</td>\n      <td>0.148437</td>\n      <td>0.760742</td>\n      <td>0.539062</td>\n      <td>0.890625</td>\n      <td>0.454590</td>\n      <td>0.714844</td>\n      <td>0.612305</td>\n      <td>0.351562</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>80caa435b6ab5edaff4a0a758ffaec6e_rotated</td>\n      <td>1</td>\n      <td>Atelectasis</td>\n      <td>0.102539</td>\n      <td>0.260742</td>\n      <td>0.216797</td>\n      <td>0.382812</td>\n      <td>0.181641</td>\n      <td>0.299805</td>\n      <td>0.158203</td>\n      <td>0.166016</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>c394eadea89e5795c8037280492d116d_rotated</td>\n      <td>12</td>\n      <td>Pneumothorax</td>\n      <td>0.215820</td>\n      <td>0.677734</td>\n      <td>0.566406</td>\n      <td>0.880859</td>\n      <td>0.446777</td>\n      <td>0.723633</td>\n      <td>0.461914</td>\n      <td>0.314453</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>27671</th>\n      <td>be53fe5a49231f1c1be020b0bdd8561f</td>\n      <td>7</td>\n      <td>Lung Opacity</td>\n      <td>0.210938</td>\n      <td>0.275391</td>\n      <td>0.378906</td>\n      <td>0.440430</td>\n      <td>0.243164</td>\n      <td>0.409668</td>\n      <td>0.064453</td>\n      <td>0.061523</td>\n    </tr>\n    <tr>\n      <th>27672</th>\n      <td>380d07a94cc4b012812119370de47192</td>\n      <td>0</td>\n      <td>Aortic enlargement</td>\n      <td>0.562500</td>\n      <td>0.692383</td>\n      <td>0.277344</td>\n      <td>0.388672</td>\n      <td>0.627441</td>\n      <td>0.333008</td>\n      <td>0.129883</td>\n      <td>0.111328</td>\n    </tr>\n    <tr>\n      <th>27673</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>3</td>\n      <td>Cardiomegaly</td>\n      <td>0.313477</td>\n      <td>0.701172</td>\n      <td>0.559570</td>\n      <td>0.664062</td>\n      <td>0.507324</td>\n      <td>0.611816</td>\n      <td>0.387695</td>\n      <td>0.104492</td>\n    </tr>\n    <tr>\n      <th>27674</th>\n      <td>52951d7de2485aba8ed62629eee4d254</td>\n      <td>9</td>\n      <td>Other lesion</td>\n      <td>0.130859</td>\n      <td>0.166016</td>\n      <td>0.500000</td>\n      <td>0.523438</td>\n      <td>0.148438</td>\n      <td>0.511719</td>\n      <td>0.035156</td>\n      <td>0.023438</td>\n    </tr>\n    <tr>\n      <th>27675</th>\n      <td>1224f07d895107573588225f692e94f9</td>\n      <td>0</td>\n      <td>Aortic enlargement</td>\n      <td>0.502930</td>\n      <td>0.623047</td>\n      <td>0.306641</td>\n      <td>0.430664</td>\n      <td>0.562988</td>\n      <td>0.368652</td>\n      <td>0.120117</td>\n      <td>0.124023</td>\n    </tr>\n  </tbody>\n</table>\n<p>27676 rows × 11 columns</p>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"new_df['class_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:26:43.84921Z","iopub.execute_input":"2024-09-26T01:26:43.849479Z","iopub.status.idle":"2024-09-26T01:26:43.857986Z","shell.execute_reply.started":"2024-09-26T01:26:43.849457Z","shell.execute_reply":"2024-09-26T01:26:43.857011Z"},"trusted":true},"execution_count":38,"outputs":[{"execution_count":38,"output_type":"execute_result","data":{"text/plain":"class_id\n11    3793\n0     3203\n13    3125\n3     2336\n7     1941\n6     1828\n8     1826\n9     1776\n10    1604\n2     1454\n5     1382\n1     1356\n4     1284\n12     768\nName: count, dtype: int64"},"metadata":{}}]},{"cell_type":"code","source":"# ncount = len(new_df) #sample의 개수\n# plt.figure(figsize = (12,8))\n# #Seaborn으로 class name기준으로 bar chart 그리기\n# ax = sns.countplot(x = 'class_name', data = new_df, order = new_df.class_name.value_counts().index)\n# plt.title('Distribution of class names')\n# ax.set_xticklabels(ax.get_xticklabels(), rotation = 45, ha ='right')\n# #x축만 공유하고 y축은 따로 쓰는 'twinx()'\n# ax2 = ax.twinx()\n# # tick: 축에 간격을 구분하기 위해 표시하는 눈금\n# # ax는 오른쪽, ax2는 왼쪽\n# ax2.yaxis.tick_left() \n# ax.yaxis.tick_right() \n\n# ax.yaxis.set_label_position('right')\n# ax2.yaxis.set_label_position('left')\n\n# ax2.set_ylabel('Frequency [%]')\n# # patches 모듈은 (x, y, width, height)의 형태로 도형으로 시각화 하는 방법\n# for p in ax.patches:\n#   x = p.get_bbox().get_points()[:,0] # get_points: bbox의 점을 [[x0,y0],[x1,y1]]형식의 numpy배열로 직접가져옴\n#   y = p.get_bbox().get_points()[1,1]\n#   # annotate: 주석달기. (주석 내용, 좌표는 필수로 전달해야함)\n#   # ha = horizontal alignmnet, va = vertical alignment\n#   ax.annotate('{:.1f}%'.format(100.*y/ncount), (x.mean(), y), ha = 'center', va = 'bottom')\n\n# # linearlocator: min에서 max까지 균일 분포 눈금.\n# ax.yaxis.set_major_locator(ticker.LinearLocator(11))\n# # y값의 limitation\n# ax2.set_ylim(0,100)\n# ax.set_ylim(0,ncount)\n# # multiplelocator: 눈금과 범위는 모두 base의 배수\n# ax2.yaxis.set_major_locator(ticker.MultipleLocator(10))\n\n# ax2.grid(None)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:26:43.859257Z","iopub.execute_input":"2024-09-26T01:26:43.859602Z","iopub.status.idle":"2024-09-26T01:26:43.867246Z","shell.execute_reply.started":"2024-09-26T01:26:43.859571Z","shell.execute_reply":"2024-09-26T01:26:43.866556Z"},"trusted":true},"execution_count":39,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport shutil\nimport yaml\nfrom sklearn.model_selection import GroupShuffleSplit\nfrom tqdm import tqdm\n\n# Training/validation splitting\nsplitter = GroupShuffleSplit(test_size=0.1)\nsplit = splitter.split(new_df, groups=new_df['image_id'])\ntrain_inds, valid_inds = next(split)\n\nvalid_df = new_df.iloc[valid_inds]\ntrain_df = new_df.iloc[train_inds]\n\n# Create images and labels directories for both training and validation\nfor folder in [\n    TRAIN_IMAGES_PATH_REFACTORED,\n    TRAIN_LABELS_PATH_REFACTORED,\n    VALID_IMAGES_PATH_REFACTORED,\n    VALID_LABELS_PATH_REFACTORED,\n]:\n    os.makedirs(folder, exist_ok=True)\n\n# Copy training images to its designated directory, and create a txt file for each image denoting its classes and their positions\nfor image in tqdm(train_df['image_id'].unique()):\n    records = train_df[train_df['image_id'] == image]\n    attributes = records[['class_id', 'x_center', 'y_center', 'w', 'h']].values\n    np.savetxt(\n        os.path.join(TRAIN_LABELS_PATH_REFACTORED, f'{image}.txt'),\n        attributes,\n        fmt=['%d', '%f', '%f', '%f', '%f']\n    )\n    shutil.copy(\n        os.path.join(TRAIN_IMAGES_DIRECTORY, f'{image}.png'),\n        TRAIN_IMAGES_PATH_REFACTORED\n    )\n\n# Copy validation images to its designated directory, and create a txt file for each image denoting its classes and their positions\nfor image in tqdm(valid_df['image_id'].unique()):\n    records = valid_df[valid_df['image_id'] == image]\n    attributes = records[['class_id', 'x_center', 'y_center', 'w', 'h']].values\n    np.savetxt(\n        os.path.join(VALID_LABELS_PATH_REFACTORED, f'{image}.txt'),\n        attributes,\n        fmt=['%d', '%f', '%f', '%f', '%f']\n    )\n    shutil.copy(\n        os.path.join(TRAIN_IMAGES_DIRECTORY, f'{image}.png'),\n        VALID_IMAGES_PATH_REFACTORED\n    )\n\n# Order the classes based on the class ID numerical value (in an ascending order)\nclass_ids, class_names = list(zip(*set(zip(new_df['class_id'], new_df['class_name']))))\nclasses = list(np.array(class_names)[np.argsort(class_ids)])\nclasses = list(map(lambda x: str(x), classes))\n\n# Store a list containing the path of each training image in a TXT file\nwith open('train.txt', 'w') as f:\n    for path in os.listdir(TRAIN_IMAGES_PATH_REFACTORED):\n        f.write(f'{TRAIN_IMAGES_PATH_REFACTORED}{path}\\n')\n\n# Store a list containing the path of each validation image in a TXT file\nwith open('valid.txt', 'w') as f:\n    for path in os.listdir(VALID_IMAGES_PATH_REFACTORED):\n        f.write(f'{VALID_IMAGES_PATH_REFACTORED}{path}\\n')\n\n# Create a dictionary containing the necessary configurations to run YOLO\ndata = dict(\n    train='train.txt',\n    val='valid.txt',\n    nc=14,  # Assuming 14 classes, adjust if needed\n    names=classes\n)\n\n# Store the configurations in a YAML file, to be absorbed later by YOLO\nwith open('yolo.yaml', 'w') as outfile:\n    yaml.dump(data, outfile, default_flow_style=False)\n\n# Return train dataframe and the original dataframe loaded from a CSV (assuming the CSV path is '/kaggle/working/data/train.csv')\n# return [pd.read_csv('/kaggle/working/data/train.csv'), train_df]\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:26:43.868526Z","iopub.execute_input":"2024-09-26T01:26:43.869032Z","iopub.status.idle":"2024-09-26T01:27:31.331855Z","shell.execute_reply.started":"2024-09-26T01:26:43.868998Z","shell.execute_reply":"2024-09-26T01:27:31.330856Z"},"trusted":true},"execution_count":40,"outputs":[{"name":"stderr","text":"100%|██████████| 6547/6547 [00:45<00:00, 145.31it/s]\n100%|██████████| 728/728 [00:02<00:00, 313.76it/s]\n","output_type":"stream"}]},{"cell_type":"code","source":"# def train_model(\n#         train_step: int = 2,  # Chọn lần huấn luyện (1 cho lần 1, 2 cho lần 2)\n#         visualize: bool = False,\n#         export_as_onnx: bool = False,\n#         export_as_tf: bool = False,\n#         export_as_tflite: bool = False,\n#         conf_thres: float = 0.3,\n#         iou_thres: float = 0.4,\n# ):\n\n\n#     if train_step == 1:\n#         # Train lần 1\n#         print(\"Training lần 1...\")\n#         !torchrun --nproc_per_node=2 --master_port 9527 /kaggle/working/yolov9/train_dual.py --device 0,1 --sync-bn --img {IMAGE_SIZE} --cfg /kaggle/working/yolov9/models/detect/yolov9-c.yaml --batch-size 8 --epochs 10 --data yolo.yaml --weights /kaggle/working/yolov9-c.pt --hyp /kaggle/working/yolov9/data/hyps/hyp.scratch-high.yaml --min-items 0 --close-mosaic 15\n\n#         # Lưu trọng số sau lần train 1 vào /kaggle/outputs\n#         print(\"Lưu mô hình sau lần 1...\")\n#         !cp /kaggle/working/yolov9/runs/train/exp/weights/last.pt /kaggle/working/yolov9-c-lan1.pt\n\n#         print(\"Đã hoàn thành lần huấn luyện 1. Lưu mô hình và dừng lại.\")\n\n#         return  # Dừng lại ngay sau khi huấn luyện lần 1 hoàn thành\n\n#     elif train_step == 2:\n#         # Load mô hình từ lần train 1 và tiếp tục train lần 2\n#         print(\"Training lần 2...\")\n#         !torchrun --nproc_per_node=2 --master_port 9527 /kaggle/working/yolov9/train_dual.py --device 0,1 --sync-bn --img {IMAGE_SIZE} --cfg /kaggle/working/yolov9/models/detect/yolov9-c.yaml --batch-size 8 --epochs 20 --data yolo.yaml --weights /kaggle/working/last.pt \n\n#     if visualize:\n#         # Visualize các kết quả sau lần huấn luyện\n#         plt.rcParams.update({\n#             'figure.figsize': (15, 8),\n#             'axes.spines.left': False,\n#             'axes.spines.right': False,\n#             'axes.spines.bottom': False,\n#             'axes.spines.top': False,\n#             'xtick.bottom': False,\n#             'xtick.labelbottom': False,\n#             'ytick.labelleft': False,\n#             'ytick.left': False,\n#         })\n\n#         for img_name in ['results.png', 'PR_curve.png', 'confusion_matrix.png']:\n#             img_path = f'runs/train/exp/{img_name}'\n#             plt.figure(figsize=(15, 8))\n#             plt.imshow(plt.imread(img_path))\n#             plt.title(f\"{img_name} (conf_thres={conf_thres}, iou_thres={iou_thres})\")\n#             plt.axis('off')\n\n#         plt.rcParams.update(plt.rcParamsDefault)\n#         plt.rcParams.update({'figure.figsize': (15, 8)})\n\n#     if export_as_onnx or export_as_tf or export_as_tflite:\n#         print(\"Exporting mô hình...\")\n#         !python yolov9/export.py --weights yolov9-c.pt --img-size {IMAGE_SIZE} {IMAGE_SIZE} --max-wh {IMAGE_SIZE} --grid --end2end --simplify\n\n#         if export_as_tf or export_as_tflite:\n#             !onnx-tf convert -i yolov9.onnx -o ./\n\n#             if export_as_tflite:\n#                 converter = tf.lite.TFLiteConverter.from_saved_model('./')\n#                 tflite_model = converter.convert()\n#                 with open('yolov9.tflite', 'wb') as f:\n#                     f.write(tflite_model)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:27:31.333035Z","iopub.execute_input":"2024-09-26T01:27:31.333326Z","iopub.status.idle":"2024-09-26T01:27:31.340333Z","shell.execute_reply.started":"2024-09-26T01:27:31.333301Z","shell.execute_reply":"2024-09-26T01:27:31.339411Z"},"trusted":true},"execution_count":41,"outputs":[]},{"cell_type":"code","source":"# def train_model(\n#         train_step: int = 1,\n#         visualize: bool = True,\n#         export_as_onnx: bool = True,\n#         export_as_tf: bool = True,\n#         export_as_tflite: bool = True,\n# ):\n#     if train_step == 1:\n#         # Train lần 1\n#         print(\"Training lần 1...\")\n#         !python /kaggle/working/yolov7/train.py --workers 0 --device 0 --img {IMAGE_SIZE} --batch-size 20 --epochs 30 --data /kaggle/working/yolo.yaml --weights /kaggle/working/yolov7.pt --hyp /kaggle/working/yolov7/data/hyp.scratch.p5.yaml\n#         # Lưu trọng số sau lần train 1 vào /kaggle/outputs\n#         print(\"Lưu mô hình sau lần 1...\")\n\n#         print(\"Đã hoàn thành lần huấn luyện 1. Lưu mô hình và dừng lại.\")\n\n#         return  # Dừng lại ngay sau khi huấn luyện lần 1 hoàn thành\n\n#     elif train_step == 2:\n#         # Load mô hình từ lần train 1 và tiếp tục train lần 2\n#         print(\"Training lần 2...\")\n#         !torchrun --nproc_per_node=2 --master_port 9527 /kaggle/working/yolov9/train_dual.py --device 0,1 --sync-bn --img {IMAGE_SIZE} --cfg /kaggle/working/yolov9/models/detect/yolov9-c.yaml --batch-size 16 --epochs 40 --data yolo.yaml --weights /kaggle/working/last.pt \n\n\n#     if visualize:\n#         # Modify matplotlib figure size, and remove axis lines and ticks\n#         plt.rcParams.update({\n#             'figure.figsize': (15, 8),\n#             'axes.spines.left': False,\n#             'axes.spines.right': False,\n#             'axes.spines.bottom': False,\n#             'axes.spines.top': False,\n#             'xtick.bottom': False,\n#             'xtick.labelbottom': False,\n#             'ytick.labelleft': False,\n#             'ytick.left': False,\n#         })\n\n#         plt.imshow(plt.imread('runs/train/exp/results.png'))\n#         plt.imshow(plt.imread('runs/train/exp/PR_curve.png'))\n#         plt.imshow(plt.imread('runs/train/exp/confusion_matrix.png'))\n\n#         plt.rcParams.update(plt.rcParamsDefault)\n#         plt.rcParams.update({'figure.figsize': (15, 8)})\n\n#     if export_as_onnx or export_as_tf or export_as_tflite:\n#         !python yolov7/export.py --weights yolov7.pt --img-size {IMAGE_SIZE} {IMAGE_SIZE} --max-wh {IMAGE_SIZE} --grid --end2end --simplify\n\n#         if export_as_tf or export_as_tflite:\n#             !onnx-tf convert -i yolov7.onnx -o ./\n\n#             if export_as_tflite:\n#                 converter = tf.lite.TFLiteConverter.from_saved_model('./')\n#                 tflite_model = converter.convert()\n#                 with open('yolov7.tflite', 'wb') as f:\n#                     f.write(tflite_model)","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:27:31.341837Z","iopub.execute_input":"2024-09-26T01:27:31.342164Z","iopub.status.idle":"2024-09-26T01:27:31.358403Z","shell.execute_reply.started":"2024-09-26T01:27:31.342135Z","shell.execute_reply":"2024-09-26T01:27:31.357623Z"},"trusted":true},"execution_count":42,"outputs":[]},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:27:31.359536Z","iopub.execute_input":"2024-09-26T01:27:31.360052Z","iopub.status.idle":"2024-09-26T01:27:45.074244Z","shell.execute_reply.started":"2024-09-26T01:27:31.36002Z","shell.execute_reply":"2024-09-26T01:27:45.073219Z"},"trusted":true},"execution_count":43,"outputs":[{"name":"stdout","text":"Collecting ultralytics\n  Downloading ultralytics-8.2.101-py3-none-any.whl.metadata (39 kB)\nRequirement already satisfied: numpy<2.0.0,>=1.23.0 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (1.26.4)\nRequirement already satisfied: matplotlib>=3.3.0 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (3.7.5)\nRequirement already satisfied: opencv-python>=4.6.0 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (4.9.0.80)\nRequirement already satisfied: pillow>=7.1.2 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (9.5.0)\nRequirement already satisfied: pyyaml>=5.3.1 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (6.0.1)\nRequirement already satisfied: requests>=2.23.0 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (2.31.0)\nRequirement already satisfied: scipy>=1.4.1 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (1.11.4)\nRequirement already satisfied: torch>=1.8.0 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (2.1.2)\nRequirement already satisfied: torchvision>=0.9.0 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (0.16.2)\nRequirement already satisfied: tqdm>=4.64.0 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (4.66.1)\nRequirement already satisfied: psutil in /opt/conda/lib/python3.10/site-packages (from ultralytics) (5.9.3)\nRequirement already satisfied: py-cpuinfo in /opt/conda/lib/python3.10/site-packages (from ultralytics) (9.0.0)\nRequirement already satisfied: pandas>=1.1.4 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (2.1.4)\nRequirement already satisfied: seaborn>=0.11.0 in /opt/conda/lib/python3.10/site-packages (from ultralytics) (0.12.2)\nCollecting ultralytics-thop>=2.0.0 (from ultralytics)\n  Downloading ultralytics_thop-2.0.8-py3-none-any.whl.metadata (9.3 kB)\nRequirement already satisfied: contourpy>=1.0.1 in /opt/conda/lib/python3.10/site-packages (from matplotlib>=3.3.0->ultralytics) (1.2.0)\nRequirement already satisfied: cycler>=0.10 in /opt/conda/lib/python3.10/site-packages (from matplotlib>=3.3.0->ultralytics) (0.12.1)\nRequirement already satisfied: fonttools>=4.22.0 in /opt/conda/lib/python3.10/site-packages (from matplotlib>=3.3.0->ultralytics) (4.47.0)\nRequirement already satisfied: kiwisolver>=1.0.1 in /opt/conda/lib/python3.10/site-packages (from matplotlib>=3.3.0->ultralytics) (1.4.5)\nRequirement already satisfied: packaging>=20.0 in /opt/conda/lib/python3.10/site-packages (from matplotlib>=3.3.0->ultralytics) (21.3)\nRequirement already satisfied: pyparsing>=2.3.1 in /opt/conda/lib/python3.10/site-packages (from matplotlib>=3.3.0->ultralytics) (3.1.1)\nRequirement already satisfied: python-dateutil>=2.7 in /opt/conda/lib/python3.10/site-packages (from matplotlib>=3.3.0->ultralytics) (2.9.0.post0)\nRequirement already satisfied: pytz>=2020.1 in /opt/conda/lib/python3.10/site-packages (from pandas>=1.1.4->ultralytics) (2023.3.post1)\nRequirement already satisfied: tzdata>=2022.1 in /opt/conda/lib/python3.10/site-packages (from pandas>=1.1.4->ultralytics) (2023.4)\nRequirement already satisfied: charset-normalizer<4,>=2 in /opt/conda/lib/python3.10/site-packages (from requests>=2.23.0->ultralytics) (3.3.2)\nRequirement already satisfied: idna<4,>=2.5 in /opt/conda/lib/python3.10/site-packages (from requests>=2.23.0->ultralytics) (3.6)\nRequirement already satisfied: urllib3<3,>=1.21.1 in /opt/conda/lib/python3.10/site-packages (from requests>=2.23.0->ultralytics) (1.26.18)\nRequirement already satisfied: certifi>=2017.4.17 in /opt/conda/lib/python3.10/site-packages (from requests>=2.23.0->ultralytics) (2024.2.2)\nRequirement already satisfied: filelock in /opt/conda/lib/python3.10/site-packages (from torch>=1.8.0->ultralytics) (3.13.1)\nRequirement already satisfied: typing-extensions in /opt/conda/lib/python3.10/site-packages (from torch>=1.8.0->ultralytics) (4.9.0)\nRequirement already satisfied: sympy in /opt/conda/lib/python3.10/site-packages (from torch>=1.8.0->ultralytics) (1.12)\nRequirement already satisfied: networkx in /opt/conda/lib/python3.10/site-packages (from torch>=1.8.0->ultralytics) (3.2.1)\nRequirement already satisfied: jinja2 in /opt/conda/lib/python3.10/site-packages (from torch>=1.8.0->ultralytics) (3.1.2)\nRequirement already satisfied: fsspec in /opt/conda/lib/python3.10/site-packages (from torch>=1.8.0->ultralytics) (2024.2.0)\nRequirement already satisfied: six>=1.5 in /opt/conda/lib/python3.10/site-packages (from python-dateutil>=2.7->matplotlib>=3.3.0->ultralytics) (1.16.0)\nRequirement already satisfied: MarkupSafe>=2.0 in /opt/conda/lib/python3.10/site-packages (from jinja2->torch>=1.8.0->ultralytics) (2.1.3)\nRequirement already satisfied: mpmath>=0.19 in /opt/conda/lib/python3.10/site-packages (from sympy->torch>=1.8.0->ultralytics) (1.3.0)\nDownloading ultralytics-8.2.101-py3-none-any.whl (874 kB)\n\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m874.1/874.1 kB\u001b[0m \u001b[31m18.9 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m00:01\u001b[0m\n\u001b[?25hDownloading ultralytics_thop-2.0.8-py3-none-any.whl (26 kB)\nInstalling collected packages: ultralytics-thop, ultralytics\nSuccessfully installed ultralytics-8.2.101 ultralytics-thop-2.0.8\n","output_type":"stream"}]},{"cell_type":"code","source":"import ultralytics\nultralytics.checks()","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:27:45.075665Z","iopub.execute_input":"2024-09-26T01:27:45.076025Z","iopub.status.idle":"2024-09-26T01:27:51.228069Z","shell.execute_reply.started":"2024-09-26T01:27:45.075993Z","shell.execute_reply":"2024-09-26T01:27:51.227037Z"},"trusted":true},"execution_count":44,"outputs":[{"name":"stdout","text":"Ultralytics YOLOv8.2.101 🚀 Python-3.10.13 torch-2.1.2 CUDA:0 (Tesla P100-PCIE-16GB, 16269MiB)\nSetup complete ✅ (4 CPUs, 31.4 GB RAM, 5889.5/8062.4 GB disk)\n","output_type":"stream"}]},{"cell_type":"code","source":"rm -rf /kaggle/working/data","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:27:51.229204Z","iopub.execute_input":"2024-09-26T01:27:51.230004Z","iopub.status.idle":"2024-09-26T01:27:53.810537Z","shell.execute_reply.started":"2024-09-26T01:27:51.229977Z","shell.execute_reply":"2024-09-26T01:27:53.808848Z"},"trusted":true},"execution_count":45,"outputs":[]},{"cell_type":"code","source":"!wget https://github.com/THU-MIG/yolov10/releases/download/v1.1/yolov10n.pt","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:27:53.812987Z","iopub.execute_input":"2024-09-26T01:27:53.813512Z","iopub.status.idle":"2024-09-26T01:27:55.490226Z","shell.execute_reply.started":"2024-09-26T01:27:53.813459Z","shell.execute_reply":"2024-09-26T01:27:55.489127Z"},"trusted":true},"execution_count":46,"outputs":[{"name":"stdout","text":"--2024-09-26 01:27:54--  https://github.com/THU-MIG/yolov10/releases/download/v1.1/yolov10n.pt\nResolving github.com (github.com)... 140.82.116.3\nConnecting to github.com (github.com)|140.82.116.3|:443... connected.\nHTTP request sent, awaiting response... 302 Found\nLocation: https://objects.githubusercontent.com/github-production-release-asset-2e65be/804788522/411e0d4f-1023-40ad-bfdd-c99f0dddb73b?X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential=releaseassetproduction%2F20240926%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Date=20240926T012754Z&X-Amz-Expires=300&X-Amz-Signature=09d1593d6249bc8454bb05c0d8c2959fc0817383f106657458fd44a282dd8539&X-Amz-SignedHeaders=host&response-content-disposition=attachment%3B%20filename%3Dyolov10n.pt&response-content-type=application%2Foctet-stream [following]\n--2024-09-26 01:27:54--  https://objects.githubusercontent.com/github-production-release-asset-2e65be/804788522/411e0d4f-1023-40ad-bfdd-c99f0dddb73b?X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Credential=releaseassetproduction%2F20240926%2Fus-east-1%2Fs3%2Faws4_request&X-Amz-Date=20240926T012754Z&X-Amz-Expires=300&X-Amz-Signature=09d1593d6249bc8454bb05c0d8c2959fc0817383f106657458fd44a282dd8539&X-Amz-SignedHeaders=host&response-content-disposition=attachment%3B%20filename%3Dyolov10n.pt&response-content-type=application%2Foctet-stream\nResolving objects.githubusercontent.com (objects.githubusercontent.com)... 185.199.111.133, 185.199.110.133, 185.199.108.133, ...\nConnecting to objects.githubusercontent.com (objects.githubusercontent.com)|185.199.111.133|:443... connected.\nHTTP request sent, awaiting response... 200 OK\nLength: 11448431 (11M) [application/octet-stream]\nSaving to: 'yolov10n.pt'\n\nyolov10n.pt         100%[===================>]  10.92M  --.-KB/s    in 0.1s    \n\n2024-09-26 01:27:55 (87.7 MB/s) - 'yolov10n.pt' saved [11448431/11448431]\n\n","output_type":"stream"}]},{"cell_type":"code","source":"!yolo detect train data=/kaggle/working/yolo.yaml model=yolov10n.pt epochs=150 batch=24 imgsz=1024 device=0","metadata":{"execution":{"iopub.status.busy":"2024-09-26T01:27:55.491794Z","iopub.execute_input":"2024-09-26T01:27:55.492083Z"},"trusted":true},"execution_count":null,"outputs":[{"name":"stdout","text":"Ultralytics YOLOv8.2.101 🚀 Python-3.10.13 torch-2.1.2 CUDA:0 (Tesla P100-PCIE-16GB, 16269MiB)\n\u001b[34m\u001b[1mengine/trainer: \u001b[0mtask=detect, mode=train, model=yolov10n.pt, data=/kaggle/working/yolo.yaml, epochs=150, time=None, patience=100, batch=24, imgsz=1024, save=True, save_period=-1, cache=False, device=0, workers=8, project=None, name=train, exist_ok=False, pretrained=True, optimizer=auto, verbose=True, seed=0, deterministic=True, single_cls=False, rect=False, cos_lr=False, close_mosaic=10, resume=False, amp=True, fraction=1.0, profile=False, freeze=None, multi_scale=False, overlap_mask=True, mask_ratio=4, dropout=0.0, val=True, split=val, save_json=False, save_hybrid=False, conf=None, iou=0.7, max_det=300, half=False, dnn=False, plots=True, source=None, vid_stride=1, stream_buffer=False, visualize=False, augment=False, agnostic_nms=False, classes=None, retina_masks=False, embed=None, show=False, save_frames=False, save_txt=False, save_conf=False, save_crop=False, show_labels=True, show_conf=True, show_boxes=True, line_width=None, format=torchscript, keras=False, optimize=False, int8=False, dynamic=False, simplify=True, opset=None, workspace=4, nms=False, lr0=0.01, lrf=0.01, momentum=0.937, weight_decay=0.0005, warmup_epochs=3.0, warmup_momentum=0.8, warmup_bias_lr=0.1, box=7.5, cls=0.5, dfl=1.5, pose=12.0, kobj=1.0, label_smoothing=0.0, nbs=64, hsv_h=0.015, hsv_s=0.7, hsv_v=0.4, degrees=0.0, translate=0.1, scale=0.5, shear=0.0, perspective=0.0, flipud=0.0, fliplr=0.5, bgr=0.0, mosaic=1.0, mixup=0.0, copy_paste=0.0, auto_augment=randaugment, erasing=0.4, crop_fraction=1.0, cfg=None, tracker=botsort.yaml, save_dir=runs/detect/train\nDownloading https://ultralytics.com/assets/Arial.ttf to '/root/.config/Ultralytics/Arial.ttf'...\n100%|████████████████████████████████████████| 755k/755k [00:00<00:00, 19.7MB/s]\nOverriding model.yaml nc=80 with nc=14\n\n                   from  n    params  module                                       arguments                     \n  0                  -1  1       464  ultralytics.nn.modules.conv.Conv             [3, 16, 3, 2]                 \n  1                  -1  1      4672  ultralytics.nn.modules.conv.Conv             [16, 32, 3, 2]                \n  2                  -1  1      7360  ultralytics.nn.modules.block.C2f             [32, 32, 1, True]             \n  3                  -1  1     18560  ultralytics.nn.modules.conv.Conv             [32, 64, 3, 2]                \n  4                  -1  2     49664  ultralytics.nn.modules.block.C2f             [64, 64, 2, True]             \n  5                  -1  1      9856  ultralytics.nn.modules.block.SCDown          [64, 128, 3, 2]               \n  6                  -1  2    197632  ultralytics.nn.modules.block.C2f             [128, 128, 2, True]           \n  7                  -1  1     36096  ultralytics.nn.modules.block.SCDown          [128, 256, 3, 2]              \n  8                  -1  1    460288  ultralytics.nn.modules.block.C2f             [256, 256, 1, True]           \n  9                  -1  1    164608  ultralytics.nn.modules.block.SPPF            [256, 256, 5]                 \n 10                  -1  1    249728  ultralytics.nn.modules.block.PSA             [256, 256]                    \n 11                  -1  1         0  torch.nn.modules.upsampling.Upsample         [None, 2, 'nearest']          \n 12             [-1, 6]  1         0  ultralytics.nn.modules.conv.Concat           [1]                           \n 13                  -1  1    148224  ultralytics.nn.modules.block.C2f             [384, 128, 1]                 \n 14                  -1  1         0  torch.nn.modules.upsampling.Upsample         [None, 2, 'nearest']          \n 15             [-1, 4]  1         0  ultralytics.nn.modules.conv.Concat           [1]                           \n 16                  -1  1     37248  ultralytics.nn.modules.block.C2f             [192, 64, 1]                  \n 17                  -1  1     36992  ultralytics.nn.modules.conv.Conv             [64, 64, 3, 2]                \n 18            [-1, 13]  1         0  ultralytics.nn.modules.conv.Concat           [1]                           \n 19                  -1  1    123648  ultralytics.nn.modules.block.C2f             [192, 128, 1]                 \n 20                  -1  1     18048  ultralytics.nn.modules.block.SCDown          [128, 128, 3, 2]              \n 21            [-1, 10]  1         0  ultralytics.nn.modules.conv.Concat           [1]                           \n 22                  -1  1    282624  ultralytics.nn.modules.block.C2fCIB          [384, 256, 1, True, True]     \n 23        [16, 19, 22]  1    866788  ultralytics.nn.modules.head.v10Detect        [14, [64, 128, 256]]          \nYOLOv10n summary: 385 layers, 2,712,500 parameters, 2,712,484 gradients, 8.4 GFLOPs\n\nTransferred 493/595 items from pretrained weights\n\u001b[34m\u001b[1mTensorBoard: \u001b[0mStart with 'tensorboard --logdir runs/detect/train', view at http://localhost:6006/\nFreezing layer 'model.23.dfl.conv.weight'\n\u001b[34m\u001b[1mAMP: \u001b[0mrunning Automatic Mixed Precision (AMP) checks with YOLOv8n...\nDownloading https://github.com/ultralytics/assets/releases/download/v8.2.0/yolov8n.pt to 'yolov8n.pt'...\n100%|██████████████████████████████████████| 6.25M/6.25M [00:00<00:00, 98.4MB/s]\n\u001b[34m\u001b[1mAMP: \u001b[0mchecks passed ✅\n\u001b[34m\u001b[1mtrain: \u001b[0mScanning /kaggle/working/refactored_data/labels/train... 6547 images, 0 b\u001b[0m\n\u001b[34m\u001b[1mtrain: \u001b[0mNew cache created: /kaggle/working/refactored_data/labels/train.cache\n\u001b[34m\u001b[1malbumentations: \u001b[0mBlur(p=0.01, blur_limit=(3, 7)), MedianBlur(p=0.01, blur_limit=(3, 7)), ToGray(p=0.01), CLAHE(p=0.01, clip_limit=(1, 4.0), tile_grid_size=(8, 8))\n\u001b[34m\u001b[1mval: \u001b[0mScanning /kaggle/working/refactored_data/labels/val... 728 images, 0 backgr\u001b[0m\n\u001b[34m\u001b[1mval: \u001b[0mNew cache created: /kaggle/working/refactored_data/labels/val.cache\nPlotting labels to runs/detect/train/labels.jpg... \n\u001b[34m\u001b[1moptimizer:\u001b[0m 'optimizer=auto' found, ignoring 'lr0=0.01' and 'momentum=0.937' and determining best 'optimizer', 'lr0' and 'momentum' automatically... \n\u001b[34m\u001b[1moptimizer:\u001b[0m SGD(lr=0.01, momentum=0.9) with parameter groups 95 weight(decay=0.0), 108 weight(decay=0.0005625000000000001), 107 bias(decay=0.0)\n\u001b[34m\u001b[1mTensorBoard: \u001b[0mmodel graph visualization added ✅\nImage sizes 1024 train, 1024 val\nUsing 4 dataloader workers\nLogging results to \u001b[1mruns/detect/train\u001b[0m\nStarting training for 150 epochs...\n\n      Epoch    GPU_mem   box_loss   cls_loss   dfl_loss  Instances       Size\n      1/150      13.5G       4.05       13.9      4.009         83       1024: 1\n                 Class     Images  Instances      Box(P          R      mAP50  m\n                   all        728       2845      0.899     0.0175     0.0332     0.0192\n\n      Epoch    GPU_mem   box_loss   cls_loss   dfl_loss  Instances       Size\n      2/150      11.8G      3.639      10.74       3.52        114       1024:  ","output_type":"stream"}]},{"cell_type":"code","source":"# from ultralytics import YOLOv10\n\n# model = YOLOv10()\n# # If you want to finetune the model with pretrained weights, you could load the \n# # pretrained weights like below\n# # model = YOLOv10.from_pretrained('jameslahm/yolov10{n/s/m/b/l/x}')\n# # or\n# # wget https://github.com/THU-MIG/yolov10/releases/download/v1.1/yolov10{n/s/m/b/l/x}.pt\n# # model = YOLOv10('yolov10{n/s/m/b/l/x}.pt')\n\n# model.train(data='yolo.yaml', epochs=150, batch=24, imgsz=1024)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !WANDB_MODE=\"dryrun\" python train_dual.py \n# !yolo train model=yolov10n.pt workers=10 device=0 batch=24 data = /kaggle/working/yolo.yaml imgsz = 1024 epochs = 150","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# raw_df, preprocessed_df = prime_dataset()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip uninstall -y wandb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !cp /kaggle/working/valid.txt /kaggle/working/yolov7/valid.txt\n# !cp /kaggle/working/train.txt /kaggle/working/yolov7/train.txt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rm -rf /kaggle/working/runs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rm -rf /kaggle/working/yolov8n.pt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rm -rf /kaggle/working/exp.zip","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_model()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp /kaggle/working/yolov9/runs/train/exp/weights/last.pt /kaggle/working/yolov7_lan1.pt","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working/\nfrom IPython.display import FileLink\nFileLink(r'train.zip')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/working/yolov7/runs/train/exp /kaggle/working/","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nfrom IPython.display import FileLink\n\n# Đường dẫn đến thư mục cần nén\nsource_dir = '/kaggle/working/runs/detect/train'\n\n# Nén thư mục 'runs' thành file 'runs.zip'\nshutil.make_archive('/kaggle/working/train', 'zip', source_dir)\n\n# Tạo liên kết tải về cho file nén\nFileLink(r'/kaggle/working/train.zip')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -rf /kaggle/working/exp.zip","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}