{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":24800,"databundleVersionId":1831594,"sourceType":"competition"},{"sourceId":1799839,"sourceType":"datasetVersion","datasetId":1069682},{"sourceId":1800777,"sourceType":"datasetVersion","datasetId":1069787},{"sourceId":7483723,"sourceType":"datasetVersion","datasetId":4356112}],"dockerImageVersionId":30636,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np, pandas as pd\nfrom glob import glob\nimport shutil, os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom tqdm.notebook import tqdm\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:30:44.51781Z","iopub.execute_input":"2024-01-29T06:30:44.518485Z","iopub.status.idle":"2024-01-29T06:30:45.552487Z","shell.execute_reply.started":"2024-01-29T06:30:44.518447Z","shell.execute_reply":"2024-01-29T06:30:45.551762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dim = 1024 #512, 256, 'original'\nfold = 4\ntrain_df = pd.read_csv(f'../input/vinbigdata-{dim}-image-dataset/vinbigdata/train.csv')\ntrain_df['image_path'] = f'../input/vinbigdata-{dim}-image-dataset/vinbigdata/train/'+train_df.image_id+('.png' if dim!='original' else '.jpg')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:30:46.861465Z","iopub.execute_input":"2024-01-29T06:30:46.862654Z","iopub.status.idle":"2024-01-29T06:30:47.106622Z","shell.execute_reply.started":"2024-01-29T06:30:46.862617Z","shell.execute_reply":"2024-01-29T06:30:47.105691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[train_df.class_id!=14].reset_index(drop = True)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:30:49.437238Z","iopub.execute_input":"2024-01-29T06:30:49.43759Z","iopub.status.idle":"2024-01-29T06:30:49.456988Z","shell.execute_reply.started":"2024-01-29T06:30:49.43756Z","shell.execute_reply":"2024-01-29T06:30:49.456134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['x_min'] = train_df.apply(lambda row: (row.x_min)/row.width, axis =1)\ntrain_df['y_min'] = train_df.apply(lambda row: (row.y_min)/row.height, axis =1)\n\ntrain_df['x_max'] = train_df.apply(lambda row: (row.x_max)/row.width, axis =1)\ntrain_df['y_max'] = train_df.apply(lambda row: (row.y_max)/row.height, axis =1)\n\ntrain_df['x_mid'] = train_df.apply(lambda row: (row.x_max+row.x_min)/2, axis =1)\ntrain_df['y_mid'] = train_df.apply(lambda row: (row.y_max+row.y_min)/2, axis =1)\n\ntrain_df['w'] = train_df.apply(lambda row: (row.x_max-row.x_min), axis =1)\ntrain_df['h'] = train_df.apply(lambda row: (row.y_max-row.y_min), axis =1)\n\ntrain_df['area'] = train_df['w']*train_df['h']\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:30:51.115832Z","iopub.execute_input":"2024-01-29T06:30:51.116202Z","iopub.status.idle":"2024-01-29T06:30:58.459687Z","shell.execute_reply.started":"2024-01-29T06:30:51.11617Z","shell.execute_reply":"2024-01-29T06:30:58.458745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['x_min', 'y_min', 'x_max', 'y_max', 'x_mid', 'y_mid', 'w', 'h', 'area']\nX = train_df[features]\ny = train_df['class_id']\nX.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:31:00.180982Z","iopub.execute_input":"2024-01-29T06:31:00.181342Z","iopub.status.idle":"2024-01-29T06:31:00.190605Z","shell.execute_reply.started":"2024-01-29T06:31:00.181311Z","shell.execute_reply":"2024-01-29T06:31:00.18957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_ids, class_names = list(zip(*set(zip(train_df.class_id, train_df.class_name))))\nclasses = list(np.array(class_names)[np.argsort(class_ids)])\nclasses = list(map(lambda x: str(x), classes))\nclasses","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:31:02.87639Z","iopub.execute_input":"2024-01-29T06:31:02.876971Z","iopub.status.idle":"2024-01-29T06:31:02.89668Z","shell.execute_reply.started":"2024-01-29T06:31:02.876935Z","shell.execute_reply":"2024-01-29T06:31:02.895769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf  = GroupKFold(n_splits = 5)\ntrain_df['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(train_df, groups = train_df.image_id.tolist())):\n    train_df.loc[val_idx, 'fold'] = fold\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:31:05.407816Z","iopub.execute_input":"2024-01-29T06:31:05.408682Z","iopub.status.idle":"2024-01-29T06:31:05.512232Z","shell.execute_reply.started":"2024-01-29T06:31:05.408636Z","shell.execute_reply":"2024-01-29T06:31:05.510992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = []\nval_files   = []\nval_files += list(train_df[train_df.fold==fold].image_path.unique())\ntrain_files += list(train_df[train_df.fold!=fold].image_path.unique())","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:31:08.499609Z","iopub.execute_input":"2024-01-29T06:31:08.500486Z","iopub.status.idle":"2024-01-29T06:31:08.525273Z","shell.execute_reply.started":"2024-01-29T06:31:08.50045Z","shell.execute_reply":"2024-01-29T06:31:08.524523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs('/kaggle/working/vinbigdata/labels/train', exist_ok = True)\nos.makedirs('/kaggle/working/vinbigdata/labels/val', exist_ok = True)\nos.makedirs('/kaggle/working/vinbigdata/images/train', exist_ok = True)\nos.makedirs('/kaggle/working/vinbigdata/images/val', exist_ok = True)\nlabel_dir = '/kaggle/input/vinbigdata-yolo-labels-dataset/labels'\nfor file in tqdm(train_files):\n    shutil.copy(file, '/kaggle/working/vinbigdata/images/train/')\n    filename = file.split('/')[-1].split('.')[0]\n    shutil.copy(os.path.join(label_dir, filename+'.txt'), '/kaggle/working/vinbigdata/labels/train')\n        \nfor file in tqdm(val_files):\n    shutil.copy(file, '/kaggle/working/vinbigdata/images/val/')\n    filename = file.split('/')[-1].split('.')[0]\n    shutil.copy(os.path.join(label_dir, filename+'.txt'), '/kaggle/working/vinbigdata/labels/val')","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:31:19.573989Z","iopub.execute_input":"2024-01-29T06:31:19.574366Z","iopub.status.idle":"2024-01-29T06:33:02.165339Z","shell.execute_reply.started":"2024-01-29T06:31:19.574337Z","shell.execute_reply":"2024-01-29T06:33:02.164458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\ndef filter_categories(input_dir, output_dir, category_mapping):\n    # 確保輸出目錄存在\n    if not os.path.exists(output_dir):\n        os.makedirs(output_dir)\n\n    # 遍歷輸入目錄中的所有文件\n    for file_name in os.listdir(input_dir):\n        if file_name.endswith('.txt'):\n            input_file_path = os.path.join(input_dir, file_name)\n            output_file_path = os.path.join(output_dir, file_name)\n\n            with open(input_file_path, 'r') as file:\n                filtered_data = []\n                for line in file:\n                    parts = line.split()\n                    if parts and int(parts[0]) in category_mapping:\n                        # 替換類別代碼\n                        parts[0] = str(category_mapping[int(parts[0])])\n                        filtered_data.append(' '.join(parts))\n\n            # 如果篩選後的數據不為空，則寫入文件\n            if filtered_data:\n                with open(output_file_path, 'w') as file:\n                    file.writelines('\\n'.join(filtered_data) + '\\n')\n            else:\n                # 如果篩選後的數據為空，則刪除已創建的空文件（如果存在）\n                if os.path.exists(output_file_path):\n                    os.remove(output_file_path)\n\n# 主目錄\nbase_dir = '/kaggle/working/vinbigdata/labels'\n\n# 類別映射\ncategory_mapping = {8: 0}\n\n# 分別處理 'train' 和 'val' 資料夾\nfor folder in ['train', 'val']:\n    input_dir = os.path.join(base_dir, folder)\n    output_dir = os.path.join('/kaggle/working/oneclass/labels', folder)\n    filter_categories(input_dir, output_dir, category_mapping)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T07:11:51.716844Z","iopub.execute_input":"2024-01-29T07:11:51.717243Z","iopub.status.idle":"2024-01-29T07:11:51.990868Z","shell.execute_reply.started":"2024-01-29T07:11:51.717202Z","shell.execute_reply":"2024-01-29T07:11:51.989795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\n\ndef select_images(txt_dir, img_dir, output_dir):\n    # 確保輸出目錄存在\n    if not os.path.exists(output_dir):\n        os.makedirs(output_dir)\n\n    # 獲取所有處理過的txt文件名（不包括擴展名）\n    processed_files = {os.path.splitext(file)[0] for file in os.listdir(txt_dir) if file.endswith('.txt')}\n\n    # 遍歷圖像目錄\n    for img_file in os.listdir(img_dir):\n        file_name_without_ext = os.path.splitext(img_file)[0]\n        if file_name_without_ext in processed_files:\n            # 複製圖像到目標目錄\n            shutil.copy(os.path.join(img_dir, img_file), os.path.join(output_dir, img_file))\n\n# 定義路徑\nbase_txt_dir = '/kaggle/working/oneclass/labels'\nbase_img_dir = '/kaggle/working/vinbigdata/images'\noutput_base_dir = '/kaggle/working/oneclass/images'\n\n# 對train和val資料夾進行處理\nfor folder in ['train', 'val']:\n    select_images(os.path.join(base_txt_dir, folder), \n                  os.path.join(base_img_dir, folder), \n                  os.path.join(output_base_dir, folder))","metadata":{"execution":{"iopub.status.busy":"2024-01-29T07:11:55.135837Z","iopub.execute_input":"2024-01-29T07:11:55.136551Z","iopub.status.idle":"2024-01-29T07:11:55.562634Z","shell.execute_reply.started":"2024-01-29T07:11:55.136516Z","shell.execute_reply":"2024-01-29T07:11:55.561501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile /kaggle/working/dataset.yaml\n# Path\npath: /kaggle/working/oneclass\ntrain: images/train\nval: images/val\n\n# Classes\nnc: 1\nnames: ['Nodule/Mass']","metadata":{"execution":{"iopub.status.busy":"2024-01-29T07:12:09.336227Z","iopub.execute_input":"2024-01-29T07:12:09.336583Z","iopub.status.idle":"2024-01-29T07:12:09.345904Z","shell.execute_reply.started":"2024-01-29T07:12:09.336555Z","shell.execute_reply":"2024-01-29T07:12:09.344634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\n\ndef zipdir(path, ziph):\n    # 遞歸地添加文件\n    for root, dirs, files in os.walk(path):\n        for file in files:\n            ziph.write(os.path.join(root, file), \n                       os.path.relpath(os.path.join(root, file), \n                                       os.path.join(path, '..')))\n\n# 要壓縮的資料夾路徑\nfolder_path = '/kaggle/working/oneclass'\n# 創建的ZIP文件的名稱\nzip_filename = '/kaggle/working/oneclass.zip'\n\n# 創建一個ZipFile對象\nzipf = zipfile.ZipFile(zip_filename, 'w', zipfile.ZIP_DEFLATED)\nzipdir(folder_path, zipf)\nzipf.close()\n\nzip_filename  # 返回壓縮文件的名稱以供確認","metadata":{"execution":{"iopub.status.busy":"2024-01-29T07:12:29.79968Z","iopub.execute_input":"2024-01-29T07:12:29.800579Z","iopub.status.idle":"2024-01-29T07:12:45.593976Z","shell.execute_reply.started":"2024-01-29T07:12:29.800544Z","shell.execute_reply":"2024-01-29T07:12:45.592972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cd /kaggle/working/\nfrom IPython.display import FileLink\nFileLink('oneclass.zip')","metadata":{"execution":{"iopub.status.busy":"2024-01-29T07:12:52.255243Z","iopub.execute_input":"2024-01-29T07:12:52.255614Z","iopub.status.idle":"2024-01-29T07:12:53.25595Z","shell.execute_reply.started":"2024-01-29T07:12:52.255581Z","shell.execute_reply":"2024-01-29T07:12:53.254771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install ultralytics\n!pip install wandb","metadata":{"execution":{"iopub.status.busy":"2024-01-29T06:39:01.309589Z","iopub.execute_input":"2024-01-29T06:39:01.310454Z","iopub.status.idle":"2024-01-29T06:39:26.784774Z","shell.execute_reply.started":"2024-01-29T06:39:01.310401Z","shell.execute_reply":"2024-01-29T06:39:26.783511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# Load a model\nmodel = YOLO('yolov8n.yaml')  # build a new model from YAML\nmodel = YOLO('yolov8n.pt')  # load a pretrained model (recommended for training)\nmodel = YOLO('yolov8n.yaml').load('yolov8n.pt')  # build from YAML and transfer weights\n\n# Train the model\nresults = model.train(data='/kaggle/working/dataset.yaml', epochs=1000, imgsz=512)","metadata":{"execution":{"iopub.status.busy":"2024-01-29T07:13:35.587613Z","iopub.execute_input":"2024-01-29T07:13:35.588027Z","iopub.status.idle":"2024-01-29T07:37:21.734541Z","shell.execute_reply.started":"2024-01-29T07:13:35.587992Z","shell.execute_reply":"2024-01-29T07:37:21.733557Z"},"trusted":true},"execution_count":null,"outputs":[]}]}