{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":24800,"databundleVersionId":1831594,"sourceType":"competition"},{"sourceId":1800067,"sourceType":"datasetVersion","datasetId":1069810},{"sourceId":8015404,"sourceType":"datasetVersion","datasetId":4722369},{"sourceId":8015921,"sourceType":"datasetVersion","datasetId":4722743},{"sourceId":8439882,"sourceType":"datasetVersion","datasetId":5027654}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\n# !pip install pandas\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-17T11:13:24.114342Z","iopub.execute_input":"2024-05-17T11:13:24.115181Z","iopub.status.idle":"2024-05-17T11:13:25.089969Z","shell.execute_reply.started":"2024-05-17T11:13:24.115144Z","shell.execute_reply":"2024-05-17T11:13:25.089161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"        \n!pip install --upgrade -q wandb\n# from kaggle_secrets import UserSecretsClient\n\n# user_secrets = UserSecretsClient()\n\n!wandb.login(key=e5facd9412e5fd3991e6d9ad0c3e716b2fc97383)\n\n\nos.environ['WANDB_MODE'] = 'dryrun'\n\n# Specify your wandb project name\nos.environ['WANDB_PROJECT'] = 'shubham-aagam'\n# os.environ['WANDB_API_KEY'] = 'e5facd9412e5fd3991e6d9ad0c3e716b2fc97383'\n\n!wandb login e5facd9412e5fd3991e6d9ad0c3e716b2fc97383","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:25.091418Z","iopub.execute_input":"2024-05-17T11:13:25.091822Z","iopub.status.idle":"2024-05-17T11:13:46.621168Z","shell.execute_reply.started":"2024-05-17T11:13:25.091795Z","shell.execute_reply":"2024-05-17T11:13:46.619918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading train_dataframe","metadata":{}},{"cell_type":"code","source":"# test_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/test\"\n# train_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"\n# train_df = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\n# train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.622893Z","iopub.execute_input":"2024-05-17T11:13:46.623321Z","iopub.status.idle":"2024-05-17T11:13:46.62768Z","shell.execute_reply.started":"2024-05-17T11:13:46.623288Z","shell.execute_reply":"2024-05-17T11:13:46.626838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## mapping image_id to its image_path","metadata":{}},{"cell_type":"code","source":"# import glob\n# from tqdm import tqdm\n# import pandas as pd\n\n# # Enable pandas progress_apply\n# tqdm.pandas()\n\n# # Load the list of training DICOM images\n# yy = glob.glob(train_dir + \"/*\")\n\n# # Apply progress_apply to create the 'ImagePath' column\n# train_df['ImagePath'] = train_df['image_id'].progress_apply(lambda x: next(filter(lambda y: x in y, yy), None))\n\n# # Filter out the 'No Finding' class (class_id == 14)\n# train_df = train_df[train_df['class_id'] != 14].reset_index(drop=True)\n\n# # Select only required columns\n# train_df = train_df[['ImagePath', 'image_id', 'class_name', 'class_id', 'rad_id', 'x_min', 'y_min', 'x_max', 'y_max']]\n","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.630444Z","iopub.execute_input":"2024-05-17T11:13:46.630739Z","iopub.status.idle":"2024-05-17T11:13:46.643745Z","shell.execute_reply.started":"2024-05-17T11:13:46.630714Z","shell.execute_reply":"2024-05-17T11:13:46.642887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analysis of the train_dataframe","metadata":{}},{"cell_type":"code","source":"# print(\"No Of The Unique ImagePath :--->\", len(set(train_df['ImagePath'])))\n# print(\"Shape Of The Data Frame :->\", train_df.shape)\n# train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.644746Z","iopub.execute_input":"2024-05-17T11:13:46.645032Z","iopub.status.idle":"2024-05-17T11:13:46.65549Z","shell.execute_reply.started":"2024-05-17T11:13:46.645009Z","shell.execute_reply":"2024-05-17T11:13:46.654579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Function To Convert Diacom To An Image\n#### Creating Train and Validation Directories and Converting the DICOM to Image Array and  Saving It in Train and Validation Directories","metadata":{}},{"cell_type":"code","source":"# import os\n# import glob\n# import numpy as np\n# import pydicom\n# import matplotlib.pyplot as plt\n# from pydicom.pixel_data_handlers.util import apply_voi_lut\n# import cv2\n# import warnings\n# warnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.65674Z","iopub.execute_input":"2024-05-17T11:13:46.657012Z","iopub.status.idle":"2024-05-17T11:13:46.666285Z","shell.execute_reply.started":"2024-05-17T11:13:46.65699Z","shell.execute_reply":"2024-05-17T11:13:46.665481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def dicom2array(path, voi_lut=True, fix_monochrome=True):\n#     dicom = pydicom.read_file(path)\n#     # VOI LUT (if available by DICOM device) is used to\n#     # transform raw DICOM data to \"human-friendly\" view\n#     if voi_lut:\n#         data = apply_voi_lut(dicom.pixel_array, dicom)\n#     else:\n#         data = dicom.pixel_array\n#     # depending on this value, X-ray may look inverted - fix that:\n#     if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n#         data = np.amax(data) - data\n#     data = data - np.min(data)\n#     data = data / np.max(data)\n#     data = (data * 255).astype(np.uint8)\n#     return data\n\n# def plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n#     rows = len(imgs)//cols + 1\n#     fig = plt.figure(figsize=(cols*size, rows*size))\n#     for i, img in enumerate(imgs):\n#         if img_size is not None:\n#             img = cv2.resize(img, img_size)\n#         fig.add_subplot(rows, cols, i+1)\n#         plt.imshow(img, cmap=cmap)\n#     plt.suptitle(title)\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.667426Z","iopub.execute_input":"2024-05-17T11:13:46.667696Z","iopub.status.idle":"2024-05-17T11:13:46.683103Z","shell.execute_reply.started":"2024-05-17T11:13:46.667672Z","shell.execute_reply":"2024-05-17T11:13:46.68217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dicom_paths =  list(set(train_df['ImagePath']))\n# imgs = [dicom2array(path) for path in dicom_paths[:4]]\n# plot_imgs(imgs)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.684279Z","iopub.execute_input":"2024-05-17T11:13:46.684586Z","iopub.status.idle":"2024-05-17T11:13:46.693383Z","shell.execute_reply.started":"2024-05-17T11:13:46.684556Z","shell.execute_reply":"2024-05-17T11:13:46.692447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Maybe, you can try some preprocess like equalize histogram. You can see the difference between before and after","metadata":{}},{"cell_type":"code","source":"# from skimage import exposure\n# imgs = [exposure.equalize_hist(img) for img in imgs]\n# plot_imgs(imgs)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.697455Z","iopub.execute_input":"2024-05-17T11:13:46.698145Z","iopub.status.idle":"2024-05-17T11:13:46.70225Z","shell.execute_reply.started":"2024-05-17T11:13:46.698118Z","shell.execute_reply":"2024-05-17T11:13:46.701336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Function to save the image in a slower way","metadata":{}},{"cell_type":"code","source":"# def saving_image(output_dir, dicom_path_list = dicom_paths):\n#     for dicom_path in tqdm(dicom_path_list):\n#         file_name = os.path.basename(dicom_path).split('.')[0]\n#         image_array = dicom2array(dicom_path)\n#         equalized_image = exposure.equalize_hist(image_array)\n#         cv2.imwrite(os.path.join(output_dir, f\"{file_name}.jpeg\"), equalized_image)\n\n# # Example usage:\n# output_dir = \"/kaggle/working/chest_detection/Images\"\n# os.makedirs(output_dir, exist_ok=True)\n\n# # Call the function to save images\n# saving_image(output_dir, dicom_paths[:3])","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.70332Z","iopub.execute_input":"2024-05-17T11:13:46.703581Z","iopub.status.idle":"2024-05-17T11:13:46.71252Z","shell.execute_reply.started":"2024-05-17T11:13:46.703558Z","shell.execute_reply":"2024-05-17T11:13:46.711616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Here we will use MultiThreading to Process and save the image in a faster way ","metadata":{}},{"cell_type":"code","source":"# import os\n# import cv2\n# import numpy as np\n# import pydicom\n# import multiprocessing\n# from tqdm import tqdm\n# from skimage import exposure\n\n# def dicom2array(path, voi_lut=True, fix_monochrome=True):\n#     dicom = pydicom.read_file(path)\n#     if voi_lut:\n#         data = apply_voi_lut(dicom.pixel_array, dicom)\n#     else:\n#         data = dicom.pixel_array\n#     if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n#         data = np.amax(data) - data\n#     data = data - np.min(data)\n#     data = data / np.max(data)\n#     data = (data * 255).astype(np.uint8)\n#     return data\n\n# def process_image(dicom_path_output_dir):\n#     dicom_path, output_dir = dicom_path_output_dir\n#     file_name = os.path.splitext(os.path.basename(dicom_path))[0]\n#     image_array = dicom2array(dicom_path)\n#     equalized_image = exposure.equalize_hist(image_array)\n#     equalized_image = (equalized_image * 255).astype(np.uint8)\n#     cv2.imwrite(os.path.join(output_dir, f\"{file_name}.jpeg\"), equalized_image)\n\n# def saving_image(output_dir, dicom_path_list):\n#     os.makedirs(output_dir, exist_ok=True)\n#     dicom_path_output_dir_list = [(path, output_dir) for path in dicom_path_list]\n\n#     # Use multiprocessing Pool for parallel processing\n#     with multiprocessing.Pool() as pool:\n#         list(tqdm(pool.imap(process_image, dicom_path_output_dir_list), total=len(dicom_path_list), desc=\"Processing Images\"))","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.713713Z","iopub.execute_input":"2024-05-17T11:13:46.714023Z","iopub.status.idle":"2024-05-17T11:13:46.72562Z","shell.execute_reply.started":"2024-05-17T11:13:46.713998Z","shell.execute_reply":"2024-05-17T11:13:46.724593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Now We Will Process The Train Diacom Images","metadata":{}},{"cell_type":"code","source":"# # Example usage:\n# output_dir = \"/kaggle/working/chest_detection/images\"\n# os.makedirs(output_dir, exist_ok=True)\n\n# # Call the function to save image\n# saving_image(output_dir, dicom_paths)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.726742Z","iopub.execute_input":"2024-05-17T11:13:46.727037Z","iopub.status.idle":"2024-05-17T11:13:46.735323Z","shell.execute_reply.started":"2024-05-17T11:13:46.727012Z","shell.execute_reply":"2024-05-17T11:13:46.734407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Now We Will Process The Test Diacom Image","metadata":{}},{"cell_type":"code","source":"# test_dicom_paths = glob.glob(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/test/*\")","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.736324Z","iopub.execute_input":"2024-05-17T11:13:46.736604Z","iopub.status.idle":"2024-05-17T11:13:46.747507Z","shell.execute_reply.started":"2024-05-17T11:13:46.736579Z","shell.execute_reply":"2024-05-17T11:13:46.74677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Example usage:\n# output_dir = \"/kaggle/working/chest_detection/test\"\n# os.makedirs(output_dir, exist_ok=True)\n\n# # Call the function to save images\n# saving_image(output_dir, test_dicom_paths)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.748453Z","iopub.execute_input":"2024-05-17T11:13:46.7487Z","iopub.status.idle":"2024-05-17T11:13:46.758447Z","shell.execute_reply.started":"2024-05-17T11:13:46.748679Z","shell.execute_reply":"2024-05-17T11:13:46.757643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### now we will visualize the image on the  images directory","metadata":{}},{"cell_type":"code","source":"# yy = glob.glob(\"/kaggle/working/chest_detection/images/*\") ## SaveImagePath\n# array = cv2.imread(yy[2])\n# plt.imshow(array)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.75945Z","iopub.execute_input":"2024-05-17T11:13:46.759698Z","iopub.status.idle":"2024-05-17T11:13:46.768245Z","shell.execute_reply.started":"2024-05-17T11:13:46.759676Z","shell.execute_reply":"2024-05-17T11:13:46.767464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking Whether The height and width of all image is same or different\n### THis Will Help In Normalizing BoudingBox","metadata":{}},{"cell_type":"code","source":"# heights = []\n# widths =  []\n\n# # Use list comprehensions for a more concise and efficient code\n# heights = [cv2.imread(i).shape[0] for i in tqdm(yy)]\n# widths = [cv2.imread(i).shape[1] for i in tqdm(yy)]","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.76929Z","iopub.execute_input":"2024-05-17T11:13:46.769551Z","iopub.status.idle":"2024-05-17T11:13:46.778318Z","shell.execute_reply.started":"2024-05-17T11:13:46.769528Z","shell.execute_reply":"2024-05-17T11:13:46.777551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"Height is \", heights[:3])\n# print(\"width is \", widths[:3])\n\n# ### As We Can Clearly See That The Height And Width Vary From Image To Image.","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.77973Z","iopub.execute_input":"2024-05-17T11:13:46.780047Z","iopub.status.idle":"2024-05-17T11:13:46.788287Z","shell.execute_reply.started":"2024-05-17T11:13:46.780018Z","shell.execute_reply":"2024-05-17T11:13:46.787557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating Dataframe which will contain the columns SaveImagePath , heights and width","metadata":{}},{"cell_type":"code","source":"# ### Now Creating Data Frame For The Height ,Width \n# df = pd.DataFrame(yy, columns =['SaveImagePath'])\n# df['image_id'] = df['SaveImagePath'].apply(lambda x: x.split('/')[-1].split('.')[0])\n# df['Height'] = heights\n# df['Width']  = widths\n# print(\"shape Of The Data Frame :->\", df.shape)\n# df.head(1)\n\n# print(\"shape of the train_df\", train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.78928Z","iopub.execute_input":"2024-05-17T11:13:46.789559Z","iopub.status.idle":"2024-05-17T11:13:46.798155Z","shell.execute_reply.started":"2024-05-17T11:13:46.789534Z","shell.execute_reply":"2024-05-17T11:13:46.797217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final_df  = train_df.merge(df, on = 'image_id')\n# final_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.799304Z","iopub.execute_input":"2024-05-17T11:13:46.799658Z","iopub.status.idle":"2024-05-17T11:13:46.808731Z","shell.execute_reply.started":"2024-05-17T11:13:46.799582Z","shell.execute_reply":"2024-05-17T11:13:46.807958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Converting DataFrame to YOLO Format:\n\n#### To prepare bounding box annotations for YOLO object detection, the DataFrame containing bounding box coordinates and class labels needs to be converted to YOLO format. \n\nIn YOLO format, bounding box coordinates are normalized to the range [0, 1], where (x_center, y_center, width, height) are relative to the image dimensions. This normalized format ensures consistency across images of different sizes and aspect ratios.\n","metadata":{}},{"cell_type":"code","source":"# def convert_to_yolo_format(df):\n#     # Normalize the bounding box coordinates\n#     df['center_x'] = (df['x_min'] + df['x_max']) / 2\n#     df['center_y'] = (df['y_min'] + df['y_max']) / 2\n#     df['b_box_width'] = df['x_max'] - df['x_min']\n#     df['b_box_height'] = df['y_max'] - df['y_min']\n    \n#     # Calculate normalized coordinates and dimensions\n#     df['normalized_x'] = df['center_x'] / df['Width']\n#     df['normalized_y'] = df['center_y'] / df['Height']\n#     df['normalized_width'] = df['b_box_width'] / df['Width']\n#     df['normalized_height'] = df['b_box_height'] / df['Height']\n    \n#     return df\n\n# ### This Function Will Return The DataFrame With Normalize Bounding Box\n# df_yolo = convert_to_yolo_format(final_df)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.80971Z","iopub.execute_input":"2024-05-17T11:13:46.810001Z","iopub.status.idle":"2024-05-17T11:13:46.819129Z","shell.execute_reply.started":"2024-05-17T11:13:46.809977Z","shell.execute_reply":"2024-05-17T11:13:46.818414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_yolo.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.820169Z","iopub.execute_input":"2024-05-17T11:13:46.820461Z","iopub.status.idle":"2024-05-17T11:13:46.83358Z","shell.execute_reply.started":"2024-05-17T11:13:46.820432Z","shell.execute_reply":"2024-05-17T11:13:46.832782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Function To Save The Bounding Box","metadata":{}},{"cell_type":"code","source":"# def Get_Bounding_Box(df, output_file):\n#     # Open the output file for writing\n#     with open(output_file, 'w') as f:\n#         # Iterate over the filtered DataFrame and write bounding box information to the file\n#         for _, row in df.iterrows():\n#             class_id = row['class_id']\n#             x_center, y_center = row['normalized_x'], row['normalized_y']\n#             width, height = row[\"normalized_width\"], row['normalized_height']\n# #             print(f\"{class_id}\\t{x_center}\\t{y_center}\\t{width}\\t{height}\\n\")\n#             f.write(f\"{class_id}\\t{x_center}\\t{y_center}\\t{width}\\t{height}\\n\")\n\n# # Example usage:\n# Get_Bounding_Box(df=df_yolo, output_file='test.txt')\n\n# Image_label_dir = \"/kaggle/working/chest_detection/labels\"\n# os.makedirs(Image_label_dir, exist_ok = True)\n\n# print(\"Storing  Training Image BoundingBox\",\"-\"*50)\n# print(\"Train label dir is \", Image_label_dir)\n# for file in tqdm(yy):\n#     filename = file.split('/')[-1].split('.')[0]\n#     Get_Bounding_Box(df=df_yolo.head(), output_file = Image_label_dir+\"/\"+filename+\".txt\")\n","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.834682Z","iopub.execute_input":"2024-05-17T11:13:46.834965Z","iopub.status.idle":"2024-05-17T11:13:46.843668Z","shell.execute_reply.started":"2024-05-17T11:13:46.834934Z","shell.execute_reply":"2024-05-17T11:13:46.842826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Conclusion :- > Finally  We Have store Images and Label in Chest Directory","metadata":{}},{"cell_type":"markdown","source":"#### Lets Check Inside the label directory\n","metadata":{}},{"cell_type":"code","source":"# label_files = glob.glob('/kaggle/working/chest_detection/labels/*')\n# '/kaggle/working/chest_detection/Images/5184bc9a54adf7c8cb707c45f21fd741.jpeg'\n\n# with open(label_files[0]) as f:\n#     file = f.read()\n#     print(file)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.844699Z","iopub.execute_input":"2024-05-17T11:13:46.844988Z","iopub.status.idle":"2024-05-17T11:13:46.861241Z","shell.execute_reply.started":"2024-05-17T11:13:46.844964Z","shell.execute_reply":"2024-05-17T11:13:46.860503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lets Check Inside the Image directory","metadata":{}},{"cell_type":"code","source":"# image_files = yy[0]\n\n# array = cv2.imread(image_files)\n# plt.imshow(array)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:46.865196Z","iopub.execute_input":"2024-05-17T11:13:46.865557Z","iopub.status.idle":"2024-05-17T11:13:46.873999Z","shell.execute_reply.started":"2024-05-17T11:13:46.865531Z","shell.execute_reply":"2024-05-17T11:13:46.87307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bounding Box","metadata":{}},{"cell_type":"code","source":"# import cv2\n# import matplotlib.pyplot as plt\n# import numpy as np\n# import random\n\n# def plot_image_with_bounding_box(image, bounding_boxes, class_dict):\n#     fig, ax = plt.subplots()\n#     ax.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    \n#     for box in bounding_boxes:\n#         class_id, x, y, width, height = map(float, box.split())\n#         image_width, image_height = image.shape[1], image.shape[0]\n#         x1 = int((x - width / 2) * image_width)\n#         y1 = int((y - height / 2) * image_height)\n#         x2 = int((x + width / 2) * image_width)\n#         y2 = int((y + height / 2) * image_height)\n        \n#         # Choose random color for bounding box\n#         color = [random.random() for _ in range(3)]\n        \n#         rect = plt.Rectangle((x1, y1), x2 - x1, y2 - y1, linewidth=2, edgecolor=color, facecolor='none')\n#         ax.add_patch(rect)\n        \n#         # Add label text using class label from dictionary\n#         label_text = class_dict[int(class_id)]\n#         ax.text(x1, y1, label_text, color='white', verticalalignment='top', bbox={'color': color, 'pad': 0})\n    \n#     plt.show()\n\n# def main(image_path, bounding_box_path, class_dict):\n#     # Read image\n#     image = cv2.imread(image_path)\n    \n#     # Read bounding boxes\n#     with open(bounding_box_path, 'r') as file:\n#         bounding_boxes = file.readlines()\n    \n#     # Plot image with bounding boxes\n#     plot_image_with_bounding_box(image, bounding_boxes, class_dict)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:50.539938Z","iopub.execute_input":"2024-05-17T11:13:50.540272Z","iopub.status.idle":"2024-05-17T11:13:50.546113Z","shell.execute_reply.started":"2024-05-17T11:13:50.540245Z","shell.execute_reply":"2024-05-17T11:13:50.545196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# yy[0], label_files[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:51.083793Z","iopub.execute_input":"2024-05-17T11:13:51.084188Z","iopub.status.idle":"2024-05-17T11:13:51.08836Z","shell.execute_reply.started":"2024-05-17T11:13:51.084158Z","shell.execute_reply":"2024-05-17T11:13:51.087367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dict_ = dict(zip(df_yolo['class_id'], df_yolo['class_name']))\n# print(dict_)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:51.244201Z","iopub.execute_input":"2024-05-17T11:13:51.244494Z","iopub.status.idle":"2024-05-17T11:13:51.248531Z","shell.execute_reply.started":"2024-05-17T11:13:51.244469Z","shell.execute_reply":"2024-05-17T11:13:51.247561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img_no = 44\n# main(yy[img_no], label_files[img_no], class_dict = dict_)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:51.680526Z","iopub.execute_input":"2024-05-17T11:13:51.681329Z","iopub.status.idle":"2024-05-17T11:13:51.685176Z","shell.execute_reply.started":"2024-05-17T11:13:51.681296Z","shell.execute_reply":"2024-05-17T11:13:51.684298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img_no = 45\n# main(yy[img_no], label_files[img_no], class_dict = dict_)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:13:52.314147Z","iopub.execute_input":"2024-05-17T11:13:52.314843Z","iopub.status.idle":"2024-05-17T11:13:52.318544Z","shell.execute_reply.started":"2024-05-17T11:13:52.31481Z","shell.execute_reply":"2024-05-17T11:13:52.317595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### In the Second Part, We Will Perform Image Analysis and Download All the Images and Text Files¶","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PART--2) .  YOLO-IMPLEMENTATION:--------","metadata":{}},{"cell_type":"markdown","source":"##  correct label_text_files, and Image_files","metadata":{}},{"cell_type":"markdown","source":"THis df Is The Actual train_data , Now from  this we will create train(90%) and valid(10%)","metadata":{}},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n# df_train, df_valid = train_test_split(df, test_size = 19 , random_state = 42)\n# print(\"shape of df_train\", df_train.shape)\n# print(\"shape of df_valid\", df_valid.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:14:07.297546Z","iopub.execute_input":"2024-05-17T11:14:07.297924Z","iopub.status.idle":"2024-05-17T11:14:07.301867Z","shell.execute_reply.started":"2024-05-17T11:14:07.297881Z","shell.execute_reply":"2024-05-17T11:14:07.30105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cloning Yolo V9 From Github","metadata":{}},{"cell_type":"code","source":"import os\nHOME = \"/kaggle/working/\"  ## Get The Current Working Directory \nprint(HOME)\nos.chdir(HOME)\n\n## Git Clone Yolo V9 \n\n!git clone https://github.com/SkalskiP/yolov9.git\n%cd yolov9\n!pip install -r requirements.txt -q","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:14:08.755949Z","iopub.execute_input":"2024-05-17T11:14:08.756776Z","iopub.status.idle":"2024-05-17T11:14:23.856638Z","shell.execute_reply.started":"2024-05-17T11:14:08.756745Z","shell.execute_reply":"2024-05-17T11:14:23.855664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Downloading Model Weights ","metadata":{}},{"cell_type":"code","source":"!wget -P {HOME}/weights -q https://github.com/WongKinYiu/yolov9/releases/download/v0.1/yolov9-c.pt\n!wget -P {HOME}/weights -q https://github.com/WongKinYiu/yolov9/releases/download/v0.1/yolov9-e.pt\n!wget -P {HOME}/weights -q https://github.com/WongKinYiu/yolov9/releases/download/v0.1/gelan-c.pt\n!wget -P {HOME}/weights -q https://github.com/WongKinYiu/yolov9/releases/download/v0.1/gelan-e.pt","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:14:23.859074Z","iopub.execute_input":"2024-05-17T11:14:23.859456Z","iopub.status.idle":"2024-05-17T11:14:30.736773Z","shell.execute_reply.started":"2024-05-17T11:14:23.859419Z","shell.execute_reply":"2024-05-17T11:14:30.735559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now We Will Create dataset folder COM","metadata":{}},{"cell_type":"code","source":"# cwd = \"/kaggle/working/\"\n# data_train_images = cwd+\"data/train/images\"\n# data_train_labels = cwd+\"data/train/labels\"\n\n# data_valid_images = cwd+\"data/valid/images\"\n# data_valid_labels = cwd+\"data/valid/labels/\"\n\n# os.makedirs(data_train_images, exist_ok = True)\n# os.makedirs(data_train_labels, exist_ok = True)\n\n# os.makedirs(data_valid_images, exist_ok = True)\n# os.makedirs(data_valid_labels, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:14:30.746989Z","iopub.execute_input":"2024-05-17T11:14:30.74764Z","iopub.status.idle":"2024-05-17T11:14:30.843618Z","shell.execute_reply.started":"2024-05-17T11:14:30.747605Z","shell.execute_reply":"2024-05-17T11:14:30.842577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_valid.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:14:30.845072Z","iopub.execute_input":"2024-05-17T11:14:30.84573Z","iopub.status.idle":"2024-05-17T11:14:30.853863Z","shell.execute_reply.started":"2024-05-17T11:14:30.84569Z","shell.execute_reply":"2024-05-17T11:14:30.852955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" Now We have Created data folder","metadata":{}},{"cell_type":"code","source":"####TO CHANGE THE NUMBER OF CLIENTS\nnum_clients = 2\nn = num_clients\n#removing to allow rerun in the same runtime\n! rm -rf /kaggle/working/masterIid\n! rm -rf /kaggle/working/masterNoniid","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:14:30.855266Z","iopub.execute_input":"2024-05-17T11:14:30.855874Z","iopub.status.idle":"2024-05-17T11:14:32.769396Z","shell.execute_reply.started":"2024-05-17T11:14:30.855838Z","shell.execute_reply":"2024-05-17T11:14:32.767981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## non-IID","metadata":{}},{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nimport tqdm\nfrom sklearn.model_selection import train_test_split\n\ndef convert_to_yolo_format(df):\n    # Normalize the bounding box coordinates\n    df['center_x'] = (df['x_min'] + df['x_max']) / 2\n    df['center_y'] = (df['y_min'] + df['y_max']) / 2\n    df['b_box_width'] = df['x_max'] - df['x_min']\n    df['b_box_height'] = df['y_max'] - df['y_min']\n    \n    # Calculate normalized coordinates and dimensions\n    df['normalized_x'] = df['center_x'] / df['width']\n    df['normalized_y'] = df['center_y'] / df['height']\n    df['normalized_width'] = df['b_box_width'] / df['width']\n    df['normalized_height'] = df['b_box_height'] / df['height']\n    \n    return df\n\ndf = pd.read_csv(\"/kaggle/input/vinbigdata-512-image-dataset/vinbigdata/train.csv\")\ndf = df.drop(df[df['class_id'] == 14].index)\ndf = convert_to_yolo_format(df)\n# df.head(3)\n\n\n# Shuffle the DataFrame rows\ndf_shuffled = df.sample(frac=1).reset_index(drop=True)\n\n# print(df_shuffled.columns)\n\ngroup_size = len(df_shuffled) // n  \ngroups = []\n\nfor i in range(n):\n    start_index = i * group_size\n    if i == n - 1:\n        groups.append(df_shuffled.iloc[start_index:])\n    else:\n        groups.append(df_shuffled.iloc[start_index:start_index + group_size])\n\nbasefolder = \"/kaggle/working/masterNoniid\"\nogDataset = \"/kaggle/input/vinbigdata-512-image-dataset/vinbigdata\"\n\nfor i, df in enumerate(groups):\n    # Create directories for images and labels\n    base_path = os.path.join(basefolder, str(i+1))\n    os.makedirs(os.path.join(base_path, \"images/train\"), exist_ok=True)\n    os.makedirs(os.path.join(base_path, \"images/val\"), exist_ok=True)\n    os.makedirs(os.path.join(base_path, \"labels/train\"), exist_ok=True)\n    os.makedirs(os.path.join(base_path, \"labels/val\"), exist_ok=True)\n    \n    # Splitting the DataFrame into train and validation sets (80/20 split)\n    train_df, val_df = train_test_split(df, test_size=0.2)\n    \n    # Save train and validation DataFrames\n    train_df.to_csv(os.path.join(base_path, \"train.csv\"))\n    val_df.to_csv(os.path.join(base_path, \"val.csv\"))\n    \n    # Function to create labels in YOLO format\n    def create_labels(df, type_):\n        for _, row in df.iterrows():\n            class_id = row['class_id']\n            x_center = row['normalized_x']\n            y_center = row['normalized_y']\n            width = row['normalized_width']\n            height = row['normalized_height']\n            image_id = row['image_id']\n            \n            label_path = os.path.join(base_path, f\"labels/{type_}\", f\"{image_id}.txt\")\n            with open(label_path, 'a') as f:\n                f.write(f\"{class_id}\\t{x_center}\\t{y_center}\\t{width}\\t{height}\\n\")\n                \n            shutil.copy(os.path.join(ogDataset, \"train\", f\"{image_id}.png\"), os.path.join(base_path, f\"images/{type_}\", f\"{image_id}.png\"))\n\n            \n    create_labels(train_df, 'train')\n    create_labels(val_df, 'val')\n    \n    print(f\"Processed group {i+1}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:14:32.7711Z","iopub.execute_input":"2024-05-17T11:14:32.771401Z","iopub.status.idle":"2024-05-17T11:15:43.7677Z","shell.execute_reply.started":"2024-05-17T11:14:32.771373Z","shell.execute_reply":"2024-05-17T11:15:43.766667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## IID","metadata":{}},{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nimport tqdm\nfrom sklearn.model_selection import train_test_split\n\ndef convert_to_yolo_format(df):\n    # Normalize the bounding box coordinates\n    df['center_x'] = (df['x_min'] + df['x_max']) / 2\n    df['center_y'] = (df['y_min'] + df['y_max']) / 2\n    df['b_box_width'] = df['x_max'] - df['x_min']\n    df['b_box_height'] = df['y_max'] - df['y_min']\n    \n    # Calculate normalized coordinates and dimensions\n    df['normalized_x'] = df['center_x'] / df['width']\n    df['normalized_y'] = df['center_y'] / df['height']\n    df['normalized_width'] = df['b_box_width'] / df['width']\n    df['normalized_height'] = df['b_box_height'] / df['height']\n    \n    return df\n\ndf = pd.read_csv(\"/kaggle/input/vinbigdata-512-image-dataset/vinbigdata/train.csv\")\ndf = df.drop(df[df['class_id'] == 14].index)\ndf = convert_to_yolo_format(df)\n\ndef create_iid_datasets(df, n, basefolder, ogDataset):\n    # Shuffle the DataFrame\n    df_shuffled = df.sample(frac=1).reset_index(drop=True)\n\n    # Initialize the list to hold \"n\" DataFrames\n    groups = [pd.DataFrame(columns=df.columns) for _ in range(n)]\n\n    # Group by 'class_id' and iterate over each class\n    for class_id, group_df in df_shuffled.groupby('class_id'):\n        # Evenly distribute the images of each class to the groups\n        group_indices = [(len(group_df) // n + (1 if x < len(group_df) % n else 0)) for x in range(n)]\n        start_idx = 0\n        for i in range(n):\n            end_idx = start_idx + group_indices[i]\n            groups[i] = pd.concat([groups[i], group_df.iloc[start_idx:end_idx]])\n            start_idx = end_idx\n\n    # Process each group DataFrame\n    for i, group_df in enumerate(groups):\n        group_df = group_df.reset_index(drop=True)\n        group_path = os.path.join(basefolder, f\"{i+1}\")\n        os.makedirs(group_path, exist_ok=True)\n        os.makedirs(os.path.join(group_path, \"images/train\"), exist_ok=True)\n        os.makedirs(os.path.join(group_path, \"images/val\"), exist_ok=True)\n        os.makedirs(os.path.join(group_path, \"labels/train\"), exist_ok=True)\n        os.makedirs(os.path.join(group_path, \"labels/val\"), exist_ok=True)\n\n        # print(group_df.groupby('class_id').count())\n\n        # Split into train and validation sets\n        train_df, val_df = train_test_split(group_df, test_size=0.2, stratify=group_df['class_id'])\n\n        # Define function to write labels and optionally copy images\n        def write_labels_and_copy_images(df, img_folder, lbl_folder):\n            for _, row in df.iterrows():\n                filename = f\"{row['image_id']}.png\"\n                label_content = f\"{row['class_id']} {row['normalized_x']} {row['normalized_y']} {row['normalized_width']} {row['normalized_height']}\\n\"\n                \n                # Write label file\n                with open(os.path.join(lbl_folder, f\"{row['image_id']}.txt\"), 'a') as label_file:\n                    label_file.write(label_content)\n                \n                # Uncomment the following line to copy the image files\n                # shutil.copy(os.path.join(ogDataset, \"train\", filename), os.path.join(base_path, f\"images/{type_}\", f\"{image_id}.png\"))\n                shutil.copy(os.path.join(ogDataset, \"train\", filename), os.path.join(img_folder, filename))\n\n        # Process train and validation datasets\n        train_img_folder = os.path.join(group_path, \"images/train\")\n        val_img_folder = os.path.join(group_path, \"images/val\")\n        train_lbl_folder = os.path.join(group_path, \"labels/train\")\n        val_lbl_folder = os.path.join(group_path, \"labels/val\")\n\n        write_labels_and_copy_images(train_df, train_img_folder, train_lbl_folder)\n        write_labels_and_copy_images(val_df, val_img_folder, val_lbl_folder)\n\n        print(f\"Group {i+1} processed with {len(train_df)} training and {len(val_df)} validation images.\")\n\n# Call the function with your dataframe and desired parameters\nbasefolder = \"/kaggle/working/masterIid\"\nogDataset = \"/kaggle/input/vinbigdata-512-image-dataset/vinbigdata\"\ncreate_iid_datasets(df, n=num_clients, basefolder=\"./masterIid\", ogDataset=\"/kaggle/input/vinbigdata-512-image-dataset/vinbigdata\")\nbasefolder = \"./masterNoniid/\"","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:15:43.769042Z","iopub.execute_input":"2024-05-17T11:15:43.769335Z","iopub.status.idle":"2024-05-17T11:16:23.09222Z","shell.execute_reply.started":"2024-05-17T11:15:43.769309Z","shell.execute_reply":"2024-05-17T11:16:23.091293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working\nimport shutil\n\n# Move the dataset folder to the Yolo folder\n! rm -rf /kaggle/working/yolov9/masterNoniid\n! rm -rf /kaggle/working/yolov9/masterIid\n\nfor i in range(num_clients):\n    dataset_dir = f\"/kaggle/working/masterNoniid/{i+1}\"\n    yolo_dir = \"/kaggle/working/yolov9\"\n    shutil.move(dataset_dir, yolo_dir)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:16:23.093263Z","iopub.execute_input":"2024-05-17T11:16:23.093552Z","iopub.status.idle":"2024-05-17T11:16:25.93624Z","shell.execute_reply.started":"2024-05-17T11:16:23.093528Z","shell.execute_reply":"2024-05-17T11:16:25.935028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conversion Ends","metadata":{}},{"cell_type":"markdown","source":"## Modify Yaml Code  :----->","metadata":{}},{"cell_type":"code","source":"for i in range(num_clients):\n    yaml_dir = f\"/kaggle/working/yolov9/{i+1}\"\n    print(\"Yaml Directory Is :--->\", yaml_dir)\n    import yaml\n\n    # Data to write to YAML file\n    data = {\n        'names': ['Aortic enlargement',\n                     'Atelectasis',\n                     'Calcification',\n                     'Cardiomegaly',\n                     'Consolidation',\n                     'ILD',\n                     'Infiltration',\n                     'Lung Opacity',\n                     'Nodule/Mass',\n                     'Other lesion',\n                     'Pleural effusion',\n                     'Pleural thickening',\n                     'Pneumothorax',\n                     'Pulmonary fibrosis'],\n        'nc': 14,\n\n        'train': f'{i+1}/images/train',\n        'val': f'{i+1}/images/val'\n    }\n\n    # Write data to YAML file\n    with open(yaml_dir+'/data.yaml', 'w') as file:\n        yaml.dump(data, file)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:16:25.940447Z","iopub.execute_input":"2024-05-17T11:16:25.940761Z","iopub.status.idle":"2024-05-17T11:16:25.975714Z","shell.execute_reply.started":"2024-05-17T11:16:25.940732Z","shell.execute_reply":"2024-05-17T11:16:25.97493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# glob.glob(\"/kaggle/working/yolov9/dataset/train/images/*\")","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:16:25.976644Z","iopub.execute_input":"2024-05-17T11:16:25.976891Z","iopub.status.idle":"2024-05-17T11:16:25.980969Z","shell.execute_reply.started":"2024-05-17T11:16:25.97687Z","shell.execute_reply":"2024-05-17T11:16:25.980092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#creating copies for each client\nfor i in range(num_clients):\n    dest = f\"/kaggle/working/weights/{i+1}.pt\"\n    shutil.copy(\"/kaggle/working/weights/gelan-c.pt\", dest)","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:16:25.982115Z","iopub.execute_input":"2024-05-17T11:16:25.982433Z","iopub.status.idle":"2024-05-17T11:16:26.072943Z","shell.execute_reply.started":"2024-05-17T11:16:25.982409Z","shell.execute_reply":"2024-05-17T11:16:26.071792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Custom Model ","metadata":{}},{"cell_type":"markdown","source":"#### opy the output and put it in the cell below, and run","metadata":{}},{"cell_type":"code","source":"import os\n\n!rm training_commands.txt\n\nnumber_of_iterations = num_clients\noutput_file_path = 'training_commands.txt'\n\nwith open(output_file_path, \"w\") as f:\n    f.write(\"\"\"\n%cd /kaggle/working/yolov9\n    \"\"\")\n\nfor N in range(1, number_of_iterations + 1):\n    command = f\"\"\"\n!python train.py \\\\\n--batch 16 --epochs 1 --img 512 --device 0 --min-items 0 --close-mosaic 15 \\\\\n--data /kaggle/working/yolov9/{N}/data.yaml \\\\\n--weights {{HOME}}/weights/{N}.pt \\\\\n--cfg models/detect/gelan-c.yaml \\\\\n--hyp hyp.scratch-high.yaml\n\ndirs = os.listdir(\"./runs/train\")\nmaxNum = 0\nfor d in dirs:\n    if d[-1].isalpha():\n        continue\n    else:\n        num = int(d[3:])\n        maxNum = max(num, maxNum)\n\n!rm /kaggle/working/weights/{N}.pt\nif maxNum != 0:\n    src = \"./runs/train/exp\" + str(maxNum) + \"/weights/best.pt\"\nelse:\n    src = \"./runs/train/exp/weights/best.pt\"\nshutil.move(src, \"/kaggle/working/weights/{N}.pt\")\n\"\"\"\n    if N > 1:\n        command = '\\n#### PART ' + str(N) + '\\n' + command\n\n    with open(output_file_path, 'a') as file:\n        file.write(command)\n\n# print(f'Commands for {number_of_iterations} iterations have been written to {output_file_path}.')\n\nprint(open(\"training_commands.txt\").read())","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:16:26.075597Z","iopub.execute_input":"2024-05-17T11:16:26.075956Z","iopub.status.idle":"2024-05-17T11:16:27.071823Z","shell.execute_reply.started":"2024-05-17T11:16:26.075903Z","shell.execute_reply":"2024-05-17T11:16:27.07072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### Copy the above output here\n%cd /kaggle/working/yolov9\n    \n!python train.py \\\n--batch 16 --epochs 1 --img 512 --device 0 --min-items 0 --close-mosaic 15 \\\n--data /kaggle/working/yolov9/1/data.yaml \\\n--weights {HOME}/weights/1.pt \\\n--cfg models/detect/gelan-c.yaml \\\n--hyp hyp.scratch-high.yaml\n\ndirs = os.listdir(\"./runs/train\")\nmaxNum = 0\nfor d in dirs:\n    if d[-1].isalpha():\n        continue\n    else:\n        num = int(d[3:])\n        maxNum = max(num, maxNum)\n\n!rm /kaggle/working/weights/1.pt\nif maxNum != 0:\n    src = \"./runs/train/exp\" + str(maxNum) + \"/weights/best.pt\"\nelse:\n    src = \"./runs/train/exp/weights/best.pt\"\nshutil.move(src, \"/kaggle/working/weights/1.pt\")\n\n#### PART 2\n\n!python train.py \\\n--batch 16 --epochs 1 --img 512 --device 0 --min-items 0 --close-mosaic 15 \\\n--data /kaggle/working/yolov9/2/data.yaml \\\n--weights {HOME}/weights/2.pt \\\n--cfg models/detect/gelan-c.yaml \\\n--hyp hyp.scratch-high.yaml\n\ndirs = os.listdir(\"./runs/train\")\nmaxNum = 0\nfor d in dirs:\n    if d[-1].isalpha():\n        continue\n    else:\n        num = int(d[3:])\n        maxNum = max(num, maxNum)\n\n!rm /kaggle/working/weights/2.pt\nif maxNum != 0:\n    src = \"./runs/train/exp\" + str(maxNum) + \"/weights/best.pt\"\nelse:\n    src = \"./runs/train/exp/weights/best.pt\"\nshutil.move(src, \"/kaggle/working/weights/2.pt\")","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:16:27.074867Z","iopub.execute_input":"2024-05-17T11:16:27.075214Z","iopub.status.idle":"2024-05-17T11:26:34.797467Z","shell.execute_reply.started":"2024-05-17T11:16:27.075185Z","shell.execute_reply":"2024-05-17T11:26:34.796259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Running Once","metadata":{}},{"cell_type":"markdown","source":"Averaging","metadata":{}},{"cell_type":"code","source":"!rm pasteInCell.py\nwith open('pasteInCell.py', 'a') as f:\n    s = '''\nimport torch\nimport copy\ndef attempt_load_weights(path):\n    return torch.load(path, map_location=torch.device(0))\n    '''\n    \n    f.write(s)\n    \n    for i in range(num_clients):\n        x = str(i+1)\n        f.write('''\npath_v{} = '/kaggle/working/weights/{}.pt'\nmodel_v{} = attempt_load_weights(path_v{})['model'].state_dict()\n        '''.format(x,x,x,x))\n    string = \"(\"\n    for i in range(num_clients-1):\n        x = str(i+1)\n        string+=\"model_v{}[key] + \".format(x)\n        \n    string+=\"model_v{}[key])/ {}.0\".format(num_clients, num_clients)\n    f.write('''\navg_weights = copy.deepcopy(model_v1)\nfor key in avg_weights.keys():\n    avg_weights[key] =''' + string + '''\nckpt_v1 = attempt_load_weights('/kaggle/working/weights/1.pt')\nckpt_v1['model'].load_state_dict(avg_weights)\ntorch.save(ckpt_v1, '/kaggle/working/weights/yoloAvg.pt')\n    ''')\n    \n    \n# with open(\"pasteInCell.txt\", \"r\") as f:\n#     file = f.read()\n#     print(file)\n\n# !python pasteInCell.py","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:26:34.799307Z","iopub.execute_input":"2024-05-17T11:26:34.799725Z","iopub.status.idle":"2024-05-17T11:26:35.798886Z","shell.execute_reply.started":"2024-05-17T11:26:34.799674Z","shell.execute_reply":"2024-05-17T11:26:35.79772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm val.py\nshutil.copy(\"/kaggle/input/valversion2/val.py\", \"val.py\")\n\nwith open(\"fedOutputs.txt\", \"w\") as f:\n    f.write(\"mAP_0.5, mAP_0.95\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:26:35.800342Z","iopub.execute_input":"2024-05-17T11:26:35.800666Z","iopub.status.idle":"2024-05-17T11:26:36.787126Z","shell.execute_reply.started":"2024-05-17T11:26:35.800637Z","shell.execute_reply":"2024-05-17T11:26:36.786023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !python val.py \\\n# --weights /kaggle/working/weights/yoloAvg.pt \\\n# --batch 8 \\\n# --data /kaggle/working/yolov9/1/data.yaml \\\n# --img 512 --iou-thres 0.65 --conf-thres 0.25 \\\n# --device 0","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:26:36.788683Z","iopub.execute_input":"2024-05-17T11:26:36.789024Z","iopub.status.idle":"2024-05-17T11:26:44.917723Z","shell.execute_reply.started":"2024-05-17T11:26:36.788995Z","shell.execute_reply":"2024-05-17T11:26:44.916645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n%cd /kaggle/working/yolov9\n            \n!rm val.py\nshutil.copy(\"/kaggle/input/valversion2/val.py\", \"val.py\")\n\nwith open(\"fedOutputs.txt\", \"w\") as f:\n    f.write(\"mAP_0.5, mAP_0.95\")\n\n    \n#### PART 1\n\n!python train.py \\\n--batch 16 --epochs 1 --img 512 --device 0 --min-items 0 --close-mosaic 15 \\\n--data /kaggle/working/yolov9/1/data.yaml \\\n--weights {HOME}/weights/1.pt \\\n--cfg models/detect/gelan-c.yaml \\\n--hyp hyp.scratch-high.yaml\n\ndirs = os.listdir(\"./runs/train\")\nmaxNum = 0\nfor d in dirs:\n    if d[-1].isalpha():\n        continue\n    else:\n        num = int(d[3:])\n        maxNum = max(num, maxNum)\n\n!rm /kaggle/working/weights/1.pt\nif maxNum != 0:\n    src = \"./runs/train/exp\" + str(maxNum) + \"/weights/best.pt\"\nelse:\n    src = \"./runs/train/exp/weights/best.pt\"\nshutil.move(src, \"/kaggle/working/weights/1.pt\")\n    \n\n#### PART 2\n\n!python train.py \\\n--batch 16 --epochs 1 --img 512 --device 0 --min-items 0 --close-mosaic 15 \\\n--data /kaggle/working/yolov9/2/data.yaml \\\n--weights {HOME}/weights/2.pt \\\n--cfg models/detect/gelan-c.yaml \\\n--hyp hyp.scratch-high.yaml\n\ndirs = os.listdir(\"./runs/train\")\nmaxNum = 0\nfor d in dirs:\n    if d[-1].isalpha():\n        continue\n    else:\n        num = int(d[3:])\n        maxNum = max(num, maxNum)\n\n!rm /kaggle/working/weights/2.pt\nif maxNum != 0:\n    src = \"./runs/train/exp\" + str(maxNum) + \"/weights/best.pt\"\nelse:\n    src = \"./runs/train/exp/weights/best.pt\"\nshutil.move(src, \"/kaggle/working/weights/2.pt\")\n    \n!python pasteInCell.py\n\n!python val.py \\\n--weights /kaggle/working/weights/yoloAvg.pt \\\n--batch 8 \\\n--data /kaggle/working/yolov9/1/data.yaml \\\n--img 512 --iou-thres 0.65 --conf-thres 0.25 \\\n--device 0\n    \n#### PART 1\n\n!python train.py \\\n--batch 16 --epochs 1 --img 512 --device 0 --min-items 0 --close-mosaic 15 \\\n--data /kaggle/working/yolov9/1/data.yaml \\\n--weights {HOME}/weights/1.pt \\\n--cfg models/detect/gelan-c.yaml \\\n--hyp hyp.scratch-high.yaml\n\ndirs = os.listdir(\"./runs/train\")\nmaxNum = 0\nfor d in dirs:\n    if d[-1].isalpha():\n        continue\n    else:\n        num = int(d[3:])\n        maxNum = max(num, maxNum)\n\n!rm /kaggle/working/weights/1.pt\nif maxNum != 0:\n    src = \"./runs/train/exp\" + str(maxNum) + \"/weights/best.pt\"\nelse:\n    src = \"./runs/train/exp/weights/best.pt\"\nshutil.move(src, \"/kaggle/working/weights/1.pt\")\n    \n\n#### PART 2\n\n!python train.py \\\n--batch 16 --epochs 1 --img 512 --device 0 --min-items 0 --close-mosaic 15 \\\n--data /kaggle/working/yolov9/2/data.yaml \\\n--weights {HOME}/weights/2.pt \\\n--cfg models/detect/gelan-c.yaml \\\n--hyp hyp.scratch-high.yaml\n\ndirs = os.listdir(\"./runs/train\")\nmaxNum = 0\nfor d in dirs:\n    if d[-1].isalpha():\n        continue\n    else:\n        num = int(d[3:])\n        maxNum = max(num, maxNum)\n\n!rm /kaggle/working/weights/2.pt\nif maxNum != 0:\n    src = \"./runs/train/exp\" + str(maxNum) + \"/weights/best.pt\"\nelse:\n    src = \"./runs/train/exp/weights/best.pt\"\nshutil.move(src, \"/kaggle/working/weights/2.pt\")\n    \n!python pasteInCell.py\n\n!python val.py \\\n--weights /kaggle/working/weights/yoloAvg.pt \\\n--batch 8 \\\n--data /kaggle/working/yolov9/1/data.yaml \\\n--img 512 --iou-thres 0.65 --conf-thres 0.25 \\\n--device 0\n    ","metadata":{"execution":{"iopub.status.busy":"2024-05-17T11:26:44.920963Z","iopub.execute_input":"2024-05-17T11:26:44.921271Z","iopub.status.idle":"2024-05-17T11:48:36.756497Z","shell.execute_reply.started":"2024-05-17T11:26:44.921241Z","shell.execute_reply":"2024-05-17T11:48:36.75529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Examine The Training Result:-->","metadata":{}},{"cell_type":"code","source":"!ls {HOME}/yolov9/runs/train/exp/","metadata":{"execution":{"iopub.status.busy":"2024-03-25T07:51:27.907644Z","iopub.execute_input":"2024-03-25T07:51:27.908264Z","iopub.status.idle":"2024-03-25T07:51:28.882135Z","shell.execute_reply.started":"2024-03-25T07:51:27.908233Z","shell.execute_reply":"2024-03-25T07:51:28.880995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Image\n\nImage(filename=f\"{HOME}/yolov9/runs/train/exp/results.png\", width=700)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Image\n\nImage(filename=f\"{HOME}/yolov9/runs/train/exp/confusion_matrix.png\", width=1000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import Image\n\nImage(filename=f\"{HOME}/yolov9/runs/train/exp/val_batch0_pred.jpg\", width=1000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Demo :--->","metadata":{}},{"cell_type":"code","source":"!python /kaggle/working/yolov9/detect.py \\\n    --weights /kaggle/input/yaml-model/yoloAvg.pt \\\n    --data /kaggle/input/yaml-model/data.yaml \\\n    --img 512 \\\n    --iou-thres 0.65 \\\n    --conf-thres 0.25 \\\n    --device 0 \\\n    --batch 8 \\\n    --source /kaggle/input/vinbigdata-512-image-dataset/vinbigdata/train/915e7fbda46ac956e163de4485710390.png","metadata":{"execution":{"iopub.status.busy":"2024-05-17T12:01:40.5485Z","iopub.execute_input":"2024-05-17T12:01:40.548892Z","iopub.status.idle":"2024-05-17T12:01:48.089042Z","shell.execute_reply.started":"2024-05-17T12:01:40.54886Z","shell.execute_reply":"2024-05-17T12:01:48.087981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}