{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":37333,"databundleVersionId":3949526,"sourceType":"competition"},{"sourceId":7274530,"sourceType":"datasetVersion","datasetId":4217325},{"sourceId":7274693,"sourceType":"datasetVersion","datasetId":4217421}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!apt -y update && apt -y upgrade","metadata":{"_uuid":"5dbbca4e-8a76-4f50-a89f-9b3a080da72b","_cell_guid":"68b32583-77c9-4dca-88b2-2a63034eda14","_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-02-15T17:03:09.349965Z","iopub.execute_input":"2024-02-15T17:03:09.350322Z","iopub.status.idle":"2024-02-15T17:03:13.864326Z","shell.execute_reply.started":"2024-02-15T17:03:09.350281Z","shell.execute_reply":"2024-02-15T17:03:13.863069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt -y install -d -o=dir::cache=/kaggle/working libvips libvips-dev libvips-tools","metadata":{"_uuid":"5c59a829-12b2-473b-9363-11a533a23cdb","_cell_guid":"ad69f1a1-32bd-4fe3-8584-4235075aa1bc","execution":{"iopub.status.busy":"2024-02-15T17:03:13.866618Z","iopub.execute_input":"2024-02-15T17:03:13.866905Z","iopub.status.idle":"2024-02-15T17:03:16.142369Z","shell.execute_reply.started":"2024-02-15T17:03:13.86688Z","shell.execute_reply":"2024-02-15T17:03:16.141112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip3 install --upgrade pip\n!pip3 download -d /kaggle/working pyvips","metadata":{"_uuid":"8f008194-acb2-44e1-ae15-55ea095ed8f6","_cell_guid":"af9232cc-9853-4e69-a653-8a8473a94319","execution":{"iopub.status.busy":"2024-02-15T17:03:16.143785Z","iopub.execute_input":"2024-02-15T17:03:16.144101Z","iopub.status.idle":"2024-02-15T17:03:32.876027Z","shell.execute_reply.started":"2024-02-15T17:03:16.144075Z","shell.execute_reply":"2024-02-15T17:03:32.874783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyvips","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:10:52.68214Z","iopub.execute_input":"2024-02-15T17:10:52.682461Z","iopub.status.idle":"2024-02-15T17:11:04.544432Z","shell.execute_reply.started":"2024-02-15T17:10:52.682434Z","shell.execute_reply":"2024-02-15T17:11:04.543377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!dpkg -i --force-depends ./archives/*.deb >/dev/null 2>&1\n# !pip3 install --quiet ./pycparser-2.21-py2.py3-none-any.whl\n# !pip3 install --quiet ./pyvips-2.2.1.tar.gz","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:10:11.951761Z","iopub.execute_input":"2024-02-15T17:10:11.952547Z","iopub.status.idle":"2024-02-15T17:10:52.68025Z","shell.execute_reply.started":"2024-02-15T17:10:11.952514Z","shell.execute_reply":"2024-02-15T17:10:52.67914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/working/cffi-1.16.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install /kaggle/working/pyvips-2.2.2.tar.gz\n!pip install /kaggle/working/pycparser-2.21-py2.py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:08:37.055506Z","iopub.execute_input":"2024-02-15T17:08:37.05644Z","iopub.status.idle":"2024-02-15T17:09:19.399324Z","shell.execute_reply.started":"2024-02-15T17:08:37.056406Z","shell.execute_reply":"2024-02-15T17:09:19.39827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyvips\nimport os\nimport cv2\nimport sys\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport zipfile\nfrom IPython.display import FileLink\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nimport tensorflow.keras.layers as l\nimport albumentations as A\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"_uuid":"b0f1850d-6943-46e9-b9a4-62bb8798ffd2","_cell_guid":"32d83000-de5f-48ac-8ce9-b77203d05498","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-15T17:11:12.116275Z","iopub.execute_input":"2024-02-15T17:11:12.116658Z","iopub.status.idle":"2024-02-15T17:11:12.123371Z","shell.execute_reply.started":"2024-02-15T17:11:12.11663Z","shell.execute_reply":"2024-02-15T17:11:12.122247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_and_resize_image(image_path):\n    # check if image is less than limit\n    if os.path.getsize(image_path) < 178956970:\n        scale_factor = 1.0\n    \n    else:\n        # Calculate the scale factor needed to achieve the target size\n        scale_factor = np.sqrt(178956970 / os.path.getsize(image_path))\n    \n    img = pyvips.Image.new_from_file(image_path, access='sequential')\n    resized_img = img.resize(1.0 / scale_factor)\n    \n    return resized_img","metadata":{"_uuid":"32663662-4baf-4949-bf17-f123ef295662","_cell_guid":"342b74a6-1265-4be8-a15b-86319f9ee70f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-15T17:11:15.211795Z","iopub.execute_input":"2024-02-15T17:11:15.212166Z","iopub.status.idle":"2024-02-15T17:11:15.21834Z","shell.execute_reply.started":"2024-02-15T17:11:15.212134Z","shell.execute_reply":"2024-02-15T17:11:15.217275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def divide_tiff_into_tiles(input_path, tile_size):\n    img = read_and_resize_image(input_path)\n\n    # Get the size of the input image\n    img_width = img.width\n    img_height = img.height\n\n    # Calculate the number of tiles in the x and y directions\n    num_tiles_x = img_width // tile_size[0]\n    num_tiles_y = img_height // tile_size[1]\n\n    # Initialize an empty list to store the tiles\n    tiles = []\n\n    # Iterate over each tile and append it to the list\n    for y in range(num_tiles_y):\n        for x in range(num_tiles_x):\n            # Define the region for the current tile\n            left = x * tile_size[0]\n            upper = y * tile_size[1]\n            width = tile_size[0]\n            height = tile_size[1]\n\n            # Crop the image to the current tile\n            tile = img.crop(left, upper, width, height)\n\n            # Remove tiles with average intensity less than threshold\n            if tile.avg() < 185:\n                # Append the tile to the list\n                tiles.append(tile)\n\n    return tiles","metadata":{"_uuid":"7199eb12-f086-4d7e-b19e-bf32c9e1c2d9","_cell_guid":"d2e7b6fa-de5b-4038-8ee8-b261db354cda","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-15T17:11:15.52875Z","iopub.execute_input":"2024-02-15T17:11:15.529107Z","iopub.status.idle":"2024-02-15T17:11:15.536475Z","shell.execute_reply.started":"2024-02-15T17:11:15.529079Z","shell.execute_reply":"2024-02-15T17:11:15.535559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/mayo-clinic-strip-ai/train.csv')\ntrain_csv[\"label\"] = train_csv[\"label\"].map({'CE': 0, 'LAA': 1})","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:11:15.666684Z","iopub.execute_input":"2024-02-15T17:11:15.66748Z","iopub.status.idle":"2024-02-15T17:11:15.67882Z","shell.execute_reply.started":"2024-02-15T17:11:15.667448Z","shell.execute_reply":"2024-02-15T17:11:15.677851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:11:15.830229Z","iopub.execute_input":"2024-02-15T17:11:15.830566Z","iopub.status.idle":"2024-02-15T17:11:15.845084Z","shell.execute_reply.started":"2024-02-15T17:11:15.830539Z","shell.execute_reply":"2024-02-15T17:11:15.844101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lookup_dic = {}\nfor i in range(len(train_csv)):\n    lookup_dic[train_csv[\"image_id\"].iloc[i]] = train_csv[\"label\"].iloc[i]","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:11:16.137435Z","iopub.execute_input":"2024-02-15T17:11:16.137754Z","iopub.status.idle":"2024-02-15T17:11:16.166699Z","shell.execute_reply.started":"2024-02-15T17:11:16.137728Z","shell.execute_reply":"2024-02-15T17:11:16.165851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the HSV range for yellow\nlower_yellow = np.array([20, 100, 100])\nupper_yellow = np.array([30, 255, 255])\n\n# Define the HSV range for brown\nlower_brown = np.array([10, 50, 50])\nupper_brown = np.array([20, 255, 255])","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:11:16.508099Z","iopub.execute_input":"2024-02-15T17:11:16.508711Z","iopub.status.idle":"2024-02-15T17:11:16.513734Z","shell.execute_reply.started":"2024-02-15T17:11:16.508678Z","shell.execute_reply":"2024-02-15T17:11:16.512811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder_path = \"/kaggle/input/mayo-clinic-strip-ai/train\"\nnames, RBC_ratios, WBC_ratios, labels = [], [], [], []\nt = 0\nall_files = os.listdir(folder_path)\n\nfor filename in all_files[100:102]:\n    fname, _ = os.path.splitext(filename)\n    img_path = os.path.join(folder_path, f\"{fname}.tif\")\n    tiles = divide_tiff_into_tiles(img_path, (512, 512))\n    total_yellow_brown_area = 0\n    total_tile_area = 0\n\n    if len(tiles) == 0:\n        continue\n\n    for i in range(len(tiles)):\n        stained_image = np.array(tiles[i])\n        hsv_image = cv2.cvtColor(stained_image, cv2.COLOR_RGB2HSV)\n\n        yellow_mask = cv2.inRange(hsv_image, lower_yellow, upper_yellow)\n        brown_mask = cv2.inRange(hsv_image, lower_brown, upper_brown)\n\n        yellow_brown_mask = cv2.bitwise_or(yellow_mask, brown_mask)\n        other_colors_mask = cv2.bitwise_not(yellow_brown_mask)\n\n        yellow_brown_area = np.sum(yellow_brown_mask > 0)\n        other_colors_area = np.sum(other_colors_mask > 0)\n\n        tile_area = yellow_brown_area + other_colors_area\n\n        total_yellow_brown_area += yellow_brown_area\n        other_colors_area += other_colors_area\n        total_tile_area += tile_area\n\n    RBC_ratio = total_yellow_brown_area / total_tile_area\n    WBC_ratio = 0  # Since we're ignoring the violet mask, set WBC ratio to 0\n\n    labels.append(lookup_dic[fname])\n    names.append(fname)\n    RBC_ratios.append(RBC_ratio)\n    WBC_ratios.append(WBC_ratio)\n    t += 1\n    print(t)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:11:16.824813Z","iopub.execute_input":"2024-02-15T17:11:16.825662Z","iopub.status.idle":"2024-02-15T17:13:17.111729Z","shell.execute_reply.started":"2024-02-15T17:11:16.825632Z","shell.execute_reply":"2024-02-15T17:13:17.110712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(RBC_ratios),len(labels),len(names), len(RBC_ratios), len(labels), len(names)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:13:17.113946Z","iopub.execute_input":"2024-02-15T17:13:17.114612Z","iopub.status.idle":"2024-02-15T17:13:17.121664Z","shell.execute_reply.started":"2024-02-15T17:13:17.114572Z","shell.execute_reply":"2024-02-15T17:13:17.120765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = {'Name': names, 'RBC_Ratio': RBC_ratios, 'Label': labels}\ndf = pd.DataFrame(data)\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:13:17.1228Z","iopub.execute_input":"2024-02-15T17:13:17.123096Z","iopub.status.idle":"2024-02-15T17:13:17.140545Z","shell.execute_reply.started":"2024-02-15T17:13:17.123071Z","shell.execute_reply":"2024-02-15T17:13:17.139606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom IPython.display import FileLink\n\n# Assuming df is your Pandas DataFrame\n\n# Save the DataFrame to a CSV file\ndf.to_csv('/kaggle/working/my_dataframe2_100_120.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:13:17.142932Z","iopub.execute_input":"2024-02-15T17:13:17.143398Z","iopub.status.idle":"2024-02-15T17:13:17.15452Z","shell.execute_reply.started":"2024-02-15T17:13:17.143361Z","shell.execute_reply":"2024-02-15T17:13:17.153375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The End","metadata":{}},{"cell_type":"code","source":"len(tiles)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T15:33:20.81959Z","iopub.status.idle":"2024-02-15T15:33:20.820047Z","shell.execute_reply.started":"2024-02-15T15:33:20.819829Z","shell.execute_reply":"2024-02-15T15:33:20.819852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport cv2\nimport os\n\n# Assume you already have the definitions for 'stained_image', 'lower_yellow', 'upper_yellow', \n# 'lower_brown', 'upper_brown', and 'hsv_image' from the previous code snippets\n\n# Create masks for yellow and brown\nyellow_mask = cv2.inRange(hsv_image, lower_yellow, upper_yellow)\nbrown_mask = cv2.inRange(hsv_image, lower_brown, upper_brown)\n\n# Combine the masks to get the yellow and brown regions\nyellow_brown_mask = cv2.bitwise_or(yellow_mask, brown_mask)\n\n# Apply the masks to the original image\nyellow_brown_part = cv2.bitwise_and(stained_image, stained_image, mask=yellow_brown_mask)\n\n# Create a mask image where yellow and brown parts are 1, and the rest are 0\nmask_image = yellow_brown_mask.copy()\nmask_image[mask_image > 0] = 1\n\n# Display the original image and the separated parts\nplt.figure(figsize=(12, 6))\n\nplt.subplot(1, 3, 1)\nplt.imshow(stained_image)\nplt.title('Original Image')\nplt.axis('off')\n\nplt.subplot(1, 3, 2)\nplt.imshow(yellow_brown_part)\nplt.title('Yellow and Brown Parts')\nplt.axis('off')\n\nplt.subplot(1, 3, 3)\nplt.imshow(mask_image, cmap='gray')\nplt.title('Mask Image')\nplt.axis('off')\n\nplt.show()\n\n# Save the original image and mask image to folders\noriginal_images_folder = \"/kaggle/working/original_images\"\nmasked_images_folder = \"/kaggle/working/masked_images\"\n\n# Create folders if they don't exist\nos.makedirs(original_images_folder, exist_ok=True)\nos.makedirs(masked_images_folder, exist_ok=True)\n\n# Save original image\noriginal_image_path = os.path.join(original_images_folder, \"original_image.jpg\")\ncv2.imwrite(original_image_path, cv2.cvtColor(stained_image, cv2.COLOR_RGB2BGR))\n\n# Save mask image\nmask_image_path = os.path.join(masked_images_folder, \"mask_image.jpg\")\ncv2.imwrite(mask_image_path, mask_image * 255)  # Multiply by 255 to convert 1 to 255 (white)\n\nprint(\"Original image and mask image saved successfully.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T14:19:19.651422Z","iopub.execute_input":"2024-02-15T14:19:19.652415Z","iopub.status.idle":"2024-02-15T14:19:20.118818Z","shell.execute_reply.started":"2024-02-15T14:19:19.652374Z","shell.execute_reply":"2024-02-15T14:19:20.117703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n\n# # Assume you already have the definitions for 'stained_image', 'lower_yellow', 'upper_yellow', \n# # 'lower_brown', 'upper_brown' from the previous code snippets\n\n# # Create masks for yellow and brown\n# yellow_mask = cv2.inRange(hsv_image, lower_yellow, upper_yellow)\n# brown_mask = cv2.inRange(hsv_image, lower_brown, upper_brown)\n\n# # Combine the masks to get the yellow and brown regions\n# yellow_brown_mask = cv2.bitwise_or(yellow_mask, brown_mask)\n\n# # Apply the masks to the original image\n# yellow_brown_part = cv2.bitwise_and(stained_image, stained_image, mask=yellow_brown_mask)\n\n# # Display the original image and the separated parts\n# plt.figure(figsize=(8, 4))\n\n# plt.subplot(1, 2, 1)\n# plt.imshow(stained_image)\n# plt.title('Original Image')\n# plt.axis('off')\n\n# plt.subplot(1, 2, 2)\n# plt.imshow(yellow_brown_part)\n# plt.title('Yellow and Brown Parts')\n# plt.axis('off')\n\n# plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:02:42.228621Z","iopub.execute_input":"2024-02-15T17:02:42.22901Z","iopub.status.idle":"2024-02-15T17:02:42.27334Z","shell.execute_reply.started":"2024-02-15T17:02:42.228978Z","shell.execute_reply":"2024-02-15T17:02:42.272135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# import cv2\n# import os\n\n\n\n# # Define your lower and upper threshold values for yellow and brown\n# lower_yellow = np.array([20, 100, 100])\n# upper_yellow = np.array([30, 255, 255])\n# lower_brown = np.array([10, 100, 20])\n# upper_brown = np.array([20, 255, 200])\n\n# # Define the folder path containing the TIFF images\n# folder_path = \"/kaggle/input/mayo-clinic-strip-ai/train\"\n\n# # Iterate over TIFF files in the folder\n# for filename in os.listdir(folder_path)[:1]:  # Limiting to first two files for demonstration\n#     fname, _ = os.path.splitext(filename)\n#     img_path = os.path.join(folder_path, f\"{fname}.tif\")\n#     tiles = divide_tiff_into_tiles(img_path, (512, 512))\n\n#     if len(tiles) == 0:\n#         continue\n\n#     # Iterate over each tile\n#     for idx, tile in enumerate(tiles):\n#         stained_image = np.array(tile)\n#         hsv_image = cv2.cvtColor(stained_image, cv2.COLOR_RGB2HSV)\n\n#         # Create masks for yellow and brown\n#         yellow_mask = cv2.inRange(hsv_image, lower_yellow, upper_yellow)\n#         brown_mask = cv2.inRange(hsv_image, lower_brown, upper_brown)\n\n#         # Combine the masks to get the yellow and brown regions\n#         yellow_brown_mask = cv2.bitwise_or(yellow_mask, brown_mask)\n\n#         # Apply the masks to the original image\n#         yellow_brown_part = cv2.bitwise_and(stained_image, stained_image, mask=yellow_brown_mask)\n\n#         # Create a mask image where yellow and brown parts are 1, and the rest are 0\n#         mask_image = yellow_brown_mask.copy()\n#         mask_image[mask_image > 0] = 1\n\n#         # Save the original image and mask image to folders\n#         original_images_folder = \"/kaggle/working/original_images\"\n#         masked_images_folder = \"/kaggle/working/masked_images\"\n\n#         # Create folders if they don't exist\n#         os.makedirs(original_images_folder, exist_ok=True)\n#         os.makedirs(masked_images_folder, exist_ok=True)\n\n#         # Save original image\n#         original_image_path = os.path.join(original_images_folder, f\"{fname}_{idx}_original.jpg\")\n#         cv2.imwrite(original_image_path, cv2.cvtColor(stained_image, cv2.COLOR_RGB2BGR))\n\n#         # Save mask image\n#         mask_image_path = os.path.join(masked_images_folder, f\"{fname}_{idx}_mask.jpg\")\n#         cv2.imwrite(mask_image_path, mask_image * 255)  # Multiply by 255 to convert 1 to 255 (white)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T17:13:17.155911Z","iopub.execute_input":"2024-02-15T17:13:17.156182Z","iopub.status.idle":"2024-02-15T17:15:50.270769Z","shell.execute_reply.started":"2024-02-15T17:13:17.156159Z","shell.execute_reply":"2024-02-15T17:15:50.269859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pillow","metadata":{"execution":{"iopub.status.busy":"2024-02-15T15:27:26.552063Z","iopub.execute_input":"2024-02-15T15:27:26.552327Z","iopub.status.idle":"2024-02-15T15:27:39.509936Z","shell.execute_reply.started":"2024-02-15T15:27:26.552301Z","shell.execute_reply":"2024-02-15T15:27:39.508865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from PIL import Image\n\n# # Define your lower and upper threshold values for yellow and brown\n# lower_yellow = np.array([20, 100, 100])\n# upper_yellow = np.array([30, 255, 255])\n# lower_brown = np.array([10, 100, 20])\n# upper_brown = np.array([20, 255, 200])\n\n# # Define the input path for the TIFF image\n# image_path = \"/kaggle/input/mayo-clinic-strip-ai/train/008e5c_0.tif\"\n\n# # Define the tile size\n# tile_size = (512, 512)\n\n# # Divide the TIFF image into tiles\n# tiles = divide_tiff_into_tiles(image_path, tile_size)\n\n# # Initialize empty array to store the processed tiles\n# processed_tiles = []\n\n# # Apply the combined mask to each tile individually\n# for tile in tiles:\n#     # Convert tile to numpy array\n#     tile_np = np.array(tile)\n#     hsv_image = cv2.cvtColor(tile_np, cv2.COLOR_RGB2HSV)\n#     yellow_mask = cv2.inRange(hsv_image, lower_yellow, upper_yellow)\n#     brown_mask = cv2.inRange(hsv_image, lower_brown, upper_brown)\n    \n#     # Combine the masks\n#     yellow_brown_mask = cv2.bitwise_or(yellow_mask, brown_mask)\n    \n#     # Apply the mask to the tile\n#     masked_tile = cv2.bitwise_and(tile_np, tile_np, mask=yellow_brown_mask)\n    \n#     # Append the processed tile to the list\n#     processed_tiles.append(masked_tile)\n\n# # Concatenate the processed tiles into a single image\n# processed_image = np.concatenate(processed_tiles, axis=1)\n\n# # Resize the image to make it low resolution\n# scale_percent = 10  # Define the scale percentage\n# width = int(processed_image.shape[1] * scale_percent / 100)\n# height = int(processed_image.shape[0] * scale_percent / 100)\n# resized_image = cv2.resize(processed_image, (width, height), interpolation=cv2.INTER_AREA)\n\n# import cv2\n\n# # Define the output path for the resized image\n# output_image_path = \"/kaggle/working/yellow_brown_low_resolution.jpg\"\n\n# # Save the resized image\n# cv2.imwrite(output_image_path, resized_image)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T14:01:41.388363Z","iopub.status.idle":"2024-02-15T14:01:41.388712Z","shell.execute_reply.started":"2024-02-15T14:01:41.388542Z","shell.execute_reply":"2024-02-15T14:01:41.388559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Select one of the processed tiles (e.g., the first one)\nselected_tile = processed_tiles[0]\n\n# Display the selected tile\nplt.imshow(selected_tile)\nplt.title('Processed Tile with Yellow and Brown Parts')\nplt.axis('off')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T14:01:41.389702Z","iopub.status.idle":"2024-02-15T14:01:41.390039Z","shell.execute_reply.started":"2024-02-15T14:01:41.38987Z","shell.execute_reply":"2024-02-15T14:01:41.389887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyradiomics","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:49:54.189381Z","iopub.execute_input":"2024-02-15T16:49:54.189731Z","iopub.status.idle":"2024-02-15T16:50:43.065631Z","shell.execute_reply.started":"2024-02-15T16:49:54.189701Z","shell.execute_reply":"2024-02-15T16:50:43.064353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(other_colors_mask)","metadata":{"execution":{"iopub.status.busy":"2023-12-26T12:35:32.701778Z","iopub.execute_input":"2023-12-26T12:35:32.702169Z","iopub.status.idle":"2023-12-26T12:35:32.719178Z","shell.execute_reply.started":"2023-12-26T12:35:32.702134Z","shell.execute_reply":"2023-12-26T12:35:32.71798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport SimpleITK as sitk\nfrom radiomics import featureextractor\nimport csv\nfrom radiomics import imageoperations\nimport numpy as np\nimport SimpleITK as sitk\nimport pandas as pd\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-02-15T16:52:08.145556Z","iopub.execute_input":"2024-02-15T16:52:08.146225Z","iopub.status.idle":"2024-02-15T16:52:08.150827Z","shell.execute_reply.started":"2024-02-15T16:52:08.146196Z","shell.execute_reply":"2024-02-15T16:52:08.150012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import cv2\n# import pandas as pd\n# import SimpleITK as sitk\n# from radiomics import featureextractor\n# # \n# # Define the directories for original images and masked images\n# original_images_folder = \"/kaggle/working/original_images\"\n# masked_images_folder = \"/kaggle/working/masked_images\"\n\n# # Load the train CSV data into a DataFrame (assuming `train_csv` is already defined)\n# # train_df = train_csv\n\n# # Initialize PyRadiomics feature extractor\n# extractor = featureextractor.RadiomicsFeatureExtractor()\n\n# # Initialize list to store PyRadiomics data for all images\n# pyradiomics_data_for_all_images = []\n\n# # List all files in the original images directory\n# original_image_files = os.listdir(original_images_folder)\n\n# # Iterate over each image file\n# for image_file in original_image_files:\n#     # Extract image ID from file name\n#     image_id = os.path.splitext(image_file)[0]\n\n#     # Remove \"original\" from the image ID to match the masked image file name\n#     image_id = image_id.replace(\"_original\", \"\")\n\n#     # Read original image and corresponding mask\n#     original_image_path = os.path.join(original_images_folder, image_file)\n#     mask_image_path = os.path.join(masked_images_folder, f\"{image_id}_mask.jpg\")\n#     # Check if image files exist\n#     if not os.path.exists(original_image_path) or not os.path.exists(mask_image_path):\n#         print(f\"Image files not found for image {image_id}\")\n#         continue\n\n#     try:\n#         # Read images\n#         original_image = cv2.imread(original_image_path)\n#         mask_image = cv2.imread(mask_image_path, cv2.IMREAD_GRAYSCALE)\n\n#         # Convert images to SimpleITK format\n#         original_image_sitk = sitk.GetImageFromArray(cv2.cvtColor(original_image, cv2.COLOR_BGR2GRAY))\n#         mask_image_sitk = sitk.GetImageFromArray(mask_image)\n\n#         # Extract features using PyRadiomics\n#         features = extractor.execute(original_image_sitk, mask_image_sitk)\n\n#         # Append data to list\n#         pyradiomics_data_for_all_images.append({\n#             'Image_ID': image_id,\n#             **features\n#         })\n#     except Exception as e:\n#         print(f\"Error extracting features for image {image_id}: {e}\")\n#         continue\n\n# # Create DataFrame from PyRadiomics data\n# pyradiomics_df_for_all_images = pd.DataFrame(pyradiomics_data_for_all_images)\n\n# # Save DataFrame to CSV file\n# pyradiomics_df_for_all_images.to_csv(\"/kaggle/working/pyradiomics_features.csv\", index=False)\n\n# print(\"PyRadiomics features saved successfully.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T18:13:38.377388Z","iopub.status.idle":"2024-02-15T18:13:38.377829Z","shell.execute_reply.started":"2024-02-15T18:13:38.377614Z","shell.execute_reply":"2024-02-15T18:13:38.377638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result","metadata":{"execution":{"iopub.status.busy":"2023-12-26T12:35:34.142973Z","iopub.execute_input":"2023-12-26T12:35:34.143414Z","iopub.status.idle":"2023-12-26T12:35:34.171504Z","shell.execute_reply.started":"2023-12-26T12:35:34.143361Z","shell.execute_reply":"2023-12-26T12:35:34.170148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"0531006b-ffcf-4524-89db-5cc2a22a1051","_cell_guid":"64062482-0a85-4718-98b8-f5390c0d19fc","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"86402e45-ce8a-4f9c-b4f6-2d5055eccdf3","_cell_guid":"2b861f68-494f-42e8-b35a-e74ce4352d92","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# total_yellow_brown_area = 0\n# total_other_colors_area = 0\n# total_tile_area = 0\n\n# # Define the HSV range for yellow\n# lower_yellow = np.array([20, 100, 100])\n# upper_yellow = np.array([30, 255, 255])\n\n# # Define the HSV range for brown\n# lower_brown = np.array([10, 50, 50])\n# upper_brown = np.array([20, 255, 255])\n\n# for i in range(len(tiles)):\n#             stained_image = np.array(tiles[i])  \n#             # Convert the image to the HSV color space\n#             hsv_image = cv2.cvtColor(stained_image, cv2.COLOR_RGB2HSV)\n\n            \n#             # Create masks for yellow and brown\n#             yellow_mask = cv2.inRange(hsv_image, lower_yellow, upper_yellow)\n#             brown_mask = cv2.inRange(hsv_image, lower_brown, upper_brown)\n\n#             # Combine the masks to get the yellow and brown regions\n#             yellow_brown_mask = cv2.bitwise_or(yellow_mask, brown_mask)\n#             other_colors_mask = cv2.bitwise_not(yellow_brown_mask)\n\n#             # Count yellow-brown area for the current tile\n#             yellow_brown_area = np.sum(yellow_brown_mask > 0)\n#             # Count other colors area for the current tile\n#             other_colors_area = np.sum(other_colors_mask > 0)\n#             tile_area = yellow_brown_area + other_colors_area\n    \n#             # Sum up areas for all tiles\n#             total_yellow_brown_area += yellow_brown_area\n#             total_other_colors_area += other_colors_area\n#             total_tile_area += tile_area\n            \n            \n\n# RBC_ratio = total_yellow_brown_area / total_tile_area\n# print(\"RBC Area Ratio:\", RBC_ratio)           \n# #Other ce","metadata":{"_uuid":"1f5204c4-58de-47ad-8adb-a077ad30ccbd","_cell_guid":"5f35d183-9c6b-4b9b-932b-17620daf041d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-12-25T21:52:56.004482Z","iopub.status.idle":"2023-12-25T21:52:56.004919Z","shell.execute_reply.started":"2023-12-25T21:52:56.004705Z","shell.execute_reply":"2023-12-25T21:52:56.004723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport pandas as pd\nimport SimpleITK as sitk\nfrom radiomics import featureextractor\nimport shutil\n\n# Define directories for storing original and masked images\noriginal_images_folder = \"/kaggle/working/original_images\"\nmasked_images_folder = \"/kaggle/working/masked_images\"\nfeatures_folder = \"/kaggle/working/features\"\n\n# Create directories if they don't exist\nos.makedirs(original_images_folder, exist_ok=True)\nos.makedirs(masked_images_folder, exist_ok=True)\nos.makedirs(features_folder, exist_ok=True)\n\n# Define your lower and upper threshold values for yellow and brown\nlower_yellow = np.array([20, 100, 100])\nupper_yellow = np.array([30, 255, 255])\nlower_brown = np.array([10, 100, 20])\nupper_brown = np.array([20, 255, 200])\n\n# Define the folder path containing the TIFF images\nfolder_path = \"/kaggle/input/mayo-clinic-strip-ai/train\"\n\n# Sort the list of filenames in folder_path\nfilenames = sorted(os.listdir(folder_path))\n\n\n# Initialize PyRadiomics feature extractor\nextractor = featureextractor.RadiomicsFeatureExtractor()\n\n# Iterate over TIFF files in the folder\nfor filename in filenames[1:5]:  # Process first 5 files\n    fname, _ = os.path.splitext(filename)\n    img_path = os.path.join(folder_path, f\"{fname}.tif\")\n    tiles = divide_tiff_into_tiles(img_path, (512, 512))\n\n    if len(tiles) == 0:\n        continue\n\n    # Initialize list to store PyRadiomics data for all tiles of the current image\n    pyradiomics_data_for_current_image = []\n\n    # Get label for the current image\n    label = train_csv[train_csv[\"image_id\"] == fname][\"label\"].iloc[0]\n\n    # Iterate over each tile\n    for idx, tile in enumerate(tiles):\n        stained_image = np.array(tile)\n        hsv_image = cv2.cvtColor(stained_image, cv2.COLOR_RGB2HSV)\n\n        # Create masks for yellow and brown\n        yellow_mask = cv2.inRange(hsv_image, lower_yellow, upper_yellow)\n        brown_mask = cv2.inRange(hsv_image, lower_brown, upper_brown)\n\n        # Combine the masks to get the yellow and brown regions\n        yellow_brown_mask = cv2.bitwise_or(yellow_mask, brown_mask)\n\n        # Apply the masks to the original image\n        yellow_brown_part = cv2.bitwise_and(stained_image, stained_image, mask=yellow_brown_mask)\n\n        # Create a mask image where yellow and brown parts are 1, and the rest are 0\n        mask_image = yellow_brown_mask.copy()\n        mask_image[mask_image > 0] = 1\n\n        # Save original image\n        original_image_path = os.path.join(original_images_folder, f\"{fname}_{idx}_original.jpg\")\n        cv2.imwrite(original_image_path, cv2.cvtColor(stained_image, cv2.COLOR_RGB2BGR))\n\n        # Save mask image\n        mask_image_path = os.path.join(masked_images_folder, f\"{fname}_{idx}_mask.jpg\")\n        cv2.imwrite(mask_image_path, mask_image * 255)  # Multiply by 255 to convert 1 to 255 (white)\n\n        print(f\"Original image and mask image for tile {idx} of {fname} saved successfully.\")\n\n        try:\n            # Read original image\n            original_image = cv2.imread(original_image_path)\n\n            # Read mask image\n            mask_image = cv2.imread(mask_image_path, cv2.IMREAD_GRAYSCALE)\n\n            # Convert images to SimpleITK format\n            original_image_sitk = sitk.GetImageFromArray(cv2.cvtColor(original_image, cv2.COLOR_BGR2GRAY))\n            mask_image_sitk = sitk.GetImageFromArray(mask_image)\n\n            # Extract features using PyRadiomics\n            features = extractor.execute(original_image_sitk, mask_image_sitk)\n\n            # Append data to list\n            pyradiomics_data_for_current_image.append({\n                'Image_ID': f\"{fname}_{idx}\",\n                'Label': label,\n                **features\n            })\n\n            print(f\"PyRadiomics features for tile {idx} of image {fname} saved successfully.\")\n        except Exception as e:\n            print(f\"Error extracting features for tile {idx} of image {fname}: {e}\")\n            continue\n\n    # Convert list of dictionaries to DataFrame\n    pyradiomics_df_for_current_image = pd.DataFrame(pyradiomics_data_for_current_image)\n\n    # Save DataFrame to CSV file\n    pyradiomics_df_for_current_image.to_csv(f\"{features_folder}/pyradiomics_features_{fname}.csv\", index=False)\n\n    print(f\"All PyRadiomics features for image {fname} saved successfully.\")\n        # Directory path\n    directory = \"/kaggle/working/original_images\"\n    directory1 = \"/kaggle/working/masked_images\"\n\n    # Iterate over the files in the directory and delete each file\n    for filename in os.listdir(directory):\n        file_path = os.path.join(directory, filename)\n        try:\n            if os.path.isfile(file_path):\n                os.unlink(file_path)\n        except Exception as e:\n            print(f\"Failed to delete {file_path}: {e}\")\n            \n    for filename in os.listdir(directory1):\n        file_path = os.path.join(directory1, filename)\n        try:\n            if os.path.isfile(file_path):\n                os.unlink(file_path)\n        except Exception as e:\n            print(f\"Failed to delete {file_path}: {e}\")\n    \n# Delete original and masked images\nshutil.rmtree(original_images_folder)\nshutil.rmtree(masked_images_folder)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T21:39:38.914118Z","iopub.execute_input":"2024-02-15T21:39:38.914558Z","iopub.status.idle":"2024-02-15T21:43:43.897952Z","shell.execute_reply.started":"2024-02-15T21:39:38.914506Z","shell.execute_reply":"2024-02-15T21:43:43.896382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n\n# # Get the current working directory\n# working_dir = \"/kaggle/working\"\n\n# # List all files in the working directory\n# files = os.listdir(working_dir)\n\n# # Iterate over the files and delete CSV files\n# for file in files:\n#     if file.endswith(\".csv\"):\n#         file_path = os.path.join(working_dir, file)\n#         os.remove(file_path)\n#         print(f\"Deleted file: {file_path}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T19:11:15.58338Z","iopub.execute_input":"2024-02-15T19:11:15.583719Z","iopub.status.idle":"2024-02-15T19:11:15.589639Z","shell.execute_reply.started":"2024-02-15T19:11:15.583692Z","shell.execute_reply":"2024-02-15T19:11:15.588586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n\n# # Directory path\n# directory = \"/kaggle/working/features\"\n\n# # Iterate over the files in the directory and delete each file\n# for filename in os.listdir(directory):\n#     file_path = os.path.join(directory, filename)\n#     try:\n#         if os.path.isfile(file_path):\n#             os.unlink(file_path)\n#             print(f\"Deleted file: {file_path}\")\n#     except Exception as e:\n#         print(f\"Failed to delete {file_path}: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-02-15T21:39:14.849731Z","iopub.execute_input":"2024-02-15T21:39:14.850106Z","iopub.status.idle":"2024-02-15T21:39:14.857488Z","shell.execute_reply.started":"2024-02-15T21:39:14.850078Z","shell.execute_reply":"2024-02-15T21:39:14.856376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}