{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\n\ntrain_data = pd.read_csv(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")\n\ntrain_data = train_data[(train_data[\"class_id\"]==14) | (train_data[\"class_id\"]==8)]\n\ntrain_data = train_data[['image_id', 'class_name']]\n\ntrain_data['class_name'] = train_data['class_name'].replace({'No finding': 'Non-nodule', 'Nodule/Mass': 'Nodule'})\n\nnodule = train_data[train_data['class_name'] == 'Nodule'].head(500)\n\nnodule","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-16T14:04:58.252335Z","iopub.execute_input":"2023-10-16T14:04:58.25267Z","iopub.status.idle":"2023-10-16T14:04:58.905052Z","shell.execute_reply.started":"2023-10-16T14:04:58.252643Z","shell.execute_reply":"2023-10-16T14:04:58.90438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport pandas as pd\nfrom multiprocessing import Pool\n\ndef move_and_resize_image(input_dir, output_dir, image_name, new_name):\n    source_image_path = os.path.join(input_dir, image_name)\n    destination_image_path = os.path.join(output_dir, new_name)\n\n    # Open and resize the image using OpenCV\n    image = cv2.imread(source_image_path, 0)\n    image_resized = cv2.resize(image, (1024, 1024), interpolation=cv2.INTER_AREA)  # Resize to 1024x1024\n\n    # Save the resized image with the new name in the output folder\n    cv2.imwrite(destination_image_path, image_resized)\n\ndef process_images_in_parallel(input_dir, output_dir, data, nodule):\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Create a list of arguments for the move_and_resize_image function\n    image_args = []\n    for index, row in data.iterrows():\n        image_name = row['image_id'] + '.png'  # Adjust the column name accordingly\n        if nodule:\n            new_name = f\"n{index + 1:04}.png\"\n        else:\n            new_name = f\"c{index + 1:04}.png\"\n        image_args.append((input_dir, output_dir, image_name, new_name))\n    \n    # Create a pool of worker processes\n    with Pool(processes=4) as pool:  # Adjust the number of processes as needed\n        pool.starmap(move_and_resize_image, image_args)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:04:58.906267Z","iopub.execute_input":"2023-10-16T14:04:58.907Z","iopub.status.idle":"2023-10-16T14:04:59.146569Z","shell.execute_reply.started":"2023-10-16T14:04:58.906976Z","shell.execute_reply":"2023-10-16T14:04:59.145155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example usage for nodule images\ninput1 = \"/kaggle/input/vinbigdata-chest-xray-original-png/train\"\noutput1 = '/kaggle/working/Original/Nodules'\nprocess_images_in_parallel(input1, output1, nodule, True)\n\n# Example usage for non-nodule images\n# input2 = \"/kaggle/input/vinbigdata-chest-xray-original-png/train\"\n# output2 = '/kaggle/working/Original/Non-nodule'\n# process_images_in_parallel(input2, output2, non_nodule, False)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:04:59.147747Z","iopub.execute_input":"2023-10-16T14:04:59.148069Z","iopub.status.idle":"2023-10-16T14:05:28.412955Z","shell.execute_reply.started":"2023-10-16T14:04:59.148042Z","shell.execute_reply":"2023-10-16T14:05:28.411796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\n\ndef resize_jpegs_to_png(input_folder, output_folder):\n    # Ensure the output folder exists\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n\n    # List all JPEG files in the input folder\n    jpeg_files = [f for f in os.listdir(input_folder) if f.endswith(\".jpg\") or f.endswith(\".jpeg\")]\n\n    # Initialize a counter for naming the output PNG files\n    counter = 1\n\n    for jpeg_file in tqdm(jpeg_files):\n        # Read the JPEG image using OpenCV\n        image = cv2.imread(os.path.join(input_folder, jpeg_file))\n\n        # Resize the image to 1024x1024 pixels\n        image = cv2.resize(image, (1024, 1024))\n\n        # Create the output filename with the desired naming format\n        output_filename = os.path.join(output_folder, f\"c{str(counter).zfill(4)}.png\")\n\n        # Save the image as a PNG file\n        cv2.imwrite(output_filename, image)\n\n        # Increment the counter for the next image\n        counter += 1\n\n    print(\"Conversion completed.\")\n\n# Usage example:\ninput_folder = \"/kaggle/input/chest-xray-pneumonia/chest_xray/train/NORMAL/\"\noutput_folder = \"/kaggle/working/Original/Non-Nodules/\"\nresize_jpegs_to_png(input_folder, output_folder)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:05:28.415055Z","iopub.execute_input":"2023-10-16T14:05:28.415352Z","iopub.status.idle":"2023-10-16T14:07:29.431877Z","shell.execute_reply.started":"2023-10-16T14:05:28.415323Z","shell.execute_reply":"2023-10-16T14:07:29.430705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport SimpleITK as sitk\nimport matplotlib.pyplot as plt\nimport albumentations as A\nfrom albumentations.augmentations.transforms import CLAHE\nimport matplotlib.image as mpimg\nfrom skimage.io import imsave\nimport matplotlib.patches as patches\n\n\ndef load_image(image_path):\n    \"\"\"\n    Loads a PNG image from a given file path.\n\n    Args:\n        image_path (str): The file path of the image to load.\n\n    Returns:\n        numpy.ndarray: The loaded image as a NumPy array with pixel values in the [0, 255] range.\n    \"\"\"\n    # Load the image using Matplotlib\n    image = mpimg.imread(image_path)\n\n    # Normalize the array to the [0, 1] range\n    array_min = np.min(image)\n    array_max = np.max(image)\n    array = (image - array_min) / (array_max - array_min)\n\n    # Scale the array to the [0, 255] range\n    converted_image = (array * 255).astype(np.uint8)\n\n    return converted_image\n\n\ndef apply_clahe(image, clip_limit=8.0, tile_grid_size=(4, 4), p=1.0, alpha=0):\n    \"\"\"Applies CLAHE to a given image array using Albumentation library.\n\n    Args:\n        image (numpy.ndarray): The input image array.\n        clip_limit (float): Threshold for contrast limiting. Higher values result in more contrast.\n        tile_grid_size (tuple): Size of grid for histogram equalization. Higher values result in more smoothing.\n        alpha (int): Value added to the output image after CLAHE enhancement.\n\n    Returns:\n        numpy.ndarray: The image array with CLAHE applied.\n    \"\"\"\n        \n    clahe_transform = CLAHE(clip_limit=clip_limit, tile_grid_size=tile_grid_size, always_apply=True, p=p)\n    augmented = clahe_transform(image=image)\n    final_img = augmented['image'] + alpha\n    return final_img\n\n\ndef save_image(image_array, output_directory, filename):\n    \"\"\"Saves a given image array to disk as a PNG file.\n    \n    Args:\n        image_array (numpy.ndarray): The image array to save.\n        output_directory (str): The output directory to save the image file to.\n        filename (str): The name of the image file to save.\n    \"\"\"\n    # Replace the extension with '.png'\n    file_name, file_ext = os.path.splitext(filename)\n    new_filename = file_name + '.png'\n    save_path = os.path.join(output_directory, new_filename)\n    # Save the image using Matplotlib\n    plt.imsave(save_path, image_array, cmap=\"gray\")\n    \ndef gamma_correction(image, gamma=1.0):\n    \"\"\"Applies gamma correction to a given image using the specified gamma value.\n    \n    Args:\n        image (numpy.ndarray): The input image to be corrected.\n        gamma (float): The gamma value to use for correction. Default is 1.0 (no correction applied).\n    \n    Returns:\n        numpy.ndarray: The gamma-corrected image.\n    \"\"\"\n    # Ensure gamma is non-negative\n    if gamma < 0:\n        raise ValueError(\"Gamma value should be non-negative.\")\n\n    # Normalize the image to the [0, 1] range\n        \n    normalized_image = image.astype(np.float32) / 255.0\n\n    # Apply gamma correction\n    corrected_image = np.power(normalized_image, gamma)\n\n    # Scale the image to the [0, 255] range\n    corrected_image = (corrected_image * 255.0).clip(0, 255).astype(np.uint8)\n\n    return corrected_image\n\ndef tonemap_image(image):\n    \"\"\"Applies automatic gamma correction to a given image using OpenCV's cv2.createTonemap() function.\n    \n    Args:\n        image (numpy.ndarray): The input image to be corrected.\n    \n    Returns:\n        numpy.ndarray: The gamma-corrected image.\n    \"\"\"\n    # Convert the image to a floating-point format\n    image_float = image.astype(np.float32) / 255.0\n\n    # Create a tonemapping object and apply it to the image\n    tonemap = cv2.createTonemap()\n    tonemapped_image = tonemap.process(image_float)\n\n    # Scale the image to the [0, 255] range and convert back to uint8 format\n    tonemapped_image = (tonemapped_image * 255.0).astype(np.uint8)\n\n    return tonemapped_image\n\ndef equalize_histogram(image):\n    # Apply histogram equalization\n    equalized_image = cv2.equalizeHist(image)\n    \n    return equalized_image\n\n\ndef apply_bcet(image, alpha=0.5, beta=0.5):\n    \"\"\"\n    Applies Balance Contrast Enhancement Technique (BCET) to an input image.\n\n    Args:\n    - image: a numpy array representing the input image.\n    - alpha: the alpha parameter for BCET. Default value is 0.5.\n    - beta: the beta parameter for BCET. Default value is 0.5.\n\n    Returns:\n    - bcet_img: a numpy array representing the BCET-enhanced image.\n    \"\"\"\n\n    # Normalize the input image to [0, 1] range\n    max_val = np.max(image)\n    min_val = np.min(image)\n    image_norm = (image - min_val) / (max_val - min_val)\n\n    # Calculate the mean and standard deviation of the normalized image\n    image_mean = np.mean(image_norm)\n    image_std = np.std(image_norm)\n\n    # Calculate the midpoint and range for contrast stretching\n    midpoint = alpha * image_mean + (1 - alpha) * 0.5\n    range_val = beta * image_std + (1 - beta) * (max_val - min_val)\n\n    # Apply contrast stretching\n    bcet_img = np.clip((image_norm - midpoint + 0.5) / range_val, 0, 1)\n\n    # Rescale the image back to the original range\n    bcet_img = bcet_img * (max_val - min_val) + min_val\n\n    return bcet_img.astype(np.uint8)\n\ndef visualize(img):\n    plt.axis('off')\n    fig = plt.imshow(img, interpolation='nearest', cmap=\"gray\")\n    fig.axes.get_xaxis().set_visible(False)\n    fig.axes.get_yaxis().set_visible(False)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:07:29.433674Z","iopub.execute_input":"2023-10-16T14:07:29.434107Z","iopub.status.idle":"2023-10-16T14:07:31.952246Z","shell.execute_reply.started":"2023-10-16T14:07:29.434068Z","shell.execute_reply":"2023-10-16T14:07:31.951273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_loc = \"/kaggle/working/Original/Nodules/n6390.png\"\nimage = load_image(image_loc)\nvisualize(gamma_correction(image, 2))\nprint(image.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:08:01.217351Z","iopub.execute_input":"2023-10-16T14:08:01.217691Z","iopub.status.idle":"2023-10-16T14:08:01.436825Z","shell.execute_reply.started":"2023-10-16T14:08:01.217665Z","shell.execute_reply":"2023-10-16T14:08:01.436197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport multiprocessing\n\ndef process_image(filename, directory, clahe_directory, gamma_directory, inverted_directory, hist_equal_directory):\n    # Load the image\n    image_path = os.path.join(directory, filename)\n    image = cv2.imread(image_path, 0)\n\n    # Apply CLAHE to the array\n    clahe_img = apply_clahe(image, clip_limit=8.0, tile_grid_size=(4, 4), p=1.0)\n\n    # Save the CLAHE-enhanced image\n    save_image(clahe_img, clahe_directory, filename)\n\n    # Gamma Correction\n    gamma_corrected_img = gamma_correction(image, gamma=2)\n\n    # Save the gamma-corrected image\n    save_image(gamma_corrected_img, gamma_directory, filename)\n    \n    # Image Inversion\n    inverted_img = cv2.bitwise_not(image)\n    \n    # Save Inverted Image\n    save_image(inverted_img, inverted_directory, filename)\n    \n    # Histogram Equalization\n    hist_equal_image = cv2.equalizeHist(image)\n    \n    # Save the histogram equalized image\n    save_image(hist_equal_image, hist_equal_directory, filename)\n\nif __name__ == '__main__':\n    # Define the path to the directory containing your PNG images\n    directory = \"/kaggle/working/Original/Nodules\"\n\n    # Define the output directories for the augmented images\n    clahe_directory = \"/kaggle/working/CLAHE_images/Nodules\"\n    gamma_directory = \"/kaggle/working/Gamma_corrected_images/Nodules\"\n    inverted_directory = \"/kaggle/working/Inverted_images/Nodules\"\n    hist_equal_directory = \"/kaggle/working/Hist_equalized_images/Nodules\"\n\n    os.makedirs(clahe_directory, exist_ok=True)\n    os.makedirs(gamma_directory, exist_ok=True)\n    os.makedirs(inverted_directory, exist_ok=True)\n    os.makedirs(hist_equal_directory, exist_ok=True)\n\n    # List PNG files in the directory\n    png_files = [filename for filename in os.listdir(directory) if filename.endswith(\".png\")]\n\n    # Create a pool of worker processes\n    pool = multiprocessing.Pool()\n\n    # Process images in parallel\n    for filename in tqdm(png_files):\n        pool.apply_async(process_image, (filename, directory, clahe_directory, gamma_directory, inverted_directory, hist_equal_directory))\n\n    # Close the pool and wait for all processes to finish\n    pool.close()\n    pool.join()\n\n    print(\"Image processing completed.\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:07:32.493452Z","iopub.status.idle":"2023-10-16T14:07:32.493807Z","shell.execute_reply.started":"2023-10-16T14:07:32.493632Z","shell.execute_reply":"2023-10-16T14:07:32.493648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__':\n    # Define the path to the directory containing your PNG images\n    directory = \"/kaggle/working/Original/Non-Nodules\"\n\n    # Define the output directories for the augmented images\n    clahe_directory = \"/kaggle/working/CLAHE_images/Non-Nodules\"\n    gamma_directory = \"/kaggle/working/Gamma_corrected_images/Non-Nodules\"\n    inverted_directory = \"/kaggle/working/Inverted_images/Non-Nodules\"\n    hist_equal_directory = \"/kaggle/working/Hist_equalized_images/Non-Nodules\"\n\n    os.makedirs(clahe_directory, exist_ok=True)\n    os.makedirs(gamma_directory, exist_ok=True)\n    os.makedirs(inverted_directory, exist_ok=True)\n    os.makedirs(hist_equal_directory, exist_ok=True)\n\n    # List PNG files in the directory\n    png_files = [filename for filename in os.listdir(directory) if filename.endswith(\".png\")]\n\n    # Create a pool of worker processes\n    pool = multiprocessing.Pool()\n\n    # Process images in parallel\n    for filename in tqdm(png_files):\n        pool.apply_async(process_image, (filename, directory, clahe_directory, gamma_directory, inverted_directory, hist_equal_directory))\n\n    # Close the pool and wait for all processes to finish\n    pool.close()\n    pool.join()\n\n    print(\"Image processing completed.\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T14:07:32.495106Z","iopub.status.idle":"2023-10-16T14:07:32.495446Z","shell.execute_reply.started":"2023-10-16T14:07:32.495291Z","shell.execute_reply":"2023-10-16T14:07:32.495309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}