{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Load the train.csv file\ntrain_data = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\n\n# Display basic information about the dataset\nprint(train_data.info())\n\n# Display the first few rows of the dataset\nprint(train_data.head())\n\n# Summary statistics of numerical columns\nprint(train_data.describe())\n\n# Count of unique values in each column\nprint(train_data.nunique())\n\n# Check the column names in the DataFrame\nprint(train_data.columns)\n\n# Corrected code for exploring aortic_hu distribution\nplt.figure(figsize=(8, 6))\nsns.histplot(data=train_data, x='any_injury', bins=20)  # Replace with the actual column name\nplt.title('Distribution of Aortic Hounsfield Units')\nplt.xlabel('Aortic Hounsfield Units')\nplt.ylabel('Count')\nplt.show()\n\n\n# Distribution of target labels\nplt.figure(figsize=(8, 6))\nsns.countplot(data=train_data, x='any_injury')\nplt.title('Distribution of Any Injury')\nplt.xlabel('Any Injury')\nplt.ylabel('Count')\nplt.show()\n\n# Distribution of injuries by organ type\ninjury_columns = ['bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n                  'kidney_healthy', 'kidney_low', 'kidney_high',\n                  'liver_healthy', 'liver_low', 'liver_high',\n                  'spleen_healthy', 'spleen_low', 'spleen_high']\nplt.figure(figsize=(12, 18))\nfor column in injury_columns:\n    plt.subplot(5, 3, injury_columns.index(column) + 1)\n    sns.countplot(data=train_data, x=column)\n    plt.title(f'Distribution of {column}')\n    plt.xlabel(column)\n    plt.ylabel('Count')\nplt.tight_layout()\nplt.show()\n\n# Load the train_series_meta.csv file\nseries_meta_data = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_series_meta.csv')\n\n# Explore incomplete organ distribution\nplt.figure(figsize=(8, 6))\nsns.countplot(data=series_meta_data, x='incomplete_organ')  # Replace with the actual column name\nplt.title('Distribution of Incomplete Organ')\nplt.xlabel('Incomplete Organ')\nplt.ylabel('Count')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T15:08:36.134432Z","iopub.execute_input":"2023-09-01T15:08:36.134927Z","iopub.status.idle":"2023-09-01T15:08:38.972082Z","shell.execute_reply.started":"2023-09-01T15:08:36.134883Z","shell.execute_reply":"2023-09-01T15:08:38.970994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pydicom\nimport matplotlib.pyplot as plt\nimport cv2\nimport seaborn as sns\nimport tensorflow as tf\nimport os\n\ntrain = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/image_level_labels.csv')\nlabels = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\ntrain_meta = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_series_meta.csv')\n# Display basic statistics of labels\nlabels.describe()\ndef random_int(shape=[], minval=0, maxval=1):\n    return tf.random.uniform(shape=shape, minval=minval, maxval=maxval, dtype=tf.int32)\ndef standardize(img):\n    mean = np.mean(img)\n    std = np.std(img)\n    standardized_img = (img - mean) / std\n    return standardized_img\n\n# Adding img_path column to the train dataframe\ndef img_path(row):\n    return f\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/{row['patient_id']}/{row['series_id']}/{row['instance_number']}.dcm\"\n\ntrain['img_path'] = train.apply(img_path, axis=1)\n\n# Display an example DICOM instance\nex_inst = pydicom.dcmread(train.loc[0, 'img_path'])\nplt.imshow(ex_inst.pixel_array, cmap='gray')\n\n# Function to standardize pixel array\ndef standardize_pixel_array(dcm):\n    pixel_array = dcm.pixel_array\n    \n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        return new_array\n        \n    return pixel_array\n\n# Function to calculate crop coordinates\ndef crop_coords(img):\n    blur = cv2.GaussianBlur(img, (5, 5), 0)\n    _, mask = cv2.threshold(blur, 0, 255, cv2.THRESH_BINARY+cv2.THRESH_OTSU)\n    \n    cnts, _ = cv2.findContours(mask.astype(np.uint8), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    cnt = max(cnts, key=cv2.contourArea)\n    \n    x, y, w, h = cv2.boundingRect(cnt)\n    \n    return (x, y, w, h)\n\n# Function for truncation normalization\ndef truncation_normalization(img):\n    pmin = np.percentile(img[img != 0], 5)\n    pmax = np.percentile(img[img != 0], 99)\n    \n    truncated = np.clip(img, pmin, pmax)  \n    \n    normalized = (truncated - pmin) / (pmax - pmin)\n    normalized[img == 0] = 0\n    \n    return normalized\n\n# Function to clip and crop\ndef clipping(img):\n    pnc = 10\n    pmin = 0.0\n    \n    while pmin < 0.15:\n        pmin = np.percentile(img, pnc)\n        pnc += 1\n        \n    pmax = np.percentile(img, 85)\n            \n    img = np.clip(img, pmin, pmax)  \n    img = (img * 255).astype('uint8')\n    \n    return img\n\n\n# Function to preprocess an image\ndef preprocess(inst):\n    dcm = pydicom.dcmread(inst['img_path'])\n    \n    img = standardize_pixel_array(dcm)\n    img = truncation_normalization(img)\n    img = (img * 255).astype('uint8')\n    (x, y, w, h) = crop_coords(img)\n    img = img[y:y+h, x:x+w]\n    \n    return img\n\n# Display the preprocessed image and the original image\nplt.imshow(preprocess(train.loc[0 , :]), cmap='gray')\nplt.imshow(pydicom.dcmread(train.loc[0 , :]['img_path']).pixel_array, cmap='gray')\n\n# Function to compute cropped dimensions\ndef cropped_sizes(row):\n    dcm = pydicom.dcmread(row['img_path'])\n    \n    img = standardize_pixel_array(dcm)\n    img = truncation_normalization(img)\n    img = (img * 255).astype('uint8')\n    (x, y, w, h) = crop_coords(img)\n        \n    return (x, y, w, h)\n\ntrain['crop_raw_size'] = train.apply(cropped_sizes, axis=1)\n\n# Display a cropped example instance\ninst = train.loc[0, 'crop_raw_size']\nimg_cropped = ex_inst.pixel_array[inst[1]:inst[1]+inst[3], inst[0]:inst[0]+inst[2]]\nplt.imshow(img_cropped, cmap='gray')\n\n# Functions to get crop dimensions statistics\ndef get_w_data(row):\n    return row['crop_raw_size'][2]\n\ndef get_h_data(row):\n    return row['crop_raw_size'][3]\n\ndef get_ratio_data(row):\n    return row['crop_raw_size'][2] / row['crop_raw_size'][3]\n\n# Compute and add columns for crop dimensions and ratios\ntrain['crop_ratio'] = train.apply(get_ratio_data, axis=1)\ntrain['crop_width'] = train.apply(get_w_data, axis=1)\ntrain['crop_height'] = train.apply(get_h_data, axis=1)\n\n# Display statistics and histograms for crop dimensions\nprint(train['crop_ratio'].describe())\nsns.histplot(train['crop_ratio'])\n\nprint(train['crop_width'].describe())\nsns.histplot(train['crop_width'])\n\nprint(train['crop_height'].describe())\nsns.histplot(train['crop_height'])\n\n# Display an outlier image with the maximum cropped width\nmax_crop_width = 767.0\nimg_data = train[train['crop_width'] == max_crop_width].iloc[0, :]\nimg = preprocess(img_data)\nplt.imshow(img, cmap='gray')\n\n# Function for lossless clipping and cropping\ndef final_preprocess(inst, png=True):\n    dcm = pydicom.dcmread(inst['img_path'])\n    \n    img = standardize_pixel_array(dcm)\n    img = standardize(img)  # Replace with your standardize function\n    \n    loss_less_crop = np.copy(img)\n    \n    img = clipping(img)\n    \n    (x, y, w, h) = crop_coords(img)\n    img = loss_less_crop[y:y+h, x:x+w]\n    \n    img = img[..., tf.newaxis]\n    img = (img * 65535).astype('uint16')\n    if png:\n        png = tf.io.encode_png(img)\n        return png\n    else:\n        img = np.squeeze(img)\n        return img\n\n# Display randomly selected instances with their base and preprocessed versions\ndef display_random_inst(inst_num, axarr):\n    rows = []\n    \n    for i in range(inst_num):\n        ind = int(random_int(maxval=train.shape[0]-1))\n        inst = train.iloc[ind, :]\n        \n        rows.append(ind)\n        \n        axarr[i][1].imshow(tf.io.decode_png(final_preprocess(inst)), cmap='gray')\n        axarr[i][0].imshow(pydicom.dcmread(inst['img_path']).pixel_array, cmap='gray')\n        \n    return rows\n\ninst_num = 10  # Set the number of iterations\nf, axarr = plt.subplots(inst_num, 2)\nf.set_figheight(inst_num*8)\nf.set_figwidth(15)\nrows = display_random_inst(inst_num, axarr)\n\nfor ax, col in zip(axarr[0], ['BASE', 'PREPROCESSED']):\n    ax.set_title(col, color='red')\n\nfor ax, row in zip(axarr[:,0], rows):\n    ax.set_ylabel(row, rotation=0, size='large', color='red')\n\n# ... (Rest of the code remains the same as before)\n\n# Creating PNG images using the final_preprocess function\nfor i in os.listdir('/kaggle/working/'):\n    if i == '.virtual_documents':\n        continue\n    os.remove(i)\n\ndef save_pngs(row):\n    img = final_preprocess(row, png=False)\n    plt.imsave(\n        f\"/kaggle/working/{row['patient_id']}-{row['series_id']}-{row['instance_number']}.png\", \n        img, format='png', cmap='gray'\n    )\n\ntrain.apply(save_pngs, axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T15:08:38.974035Z","iopub.execute_input":"2023-09-01T15:08:38.974418Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}