{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **EDA Portion**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n# import seaborn as sns\n# sns.set_color_codes('dark')\n# sns.set_style('darkgrid')\n#!pip install pydicom\nimport pydicom\n\nimport os\nfrom pathlib import Path\nfrom sklearn.decomposition import PCA\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.cluster import KMeans\nfrom pandas.plotting import scatter_matrix\n\n#import gdcm\nfrom PIL import Image\n#import cv2\nfrom tqdm import tqdm # a progress bar for keeping track\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_text_attr=pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntrain_text_attr.head(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_text_attr.isna().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_imgs = len(train_text_attr)\ntotal_patients = len(train_text_attr[\"patient_id\"].unique())\nprint(f\"Total images: {total_imgs}\\n\")\nprint(f\"Total patients: {total_patients}\\n\")\n# Is the total images per side? or set? either way \n# there's more than double the amount of patients.\nmin_age = int(train_text_attr[\"age\"].min())\nmax_age = int(train_text_attr[\"age\"].max())\ngroupby_id = train.groupby(\"patient_id\")[\"cancer\"].max() # so as to not count the anyone more than once\ntotal_neg = (groupby_id == 0).sum()\ntotal_pos = (groupby_id == 1).sum()\n\nprint(f\"Youngest age: {min_age}\\n\")\nprint(f\"Oldest age: {max_age}\\n\")\nprint(f\"Total patients w/no cancer: {total_neg}  {str(round((total_neg / total_patients) * 100, 2))}% \\n\")\nprint(f\"Total patients w/cancer: {total_pos}        {str(round((total_pos / total_patients) * 100, 2))}% \\n\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lat_L = (train_text_attr[\"laterality\"]  == 'L').sum()\nlat_R = (train_text_attr[\"laterality\"] == 'R').sum()\nimplant_0 = (train_text_attr[\"implant\"] == 0).sum()\nimplant_1 = (train_text_attr[\"implant\"] == 1).sum()\ndiff_no = (train_text_attr[\"difficult_negative_case\"] == False).sum()\ndiff_yes = (train_text_attr[\"difficult_negative_case\"] == True).sum()\ncc = (train_text_attr[\"view\"] == \"CC\").sum()\nmlo = (train_text_attr[\"view\"] == \"MLO\").sum()\nml = (train_text_attr[\"view\"] == \"ML\").sum()\nlm = (train_text_attr[\"view\"] == \"LM\").sum()\nat = (train_text_attr[\"view\"] == \"AT\").sum()\nlmo= (train_text_attr[\"view\"] == \"LMO\").sum()\na = (train_text_attr[\"density\"] == \"A\").sum()\nb = (train_text_attr[\"density\"] == \"B\").sum()\nc = (train_text_attr[\"density\"] == \"C\").sum()\nd = (train_text_attr[\"density\"] == \"D\").sum()\nsite_1 = (train_text_attr[\"site_id\"] == 1).sum()\nsite_2 = (train_text_attr[\"site_id\"] == 2).sum()\n\nprint(f\"Total Left instances: {lat_L}\")\nprint(f\"Total Right instances: {lat_R}\\n\")\nprint(f\"No implant: {implant_0}\")\nprint(f\"Implant: {implant_1}\\n\")\nprint(f\"Difficult negative: {diff_yes}\")\nprint(f\"Not difficult: {diff_no}\\n\")\nprint(f\"Views\\nCC: {cc}\")\nprint(f\"MLO: {mlo}\")\nprint(f\"ML: {ml}\")\nprint(f\"LM: {lm}\")\nprint(f\"AT: {at}\")\nprint(f\"LMO: {lmo}/n\")\nprint(f\"Densities\\nA: {a}\")\nprint(f\"B: {b}\")\nprint(f\"C: {c}\")\nprint(f\"D: {d}\\n\")\nprint(f\"Site\\nA: {site_1}\")\nprint(f\"B: {site_2}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now to print an image then try to convert it - but using the dataset thats already converted","metadata":{}},{"cell_type":"code","source":"# importing pyplot and image from matplotlib\nimport matplotlib.pyplot as plt\nimport matplotlib.image as img\nimport os\n  \n    \nimg_dir = '/kaggle/input/rsna-breast-cancer-256-pngs/'\n\ntrain_imgs = [file for file in os.listdir(img_dir) if file.endswith('.png')]\n\nsample_num = 60\nimage_of_interest = train_imgs[sample_num]\nprint(\"Img name >>\",image_of_interest)#Prints the file name with the png extension\n#get the image\nimg = img.imread(img_dir+image_of_interest)\nprint('Image Shape >>',img.shape)\nplt.imshow(img)\n# show image\n\n\n    \n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Pairing the image with the other given attributes","metadata":{}},{"cell_type":"code","source":"print(train_text_attr.head(20))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below is a way to make our data into the form we want it","metadata":{}},{"cell_type":"code","source":"train_text_attr=pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\n\n\nprint(img) #prints the image in the form an array\n\n#A sample with conjoined features (ONLY features with no NaN's) from the CSV - laterality,view,biopsy,invasive, implant,machine_id, difficult_negative_case\nsample = (img, (train_text_attr.loc[0]['laterality'],train_text_attr.loc[0]['view'],train_text_attr.loc[0]['biopsy'],train_text_attr.loc[0]['invasive'],train_text_attr.loc[0]['implant'],train_text_attr.loc[0]['machine_id'],train_text_attr.loc[0]['difficult_negative_case'] ))\nsample_label = (str(train_text_attr.loc[0]['patient_id'])+\"_\"+str(train_text_attr.loc[0]['laterality']), train_text_attr.loc[0]['cancer'])\n#This is the final train sample attribute\nprint(sample[0])\nprint(sample_label)\n\n\n\n\n\n\n\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Filtration**","metadata":{}},{"cell_type":"markdown","source":"Currently - Xtrain = [image, image, image, ...]\n\nGoal - use other attributes as as a tuple  (imagedata, (laterality,view,biopsy,invasive, implant,machine_id, difficult_negative_case))\n","metadata":{}},{"cell_type":"markdown","source":"**Filter the Train Data here**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as img\nimport os\nimport pathlib\nimport pydicom\n\n\n\n#--------------------------------------------------------------------------------------------------\n\n#CSV Attributes\ntrain_text_attr=pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\n\n\n#Image data in 256x256 format\nimg_dir = '/kaggle/input/rsna-breast-cancer-256-pngs/'\ntrain_imgs = [file for file in os.listdir(img_dir) if file.endswith('.png')]\n\n# img_dir_test = '/kaggle/input/rsna-breast-cancer-detection/test_images/10008/'\n# test_imgs_unfltrd = [file for file in os.listdir(img_dir_test) if file.endswith('.dcm')]\n'''\nPrint the Images below for testing\n'''\n# im = img.imread(img_dir+train_imgs[0])\n# print('Image Shape >>',im.shape)\n# plt.imshow(im)\n\n\nx_train = []\ny_train = []\nX_test = []\n\n\n#get the training data\nfor i in range(0,int(len(train_imgs)*.001)):\n    if (i%50)==0:\n        print(\"Processed \", i, \"images\")\n    train_img = img.imread(img_dir+train_imgs[i])\n#     X_train.append((train_img, (train_text_attr.loc[i]['laterality'],train_text_attr.loc[i]['view'],train_text_attr.loc[i]['biopsy'],train_text_attr.loc[i]['invasive'],train_text_attr.loc[i]['implant'],train_text_attr.loc[i]['machine_id'],train_text_attr.loc[i]['difficult_negative_case'], train_text_attr.loc[i]['cancer'])))\n    \n    #before adding the image to the train - make every value an int of the img - the img must be a list of ints\n    train_img = list(map(lambda x: list(map(float, x)), train_img))\n    x_train.append(train_img)\n    y_train.append(int(train_text_attr.loc[i]['cancer']))\n    \n#get the test data\nfor i in range(0,int(len(train_imgs)*.001)):\n    if (i%50)==0:\n        print(\"Processed \", i, \"images\")\n\nprint(len(test_imgs_unfltrd))\ntest_img = pydicom.dcmread(img_dir_test+test_imgs_unfltrd[0])\n# Convert the pixel array to a numpy array\ntest_img = test_img.pixel_array\n\n# Rescale the pixel values to [0, 1]\ntest_img = test_img / np.max(test_img)\n\n# Convert the numpy array to a 2D float list\ntest_img = test_img.tolist()\n\nprint(test_img)\n  \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Test Filtration Here**","metadata":{}},{"cell_type":"code","source":"\n# Import libraries\nimport os\nimport pydicom\nfrom PIL import Image\nfrom pathlib import Path\nimport numpy as np\nimport multiprocessing as mp\nimport tensorflow_io as tfio\n\n\n\nimg_dir_test = '/kaggle/input/rsna-breast-cancer-detection/test_images/'\ntest_imgs_unfltrd = [file for file in os.listdir(img_dir_test) if file.endswith('.dcm')]\n# Define a directory that contains dicom files\n# dicom_dir = 'path/to/dicom/files'\n\n\nX_test = []\n\n\n# RESIZE_TO = (256, 256)\n# !rm -rf test_images_processed_{RESIZE_TO[0]}\n# !mkdir test_images_processed_{RESIZE_TO[0]}\n\n# https://www.kaggle.com/code/tanlikesmath/brain-tumor-radiogenomic-classification-eda/notebook\ndef dicom_file_to_ary(path):\n    dicom = pydicom.read_file(path)\n#     dicom = tfio.read_file(path)\n    data = dicom.pixel_array\n#     pixel_data = tfio.image.decode_dicom_image(image_bytes,dtype=tf.uint16)\n    print(type(data))\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndirectories = list(Path(img_dir_test).iterdir())\nprint(directories)\n\ndef process_directory(directory_path):\n    parent_directory = str(directory_path).split('/')[-1]\n    !mkdir -p train_images_processed_{RESIZE_TO[0]}/{parent_directory}\n    for image_path in directory_path.iterdir():\n        processed_ary = dicom_file_to_ary(image_path)\n        im = Image.fromarray(processed_ary).resize(RESIZE_TO)\n        im.save(f'test_images_processed_{RESIZE_TO[0]}/{parent_directory}/{image_path.stem}.png')\n        \n\nwith mp.Pool(64) as p:\n    p.map(process_directory, directories)\n\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nFOR DEBUGGING BY TESTING SHAPES, VALUES, TYPES, ETC\n'''\n\n# print(type(int(train_text_attr.loc[i]['cancer'])))\n#print(type(x_train[0][0][0]))#type of structure of x_train\n# print(len(x_train))#num samples\n# print(type(x_train[0]))\n# print(x_train[0])#sample shape\n#plt.imshow(x_train[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model** ","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers\nfrom keras import backend as K\n\n# Define the pF1 metric as a custom loss function - NEEDS to be used in the loss compile method!\ndef pF1(y_true, y_pred):\n    tp = K.sum(K.cast(y_true*y_pred, 'float32'), axis=0)\n    tn = K.sum(K.cast((1-y_true)*(1-y_pred), 'float32'), axis=0)\n    fp = K.sum(K.cast((1-y_true)*y_pred, 'float32'), axis=0)\n    fn = K.sum(K.cast(y_true*(1-y_pred), 'float32'), axis=0)\n    precision = tp / (tp + fp + K.epsilon())\n    recall = tp / (tp + fn + K.epsilon())\n    f1 = 2 * precision * recall / (precision + recall + K.epsilon())\n    return 1 - K.mean(f1)\n\n# Define the model\nmodel = tf.keras.Sequential([\n    layers.Conv2D(32, (3, 3), activation='relu', input_shape=(256, 256, 1), dtype=tf.float32),\n    layers.MaxPooling2D((2, 2)),\n    layers.Conv2D(64, (3, 3), activation='relu'),\n    layers.MaxPooling2D((2, 2)),\n    layers.Flatten(),\n    layers.Dense(64, activation='relu'),\n    layers.Dense(1, activation='linear')\n])\n\n# # Compile the model\nmodel.compile(loss=\"mse\", optimizer='adam')\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T08:28:04.694058Z","iopub.execute_input":"2023-02-27T08:28:04.69476Z","iopub.status.idle":"2023-02-27T08:28:04.784936Z","shell.execute_reply.started":"2023-02-27T08:28:04.694722Z","shell.execute_reply":"2023-02-27T08:28:04.78419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model and show metrics\nhistory = model.fit(x_train, y_train, epochs=10, batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T08:31:22.290384Z","iopub.execute_input":"2023-02-27T08:31:22.290766Z","iopub.status.idle":"2023-02-27T08:31:41.592084Z","shell.execute_reply.started":"2023-02-27T08:31:22.290731Z","shell.execute_reply":"2023-02-27T08:31:41.590892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting the data","metadata":{}},{"cell_type":"code","source":"test_predictions = model.predict(test_data)","metadata":{},"execution_count":null,"outputs":[]}]}