{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import pydicom\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \n\nfrom pathlib import Path\nimport glob\nimport pandas as pd\n# import pylibjpeg\n\nimport numpy as np\nfrom tensorflow import keras\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nimport random\nfrom scipy.ndimage import gaussian_filter\nfrom scipy import ndimage\nimport cv2\nimport matplotlib.pyplot as plt\nfrom skimage.transform import rescale, resize, downscale_local_mean\nfrom tqdm import tqdm\n# import wandb\n\n\n# from kaggle_secrets import UserSecretsClient\n# from wandb.keras import WandbCallback\n\n# user_secrets = UserSecretsClient()\n\n# # I have saved my API token with \"wandb_api\" as Label. \n# # If you use some other Label make sure to change the same below. \n# wandb_api = user_secrets.get_secret(\"wandb_api\") \n\n# wandb.login(key=wandb_api)\n\n# wandb.init(project=\"kaggle-rsna-breast-cancer-detection\")\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-05T15:55:18.210737Z","iopub.execute_input":"2023-01-05T15:55:18.211416Z","iopub.status.idle":"2023-01-05T15:55:18.22105Z","shell.execute_reply.started":"2023-01-05T15:55:18.21138Z","shell.execute_reply":"2023-01-05T15:55:18.219554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# source: https://www.kaggle.com/code/allunia/rsna-csf-cervical-spine-fracture-eda/notebook\ndef rescale_img_to_hu(dcm_ds):\n    \"\"\"Rescales the image to Hounsfield unit.\"\"\"\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept\n\ndef show_images_for_patient(patient_id):\n    patient_dir = os.path.join('../input/rsna-breast-cancer-detection/train_images', str(patient_id))\n    num_images = len(glob.glob(f\"{patient_dir}/*\"))\n    print(f\"Number of images for patient: {num_images}\")\n    fig, axs = plt.subplots(5,3, figsize=(24,15))\n    axs = axs.flatten()\n    for i, img_path in enumerate(list(Path(patient_dir).iterdir())):\n        ds = pydicom.dcmread(img_path)\n        axs[i].imshow(rescale_img_to_hu(ds), cmap=\"bone\")\n        \ndef multi_class_labels(data, labels=[1]):\n    if data==1:\n        return [0,1]\n    return [1,0]\n\ndef image_resize(image, width = None, height = None, inter = cv2.INTER_LINEAR):\n\n    dim = None\n    (h, w) = image.shape[:2]\n\n    if width is None and height is None:\n        return image\n\n    if width is None:\n        r = height / float(h)\n        dim = (int(w * r), height)\n    else:\n        r = width / float(w)\n        dim = (width, int(h * r))\n    resized = cv2.resize(image, dim, interpolation = inter)\n\n    return resized\n\n\n        \n# show_images_for_patient(55706)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:27:28.969934Z","iopub.execute_input":"2023-01-05T15:27:28.970413Z","iopub.status.idle":"2023-01-05T15:27:28.992464Z","shell.execute_reply.started":"2023-01-05T15:27:28.970371Z","shell.execute_reply":"2023-01-05T15:27:28.991255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def identity_block(x, filter):\n    # copy tensor to variable called x_skip\n    x_skip = x\n    # Layer 1\n    x = tf.keras.layers.Conv2D(filter, (3,3), padding = 'same')(x)\n    x = tf.keras.layers.BatchNormalization(axis=3)(x)\n    x = tf.keras.layers.Activation('relu')(x)\n    # Layer 2\n    x = tf.keras.layers.Conv2D(filter, (3,3), padding = 'same')(x)\n    x = tf.keras.layers.BatchNormalization(axis=3)(x)\n    # Add Residue\n    x = tf.keras.layers.Add()([x, x_skip])     \n    x = tf.keras.layers.Activation('relu')(x)\n    return x\n\ndef convolutional_block(x, filter):\n    # copy tensor to variable called x_skip\n    x_skip = x\n    # Layer 1\n    x = tf.keras.layers.Conv2D(filter, (3,3), padding = 'same', strides = (2,2))(x)\n    x = tf.keras.layers.BatchNormalization(axis=3)(x)\n    x = tf.keras.layers.Activation('relu')(x)\n    # Layer 2\n    x = tf.keras.layers.Conv2D(filter, (3,3), padding = 'same')(x)\n    x = tf.keras.layers.BatchNormalization(axis=3)(x)\n    # Processing Residue with conv(1,1)\n    x_skip = tf.keras.layers.Conv2D(filter, (1,1), strides = (2,2))(x_skip)\n    # Add Residue\n    x = tf.keras.layers.Add()([x, x_skip])     \n    x = tf.keras.layers.Activation('relu')(x)\n    return x\n\ndef make_model(INPUT_SHAPE, classes, output_bias=None):\n    # Step 1 (Setup Input Layer)\n    x_input = tf.keras.layers.Input(INPUT_SHAPE)\n    x = tf.keras.layers.ZeroPadding2D((3, 3))(x_input)\n    # Step 2 (Initial Conv layer along with maxPool)\n    x = tf.keras.layers.Conv2D(64, kernel_size=7, strides=2, padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Activation('relu')(x)\n    x = tf.keras.layers.MaxPool2D(pool_size=3, strides=2, padding='same')(x)\n    # Define size of sub-blocks and initial filter size\n    block_layers = [3, 4, 6, 3]\n    filter_size = 64\n    # Step 3 Add the Resnet Blocks\n    for i in range(4):\n        if i == 0:\n            # For sub-block 1 Residual/Convolutional block not needed\n            for j in range(block_layers[i]):\n                x = identity_block(x, filter_size)\n        else:\n            # One Residual/Convolutional Block followed by Identity blocks\n            # The filter size will go on increasing by a factor of 2\n            filter_size = filter_size*2\n            x = convolutional_block(x, filter_size)\n            for j in range(block_layers[i] - 1):\n                x = identity_block(x, filter_size)\n    # Step 4 End Dense Network\n    x = tf.keras.layers.AveragePooling2D((2,2), padding = 'same')(x)\n    x = tf.keras.layers.Flatten()(x)\n    x = tf.keras.layers.Dense(512, activation = 'relu')(x)\n    x = tf.keras.layers.Dense(classes, activation = 'softmax')(x)\n    model = tf.keras.models.Model(inputs = x_input, outputs = x, name = \"ResNet34\")\n    return model\n\nmodel=make_model((256,256,3),2)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:27:29.456397Z","iopub.execute_input":"2023-01-05T15:27:29.456872Z","iopub.status.idle":"2023-01-05T15:27:30.356092Z","shell.execute_reply.started":"2023-01-05T15:27:29.456793Z","shell.execute_reply":"2023-01-05T15:27:30.355078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Generator:\n    def __init__(self, image_list):\n        np.random.seed(10)\n        train_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n#         train_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n        base_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/'\n        # saving image path into train dataframe\n        train_csv['img_path']= f'{base_path}/train_images_processed_cv2_256'\\\n                            + '/' + train_csv.patient_id.astype(str)\\\n                            + '/' + train_csv.image_id.astype(str)\\\n                            + '.png'\n        train_csv = train_csv.sample(frac=1).reset_index(drop=True)\n        case = train_csv[train_csv['cancer']==1]\n        control = train_csv[train_csv['cancer']==0]\n\n        frames = [case, control.loc[:len(case)]]\n        train_csv = pd.concat(frames)\n        \n        train_csv = train_csv.sample(frac=1).reset_index(drop=True)\n        _image=[]\n        _label=[]\n        for encounter in tqdm(train_csv.iterrows()):\n            img_path = os.path.join(f'{base_path}/train_images_processed_cv2_256', str(encounter[1]['patient_id']), str(encounter[1]['image_id'])+ '.png')\n            image=tf.io.read_file(img_path)\n            image = tf.io.decode_png(image)\n\n            _image.append(resize(image[:,:,0], (256,256)))\n            _label.append(encounter[1]['cancer'])\n        \n        self.n_image=len(_image)\n        self.image=_image\n        self.label=_label\n    def get_item(self):\n        while True:\n            randomindex = random.randint(0, self.n_image-1)\n            X_placeholder = np.zeros((256, 256, 3, 1), dtype=np.float32)\n            y_placeholder = np.zeros((2), dtype=np.int16)\n            image = self.image[randomindex]\n            label = self.label[randomindex]\n                    \n            image=(image-np.min(image))/(np.max(image)-np.min(image))\n            X = np.array(np.stack((np.array(image),)*3, axis=2))\n            y = multi_class_labels(label, np.array([0,1]))\n            X_placeholder[:, :, :, 0] = X\n            y_placeholder = y\n            yield X_placeholder, y_placeholder\n            \n\ntraining_generator=Generator('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_tf_gen = tf.data.Dataset.from_generator(training_generator.get_item, (tf.float32, tf.int16), (tf.TensorShape([256,256,3,1]), tf.TensorShape([2])))\n# train_tf_gen = train_tf_gen.cache()\ntrain_batches = train_tf_gen.batch(32)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-01-05T15:27:32.56439Z","iopub.execute_input":"2023-01-05T15:27:32.564836Z","iopub.status.idle":"2023-01-05T15:27:50.52272Z","shell.execute_reply.started":"2023-01-05T15:27:32.564779Z","shell.execute_reply":"2023-01-05T15:27:50.521842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nBATCH_SIZE=300\n\nMETRICS = [\n    keras.metrics.TruePositives(name='tp'),\n    keras.metrics.FalsePositives(name='fp'),\n    keras.metrics.TrueNegatives(name='tn'),\n    keras.metrics.FalseNegatives(name='fn'), \n    keras.metrics.BinaryAccuracy(name='accuracy'),\n    keras.metrics.Precision(name='precision'),\n    keras.metrics.Recall(name='recall'),\n    keras.metrics.AUC(name='auc'),\n    keras.metrics.AUC(name='prc', curve='PR'), # precision-recall curve\n]\n\nmodel=make_model((256,256,3), 2)\n\nmodel.compile(\n    optimizer=keras.optimizers.Adam(learning_rate=1e-3),\n    loss=keras.losses.BinaryCrossentropy(),\n    metrics=METRICS)\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_prc', \n    verbose=1,\n    patience=20,\n    mode='max',\n    restore_best_weights=True)\n\n!mkdir weights\n\ncheck_pointer = tf.keras.callbacks.ModelCheckpoint(filepath=\"./weights/iqa_seg.h5\", verbose=1, save_best_only=True, save_weights_only=True)\n\nhistory = model.fit(\n    train_batches,\n    steps_per_epoch=25,\n    epochs=100,\n    callbacks=[check_pointer],\n    validation_data=train_batches,\n    validation_steps=10,\n    )\n\n\n\nmodel.load_weights('/kaggle/working/weights/iqa_seg.h5')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:27:50.525491Z","iopub.execute_input":"2023-01-05T15:27:50.525778Z","iopub.status.idle":"2023-01-05T15:38:14.28813Z","shell.execute_reply.started":"2023-01-05T15:27:50.525751Z","shell.execute_reply":"2023-01-05T15:38:14.287094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Images\n\nDATASET_PATH='/kaggle/input/rsna-breast-cancer-detection/'\ntest_df = pd.read_csv(os.path.join(DATASET_PATH, \"test.csv\"))\ndisplay(test_df.head())\nprint(f'cases: {len(test_df)}')\n\n# Show sample submission example\n\ndf_sub = pd.read_csv(os.path.join(DATASET_PATH, \"sample_submission.csv\"))\ndisplay(df_sub.head())\nprint(f'cases: {len(df_sub)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:45:18.783743Z","iopub.execute_input":"2023-01-05T15:45:18.784147Z","iopub.status.idle":"2023-01-05T15:45:18.817141Z","shell.execute_reply.started":"2023-01-05T15:45:18.784113Z","shell.execute_reply":"2023-01-05T15:45:18.81625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path):\n    print(image_path)\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels = 3)\n    img = tf.image.resize(img, [256, 256])\n    img = tf.cast(img, dtype = tf.float32)\n    img = img/255.0\n    return img\n\n# test_paths=[]\n# test_dir='/kaggle/input/rsnatest/test_images_256/10008/'\n# img_path= os.listdir (test_dir)\n\n\n\nDF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\ndf = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\n\ntest_path='/kaggle/input/rsnatest/test_images_256/'\n\ntest_dir = f'{test_path}'\n\n\n\n\ntest_df['img_path']= f'{test_dir}/'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.png'\n\n\ntest_df['img_path'][1]\nimage = load_image(test_df['img_path'][1])\n\nprint(image.shape)\nplt.imshow(image)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:45:22.846251Z","iopub.execute_input":"2023-01-05T15:45:22.846613Z","iopub.status.idle":"2023-01-05T15:45:23.057225Z","shell.execute_reply.started":"2023-01-05T15:45:22.846581Z","shell.execute_reply":"2023-01-05T15:45:23.056319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=[]\nfor i in range (len(test_df['img_path'])):\n    image = tf.keras.preprocessing.image.load_img(test_df['img_path'][i])\n    image=tf.expand_dims(np.array(image), 0)\n    pred=np.argmax(model.predict(np.asarray(image)))\n    preds.append(pred)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:45:24.550694Z","iopub.execute_input":"2023-01-05T15:45:24.551077Z","iopub.status.idle":"2023-01-05T15:45:25.212734Z","shell.execute_reply.started":"2023-01-05T15:45:24.551045Z","shell.execute_reply":"2023-01-05T15:45:25.211843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame({'prediction_id':test_df.prediction_id,\n                        'cancer':preds})\n\npred_df['cancer']=(np.argmax(pred_df.cancer)).astype(int)\n\n\npred_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:45:28.553781Z","iopub.execute_input":"2023-01-05T15:45:28.554503Z","iopub.status.idle":"2023-01-05T15:45:28.567366Z","shell.execute_reply.started":"2023-01-05T15:45:28.554467Z","shell.execute_reply":"2023-01-05T15:45:28.566082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-01-05T15:51:32.431309Z","iopub.execute_input":"2023-01-05T15:51:32.43167Z","iopub.status.idle":"2023-01-05T15:51:32.497995Z","shell.execute_reply.started":"2023-01-05T15:51:32.431638Z","shell.execute_reply":"2023-01-05T15:51:32.497094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}