{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Dataset converted to PNGs can be found at https://www.kaggle.com/datasets/radek1/rsna-mammography-images-as-pngs","metadata":{}},{"cell_type":"code","source":"%%capture\n\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n    !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport dicomsdl\nimport numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom PIL import Image\nimport scipy.ndimage\n\nfrom os import listdir, mkdir\nfrom joblib import Parallel, delayed\nimport cv2\nimport glob\nimport sys\nimport time\nimport pydicom\n\nimport tensorflow as tf\nimport tensorflow as tf\nfrom tensorflow import keras\nimport tensorflow_io as tfio\nfrom skimage.transform import resize \nfrom kaggle_datasets import KaggleDatasets\nfrom kaggle_secrets import UserSecretsClient\nfrom sklearn.model_selection import train_test_split\n# Import Keras Libraries\n\nimport logging\nlogging.getLogger('tensorflow').disabled = True\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = '3'\nprint(f'Tensorflow Version: {tf.__version__}')\nprint(f'Python Version: {sys.version}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# float32 or mixed_float16 (mixed precision: compute float16, variable float32)\n# TPU is fast enough and has enough memory to use float32\npolicy = tf.keras.mixed_precision.Policy('float32')\ntf.keras.mixed_precision.set_global_policy(policy)\n\nprint(f'Compute dtype: {tf.keras.mixed_precision.global_policy().compute_dtype}')\nprint(f'Variable dtype: {tf.keras.mixed_precision.global_policy().variable_dtype}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() # TPU detection\nexcept ValueError:\n    tpu = None\n    gpus = tf.config.experimental.list_logical_devices(\"GPU\")\n    \nif tpu:\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    STRATEGY = tf.distribute.experimental.TPUStrategy(tpu,) \n    print('Running on TPU ', tpu.cluster_spec().as_dict()['worker'])\nelif len(gpus) > 1:\n    STRATEGY = tf.distribute.MirroredStrategy([gpu.name for gpu in gpus])\n    print('Running on multiple GPUs ', [gpu.name for gpu in gpus])\nelif len(gpus) == 1:\n    STRATEGY = tf.distribute.get_strategy() \n    print('Running on single GPU ', gpus[0].name)\nelse:\n    STRATEGY = tf.distribute.get_strategy() \n    print('Running on CPU')\nprint(\"Number of accelerators: \", STRATEGY.num_replicas_in_sync)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\nIMG_PATH = '/kaggle/input/rsna-breast-cancer-512-pngs'\n    \n    \ndf = pd.read_csv(f\"{DF_PATH}/train.csv\")\ndf['img_path'] = df.apply(\n    lambda i: os.path.join(\n        f\"{IMG_PATH}\", str(i['patient_id']) + \"_\" + str(i['image_id']) + '.png'\n    ), axis=1\n)\n\ndisplay(df.head())\nprint(df.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create train-validation split\ntrain_df, valid_df = train_test_split(df[['img_path','cancer']], test_size=0.2, random_state=42)\n\n\nprint(train_df.cancer.value_counts(normalize=True))\nprint(valid_df.cancer.value_counts(normalize=True))\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LR_MAX = 1e-3\nN_REPLICAS = STRATEGY.num_replicas_in_sync\nBATCH_SIZE = 32\nINP_SIZE = (512, 512)\n\nAUTOTUNE = tf.data.AUTOTUNE\nEPOCHS = 15\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To decode datatype\ndef preprocess_image(image,labels):\n    image = tf.image.decode_png(image, channels=3)\n\n    image = tf.cast(image, tf.float32)\n    image = tf.reshape(image, [*[512]*2, 3])\n    print(image.shape)\n    image /= 255.0  # normalize to [0,1] range         # IMAGE NORMALIZE\n\n    return image,labels\n\n# Read image, and process\ndef load_and_preprocess_image(path,labels):\n    image = tf.io.read_file(path)\n    return preprocess_image(image,labels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_dataset(df,batch_size,with_labels = False, shuffle= False):\n    \n    \n    # Create Dataset\n    if with_labels:\n        dataset = tf.data.Dataset.from_tensor_slices((df['img_path'].values, df['cancer'].values))\n    else:\n        dataset = tf.data.Dataset.from_tensor_slices(\n            (df['img_path'].values)\n        )\n\n    # Image preprocessing\n    dataset = dataset.map(load_and_preprocess_image) \n    \n\n    # Get one batch\n    dataset = dataset.shuffle(1024, reshuffle_each_iteration = True) if shuffle else dataset\n    dataset = dataset.batch(batch_size,drop_remainder=False)\n    dataset = dataset.prefetch(AUTOTUNE)\n    #dataset = dataset.shape\n\n    return dataset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_dataset = create_dataset(\n    train_df,\n    batch_size  = BATCH_SIZE, \n    with_labels = True, \n    shuffle = True\n)\n\nvalid_dataset = create_dataset(\n    valid_df,\n    batch_size  = BATCH_SIZE, \n    with_labels = True, \n    shuffle = False\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SEED = 43\n# DEBUG = False\n\n# # Image dimensions\n# IMG_HEIGHT = 1344\n# IMG_WIDTH = 768\n# N_CHANNELS = 1\n# INPUT_SHAPE = (IMG_HEIGHT, IMG_WIDTH, 1)\n# N_SAMPLES_TFRECORDS = 548\n\n# # Peak Learning Rate\n# LR_MAX = 1e-3\n\n# N_WARMUP_EPOCHS = 2\n# N_EPOCHS = 1\n\n# N_REPLICAS = STRATEGY.num_replicas_in_sync\n# #print(f'N_REPLICAS: {N_REPLICAS}, IS_TPU: {IS_TPU}')\n\n\n# # Batch size\n# BATCH_SIZE = 8*N_REPLICAS\n\n# # Is Interactive Flag and COrresponding Verbosity Method\n# IS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n# VERBOSE = 1 if IS_INTERACTIVE else 2\n\n# # Tensorflow AUTO flag\n# AUTO = tf.data.experimental.AUTOTUNE\n\n# print(f'BATCH_SIZE: {BATCH_SIZE}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seed all random number generators\nSEED = 43\ndef seed_everything(seed=SEED):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_everything()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## [Competition Metric](https://www.kaggle.com/code/sohier/probabilistic-f-score)","metadata":{}},{"cell_type":"code","source":"def tf_pfbeta(from_logits=True, beta=1.0, epsilon=1e-07):\n    \n    def pfbeta(y_true, y_pred):\n        y_pred = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(y_pred),\n            lambda: y_pred,\n        )\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n\n        ctp = tf.reduce_sum(y_true * y_pred, axis=-1)\n        cfp = tf.reduce_sum(y_pred, axis=-1) - ctp\n\n        c_precision = ctp / (ctp + cfp)\n        c_recall = ctp / tf.reduce_sum(y_true)\n        \n        def compute_fractions():\n            numerator = c_precision * c_recall\n            denominator = beta**2 * c_precision + c_recall\n            return (1 + beta**2) * tf.math.divide_no_nan(numerator, denominator)\n        \n        return tf.cond(\n            tf.logical_and(\n                tf.greater(c_precision, 0.), tf.greater(c_recall, 0.)\n            ),\n            compute_fractions,\n            lambda: tf.constant(0, dtype=tf.float32)\n        )\n    \n    return pfbeta","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Decodes the TFRecords\n# def decode_image(record_bytes):\n#     features = tf.io.parse_single_example(record_bytes, {\n#         'image': tf.io.FixedLenFeature([], tf.string),\n#         'target': tf.io.FixedLenFeature([], tf.int64),\n#         'patient_id': tf.io.FixedLenFeature([], tf.int64),\n#     })\n    \n#     # Decode PNG Image\n#     image = tf.io.decode_png(features['image'], channels=N_CHANNELS)\n#     # Explicit reshape needed for TPU\n#     image = tf.reshape(image, [IMG_HEIGHT, IMG_WIDTH, N_CHANNELS])\n\n#     target = features['target']\n    \n#     return { 'image': image }, target","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # TFRecord file paths\n# TFRECORDS_FILE_PATHS = sorted(tf.io.gfile.glob(f'{GCS_DS_PATH}/*.tfrecords'))\n# print(f'Found {len(TFRECORDS_FILE_PATHS)} TFRecords')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Train Test Split\n# TFRECORDS_TRAIN, TFRECORDS_VAL = train_test_split(TFRECORDS_FILE_PATHS, train_size=0.80, random_state=SEED, shuffle=True)\n# print(f'# TFRECORDS_TRAIN: {len(TFRECORDS_TRAIN)}, # TFRECORDS_VAL: {len(TFRECORDS_VAL)}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_dataset(tfrecords, bs=BATCH_SIZE, val=False, debug=True):\n#     ignore_order = tf.data.Options()\n#     ignore_order.experimental_deterministic = False\n    \n#     # Initialize dataset with TFRecords\n#     dataset = tf.data.TFRecordDataset(tfrecords, num_parallel_reads=AUTO, compression_type='GZIP')\n    \n#     # Decode mapping\n#     dataset = dataset.map(decode_image, num_parallel_calls=AUTO)\n\n#     if not val:\n#         dataset = dataset.filter(undersample_majority)\n#         #dataset = dataset.map(augment_image, num_parallel_calls=AUTO)\n#         dataset = dataset.with_options(ignore_order)\n#         if not debug:\n#             dataset = dataset.shuffle(1024)\n#         dataset = dataset.repeat()        \n\n#     dataset = dataset.batch(bs, drop_remainder=not val)\n#     dataset = dataset.prefetch(AUTO)\n    \n#     return dataset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Undersample majority class (0/negative) by randomly dropping them\n# def undersample_majority(X, y):\n#     # Filter 2/3 of negative samples to upsample positive samples by a factor 3\n#     return y == 1 or tf.random.uniform([]) > 0.66","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Get Train/Validation datasets\n# train_dataset = get_dataset(TFRECORDS_TRAIN, val=False, debug=False)\n# val_dataset = get_dataset(TFRECORDS_VAL, val=True, debug=False)\n\n# TRAIN_STEPS_PER_EPOCH = len(TFRECORDS_TRAIN) * N_SAMPLES_TFRECORDS // BATCH_SIZE\n# VAL_STEPS_PER_EPOCH = len(TFRECORDS_VAL) * N_SAMPLES_TFRECORDS // BATCH_SIZE\n# print(f'TRAIN_STEPS_PER_EPOCH: {TRAIN_STEPS_PER_EPOCH}, VAL_STEPS_PER_EPOCH: {VAL_STEPS_PER_EPOCH}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def binary_focal_loss(gamma=2., alpha=.25, class_weights=None):\n    \"\"\"\n    Binary form of focal loss.\n      FL(p_t) = -alpha * (1 - p_t)**gamma * log(p_t)\n      where p = sigmoid(x), p_t = p or 1 - p depending on if the label is 1 or 0, respectively.\n    References:\n        https://arxiv.org/pdf/1708.02002.pdf\n    Usage:\n     model.compile(loss=[binary_focal_loss(alpha=.25, gamma=2)], metrics=[\"accuracy\"], optimizer=adam)\n    \"\"\"\n    def binary_focal_loss_fixed(y_true, y_pred):\n        \"\"\"\n        y_true shape need be (None,1)\n        y_pred need be compute after sigmoid\n        \"\"\"\n        y_true = tf.cast(y_true, tf.float32)\n        y_pred = tf.clip_by_value(y_pred, tf.keras.backend.epsilon(), 1-tf.keras.backend.epsilon())\n        pt_1 = tf.where(tf.equal(y_true, 1), y_pred, tf.ones_like(y_pred))\n        pt_0 = tf.where(tf.equal(y_true, 0), y_pred, tf.zeros_like(y_pred))\n\n        epsilon = tf.keras.backend.epsilon()\n        # clip to prevent NaN's and Inf's\n        pt_1 = tf.keras.backend.clip(pt_1, epsilon, 1. - epsilon)\n        pt_0 = tf.keras.backend.clip(pt_0, epsilon, 1. - epsilon)\n        \n        if class_weights is not None:\n            weight_0 = class_weights[0]\n            weight_1 = class_weights[1]\n            \n            loss = -weight_0*tf.keras.backend.mean(alpha * tf.keras.backend.pow(1. - pt_1, gamma) * tf.keras.backend.log(pt_1)) \\\n               -weight_1*tf.keras.backend.mean((1 - alpha) * tf.keras.backend.pow(pt_0, gamma) * tf.keras.backend.log(1. - pt_0))\n            \n        else:\n            \n            loss = -tf.keras.backend.mean(alpha * tf.keras.backend.pow(1. - pt_1, gamma) * tf.keras.backend.log(pt_1)) \\\n               -tf.keras.backend.mean((1 - alpha) * tf.keras.backend.pow(pt_0, gamma) * tf.keras.backend.log(1. - pt_0))\n        \n        \n        return loss\n\n    return binary_focal_loss_fixed","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef create_model():\n    # Create the model\n    # Verify Mixed Policy Settings\n    print(f'Compute dtype: {tf.keras.mixed_precision.global_policy().compute_dtype}')\n    print(f'Variable dtype: {tf.keras.mixed_precision.global_policy().variable_dtype}')\n    \n    with STRATEGY.scope():\n        \n        # Set seed for deterministic weights initialization\n        seed_everything()\n        \n        model = tf.keras.Sequential()\n        \n        # Add the first convolutional layer\n        model.add(keras.layers.Conv2D(32, kernel_size=(5, 5), activation='relu', input_shape=(512,512, 3)))\n        model.add(keras.layers.MaxPooling2D(pool_size=(2, 2)))\n\n        # Add the second convolutional layer\n        model.add(keras.layers.Conv2D(32, kernel_size=(5, 5), activation='relu'))\n        model.add(keras.layers.MaxPooling2D(pool_size=(2, 2)))\n\n        # Add the third convolutional layer\n        model.add(keras.layers.Conv2D(64, kernel_size=(3, 3), activation='relu'))\n        model.add(keras.layers.MaxPooling2D(pool_size=(2, 2)))\n\n        # Flatten the output from the convolutional layers\n        model.add(keras.layers.Flatten())\n\n        # Add a dense layer\n        model.add(keras.layers.Dense(256, activation='relu'))\n\n        # Add the final dense layer for binary classification\n        model.add(keras.layers.Dense(1, activation='sigmoid'))\n        \n        # We will use the famous Adam optimizer for fast learning\n        optimizer = tf.optimizers.Adam(learning_rate=LR_MAX, epsilon=1e-7, clipnorm=10.0)\n        \n        # Loss\n        loss = binary_focal_loss(gamma=2., alpha=.25, class_weights = {0: 1., 1: 10.})\n\n        \n        # Metrics\n        metrics = [\n            #tfa.metrics.F1Score(num_classes=1, threshold=0.50),\n            tf_pfbeta(beta=1.0, from_logits=True),\n            tf.keras.metrics.Precision(),\n            tf.keras.metrics.Recall(),\n            tf.keras.metrics.AUC(),\n            tf.keras.metrics.BinaryAccuracy(),\n        ]\n        \n        model.compile(optimizer=optimizer, loss=loss, metrics=metrics)\n\n\n        return model\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pretrained File Path: '/kaggle/input/sartorius-training-dataset/model.h5'\ntf.keras.backend.clear_session()\n# enable XLA optmizations\ntf.config.optimizer.set_jit(True)\n\nmodel = create_model()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot model summary\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import clear_output\n\ntraining_callbacks = [\n    tf.keras.callbacks.ModelCheckpoint(\n       filepath='model.{epoch:02d}-{val_loss:.4f}-{val_pfbeta:.4f}.h5',\n       monitor='val_pfbeta',\n       mode='max',\n       save_best_only=True\n   ),\n]\n\nhistory = model.fit(\n        training_dataset,\n        validation_data = valid_dataset,\n        epochs = 10,\n        callbacks=training_callbacks,\n        verbose = 1,\n    );\nclear_output()","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For inference check out the [notebook](https://www.kaggle.com/kuntalpal/rsna-inference).","metadata":{}},{"cell_type":"code","source":"# Test Images\n\nDATASET_PATH='/kaggle/input/rsna-breast-cancer-detection/'\ntest_df = pd.read_csv(os.path.join(DATASET_PATH, \"test.csv\"))\ndisplay(test_df.head())\nprint(f'cases: {len(test_df)}')\n\n# Show sample submission example\n\ndf_sub = pd.read_csv(os.path.join(DATASET_PATH, \"sample_submission.csv\"))\ndisplay(df_sub.head())\nprint(f'cases: {len(df_sub)}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Images\n\nDATASET_PATH='/kaggle/input/rsna-breast-cancer-detection/'\ntest_df = pd.read_csv(os.path.join(DATASET_PATH, \"test.csv\"))\ndisplay(test_df.head())\nprint(f'cases: {len(test_df)}')\n\n# Show sample submission example\n\ndf_sub = pd.read_csv(os.path.join(DATASET_PATH, \"sample_submission.csv\"))\ndisplay(df_sub.head())\nprint(f'cases: {len(df_sub)}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\ndf = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\n\ntest_path='/kaggle/input/rsna-breast-cancer-detection/test_images'\n\ntest_dir = f'{test_path}'\n\n\n\n\ntest_df['img_path']= f'{test_dir}'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.dcm'\n\n#os.makedirs('/kaggle/working/output/test')\n\nfor f in test_df['img_path']:   # remove \"[:10]\" to convert all images \n\n    ds = pydicom.read_file(f) # read dicom image\n    img = tf.convert_to_tensor(ds.pixel_array) # get image array\n    img = tf.cast(img, dtype=tf.float32)\n    \n    img = tf.math.div(\n       tf.math.subtract(\n          img, \n          tf.reduce_min(img)\n       ), \n       tf.math.subtract(\n          tf.reduce_max(img), \n          tf.reduce_min(img)\n           )\n        )\n    print(img)\n    name = f.split('/')[-1][:-4]\n    img = ((img - img.min()) / (img.max() - img.min()))*255\n    \n    if ds.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n        \n    img = cv2.resize(img,(1344,768)) \n\n    img = np.uint8(img)\n    img = Image.fromarray(img)\n    img.save('/kaggle/working/output/test/{}.png'.format(name))\n    #cv2.imwrite('/kaggle/working/output/test/{}.png'.format(name), (img * 255).astype(np.uint8))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path):\n    img = tf.io.read_file(image_path)\n    img = tf.io.decode_image(img, channels = 1)\n    img = tf.reshape(img, [1344, 768,1])\n    img = tf.cast(img, dtype = tf.float32)\n    img = img/255.0\n    return img\n\ntest_image_paths = []\ntest_dir = os.listdir('/kaggle/working/output/test')\nfor i in range(len(test_dir)):\n    img_path = '/kaggle/working/output/test' + '/' + test_dir[i]\n    test_image_paths.append(img_path)\n    \nfor path in test_image_paths:\n    img = load_image(path)\n    \ntest_img_path_ds = tf.data.Dataset.from_tensor_slices(test_image_paths)\ntest_ds = test_img_path_ds.map(load_image)\ntest_ds = test_ds.batch(32)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}