{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":5089415,"sourceType":"datasetVersion","datasetId":2694061},{"sourceId":7389910,"sourceType":"datasetVersion","datasetId":2803290}],"dockerImageVersionId":30665,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Install ConvNextV2 Models From Keras CV Attention Models Pip Package\n!pip install -qq /kaggle/input/keras-cv-attention-models/keras_cv_attention_models-1.3.20-py3-none-any.whl\n!pip install tensorflow-addons","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:12:16.375822Z","iopub.execute_input":"2024-03-18T21:12:16.37751Z","iopub.status.idle":"2024-03-18T21:12:44.015798Z","shell.execute_reply.started":"2024-03-18T21:12:16.377471Z","shell.execute_reply":"2024-03-18T21:12:44.014775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport glob\nimport numpy as np\nfrom tqdm import tqdm\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport re\nimport pandas as pd\nimport random\nfrom keras_cv_attention_models import convnext\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nfrom sklearn.model_selection import train_test_split\nfrom torchvision import transforms\n\nDEBUG = True","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:12:50.441443Z","iopub.execute_input":"2024-03-18T21:12:50.442599Z","iopub.status.idle":"2024-03-18T21:12:55.703108Z","shell.execute_reply.started":"2024-03-18T21:12:50.442555Z","shell.execute_reply":"2024-03-18T21:12:55.702215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    TPU = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', TPU.master())\nexcept ValueError:\n    print('Running on GPU')\n    TPU = None\n\nif TPU:\n    IS_TPU = True\n    tf.config.experimental_connect_to_cluster(TPU)\n    tf.tpu.experimental.initialize_tpu_system(TPU)\n    STRATEGY = tf.distribute.experimental.TPUStrategy(TPU)\nelse:\n    IS_TPU = False\n    STRATEGY = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n\nN_REPLICAS = STRATEGY.num_replicas_in_sync\nprint(f'N_REPLICAS: {N_REPLICAS}, IS_TPU: {IS_TPU}')","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:13:32.733573Z","iopub.execute_input":"2024-03-18T21:13:32.734287Z","iopub.status.idle":"2024-03-18T21:13:32.745128Z","shell.execute_reply.started":"2024-03-18T21:13:32.734254Z","shell.execute_reply":"2024-03-18T21:13:32.743722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_HEIGHT = 1344\nIMG_WIDTH = 768\nN_CHANNELS = 1\nINPUT_SHAPE = (IMG_HEIGHT, IMG_WIDTH, 1)\nLR_MAX = 5e-6 * N_REPLICAS\nWD_RATIO = 0.01","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:13:54.537912Z","iopub.execute_input":"2024-03-18T21:13:54.538685Z","iopub.status.idle":"2024-03-18T21:13:54.543553Z","shell.execute_reply.started":"2024-03-18T21:13:54.538653Z","shell.execute_reply":"2024-03-18T21:13:54.542508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_images = glob.glob('/kaggle/input/rsna-breast-cancer-detection-poi-images/bc_1280_train_lut/*')","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:13:56.605315Z","iopub.execute_input":"2024-03-18T21:13:56.60616Z","iopub.status.idle":"2024-03-18T21:13:58.287507Z","shell.execute_reply.started":"2024-03-18T21:13:56.606125Z","shell.execute_reply":"2024-03-18T21:13:58.286478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\nlabels = []\nimages = []\n\n\nfor fname in all_images[:int(len(all_images) / 30)]:\n    elements = os.path.basename(fname).split('_')\n    patient_id = int(elements[0])\n    image_id = int(elements[1][:-4])\n    image = Image.open(fname)\n    image = image.resize((IMG_WIDTH, IMG_HEIGHT))\n    images.append(np.array(image))\n\n    row = df.loc[(df['patient_id'] == patient_id) & (df['image_id'] == image_id)]\n    labels.append([row['cancer'].values[0]])\n\nimages = np.array(images)\nlabels = np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:17:29.897712Z","iopub.execute_input":"2024-03-18T21:17:29.898742Z","iopub.status.idle":"2024-03-18T21:18:15.477706Z","shell.execute_reply.started":"2024-03-18T21:17:29.898704Z","shell.execute_reply":"2024-03-18T21:18:15.476668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seed all random number generators\ndef seed_everything(seed=43):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_everything()","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:18:56.926276Z","iopub.execute_input":"2024-03-18T21:18:56.927254Z","iopub.status.idle":"2024-03-18T21:18:56.933217Z","shell.execute_reply.started":"2024-03-18T21:18:56.927218Z","shell.execute_reply":"2024-03-18T21:18:56.93205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef normalize(image):\n    # Repeat channels to create 3 channel images required by pretrained ConvNextV2 models\n    '''grayscale_tfm = transforms.Grayscale(num_output_channels=3)\n    image = grayscale_tfm(image)'''\n\n    image = tf.repeat(image, repeats=3, axis=3)\n    # Cast to float 32\n    image = tf.cast(image, tf.float32)\n    # Normalize with respect to ImageNet mean/std\n    image = tf.keras.applications.imagenet_utils.preprocess_input(image, mode='torch')\n\n    return image\n\ndef get_model():\n    # Verify Mixed Policy Settings\n    print(f'Compute dtype: {tf.keras.mixed_precision.global_policy().compute_dtype}')\n    print(f'Variable dtype: {tf.keras.mixed_precision.global_policy().variable_dtype}')\n    \n    with STRATEGY.scope():\n        # Set seed for deterministic weights initialization\n        seed_everything()\n        \n        # Inputs, note the names are equal to the dictionary keys in the dataset\n        image = tf.keras.layers.Input(INPUT_SHAPE, name='image', dtype=tf.uint8)\n        \n        # Normalize Input\n        image_norm = normalize(image)\n        \n        # CNN Prediction in range [0,1]\n        x = convnext.ConvNeXtV2Tiny(\n            input_shape=(IMG_HEIGHT, IMG_WIDTH, 3),\n            pretrained='imagenet21k-ft1k',\n            num_classes=0,\n        )(image_norm)\n        \n        # Average Pooling BxHxWxC -> BxC\n        x = tf.keras.layers.GlobalAveragePooling2D()(x)\n        # Dropout to prevent Overfitting\n        x = tf.keras.layers.Dropout(0.30)(x)\n        # Output value between [0, 1] using Sigmoid function\n        outputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n\n        # We will use the famous AdamW optimizer for fast learning with weight decay\n        optimizer = tfa.optimizers.AdamW(learning_rate=LR_MAX, weight_decay=LR_MAX*WD_RATIO, epsilon=1e-6)\n\n        # Loss\n        loss = tf.keras.losses.BinaryCrossentropy(from_logits=False)\n        \n        # Metrics\n        metrics = [\n            tf.keras.metrics.TruePositives(name='tp'),\n            tf.keras.metrics.FalsePositives(name='fp'),\n            tf.keras.metrics.TrueNegatives(name='tn'),\n            tf.keras.metrics.FalseNegatives(name='fn'), \n            tfa.metrics.F1Score(num_classes=1, threshold=0.50),\n            tf.keras.metrics.Precision(),\n            tf.keras.metrics.Recall(),\n            tf.keras.metrics.AUC(),\n            tf.keras.metrics.BinaryAccuracy(),\n        ]\n\n        model = tf.keras.models.Model(inputs=image, outputs=outputs)\n        \n        model.compile(optimizer=optimizer, loss=loss, metrics=metrics)\n\n        return model","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:25:36.187727Z","iopub.execute_input":"2024-03-18T21:25:36.188555Z","iopub.status.idle":"2024-03-18T21:25:36.20394Z","shell.execute_reply.started":"2024-03-18T21:25:36.18851Z","shell.execute_reply":"2024-03-18T21:25:36.202736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = get_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:25:39.65694Z","iopub.execute_input":"2024-03-18T21:25:39.65784Z","iopub.status.idle":"2024-03-18T21:25:45.701865Z","shell.execute_reply.started":"2024-03-18T21:25:39.657795Z","shell.execute_reply":"2024-03-18T21:25:45.700724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(images, labels, test_size=0.25, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:19:30.788236Z","iopub.execute_input":"2024-03-18T21:19:30.788614Z","iopub.status.idle":"2024-03-18T21:19:31.336548Z","shell.execute_reply.started":"2024-03-18T21:19:30.788584Z","shell.execute_reply":"2024-03-18T21:19:31.335562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = model.evaluate(\n        X_test, y_test,\n        verbose=1,\n    )","metadata":{"execution":{"iopub.status.busy":"2024-03-18T21:19:39.263815Z","iopub.execute_input":"2024-03-18T21:19:39.264733Z","iopub.status.idle":"2024-03-18T21:21:10.639807Z","shell.execute_reply.started":"2024-03-18T21:19:39.2647Z","shell.execute_reply":"2024-03-18T21:21:10.638926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n        X_train, y_train,\n        validation_data = (X_test, y_test),\n        epochs = 10,\n        verbose = 1,\n        batch_size = 8 * N_REPLICAS,\n        class_weight = {\n            0: 1.0,\n            1: 5.0,\n        },\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grayscale_tfm = transforms.Grayscale(num_output_channels=3)\nmeh = Image.open(all_images[0])\nmeh = grayscale_tfm(meh)\nmeh = np.array([meh])\ngeh = np.array([[0]])","metadata":{"execution":{"iopub.status.busy":"2024-03-18T04:03:37.94669Z","iopub.execute_input":"2024-03-18T04:03:37.947357Z","iopub.status.idle":"2024-03-18T04:03:37.976975Z","shell.execute_reply.started":"2024-03-18T04:03:37.947323Z","shell.execute_reply":"2024-03-18T04:03:37.97596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = new_model.evaluate(\n        meh, geh,\n        verbose=1,\n    )","metadata":{"execution":{"iopub.status.busy":"2024-03-18T04:03:57.162307Z","iopub.execute_input":"2024-03-18T04:03:57.162707Z","iopub.status.idle":"2024-03-18T04:03:58.828798Z","shell.execute_reply.started":"2024-03-18T04:03:57.162674Z","shell.execute_reply":"2024-03-18T04:03:58.827357Z"},"trusted":true},"execution_count":null,"outputs":[]}]}