{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:50:58.609397Z","iopub.execute_input":"2022-12-13T06:50:58.609883Z","iopub.status.idle":"2022-12-13T06:51:20.881484Z","shell.execute_reply.started":"2022-12-13T06:50:58.609848Z","shell.execute_reply":"2022-12-13T06:51:20.879918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport gdcm\nimport glob\nfrom IPython.display import display\nimport os\nfrom pathlib import Path\nimport re\n\nfrom collections import Counter\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport seaborn as sns\nfrom tqdm.notebook import tqdm\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2022-12-13T06:51:28.315617Z","iopub.execute_input":"2022-12-13T06:51:28.316235Z","iopub.status.idle":"2022-12-13T06:51:29.603156Z","shell.execute_reply.started":"2022-12-13T06:51:28.31618Z","shell.execute_reply":"2022-12-13T06:51:29.601133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = '/kaggle/input/rsna-breast-cancer-detection'\ndf_train = pd.read_csv(ROOT + '/train.csv')\ndf_test = pd.read_csv(ROOT + '/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:49:41.660458Z","iopub.execute_input":"2022-12-13T10:49:41.661044Z","iopub.status.idle":"2022-12-13T10:49:41.748787Z","shell.execute_reply.started":"2022-12-13T10:49:41.660995Z","shell.execute_reply":"2022-12-13T10:49:41.747521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('train.csv')\ndisplay(df_train.head().style.background_gradient(cmap=\"Pastel1\"))\nprint('test.csv')\ndisplay(df_test.head().style.background_gradient(cmap=\"Pastel1\"))","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:49:44.718032Z","iopub.execute_input":"2022-12-13T10:49:44.718451Z","iopub.status.idle":"2022-12-13T10:49:44.833871Z","shell.execute_reply.started":"2022-12-13T10:49:44.718419Z","shell.execute_reply":"2022-12-13T10:49:44.832521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def base_info(df):\n    features = []\n    dtypes = []\n    unique_values = []\n    for col in df.columns:\n        features.append(col)\n        dtypes.append(df[col].dtype) \n        unique_values.append(len(df[col].unique()))\n    \n    df_ = pd.DataFrame({\n        'features': features,\n        'dtypes': dtypes,\n        'unique values':unique_values\n    })\n    display(df_.style.background_gradient(cmap=\"Pastel1\"))\n\n\ndef show_correlation(df, features):\n    df = df.loc[:, features]\n    corr = df.corr()\n    fig = plt.figure(figsize=(9, 9))\n    sns.heatmap(corr, annot=True,\n                fmt='.2f',\n                cmap=sns.color_palette('coolwarm',200))\n    plt.title('Correlation')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T11:12:30.760673Z","iopub.execute_input":"2022-12-09T11:12:30.761844Z","iopub.status.idle":"2022-12-09T11:12:30.771542Z","shell.execute_reply.started":"2022-12-09T11:12:30.761794Z","shell.execute_reply":"2022-12-09T11:12:30.770601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nbase_info(df_train)\n\nfeatures = ['site_id', 'age', 'implant', 'machine_id',\n            'biopsy', 'invasive', 'BIRADS',  \n            'difficult_negative_case', 'cancer'] \n           # remove ['patient_id', 'image_id', 'view', 'laterality', 'density']\nshow_correlation(df_train, features=features)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T11:12:36.799755Z","iopub.execute_input":"2022-12-09T11:12:36.800174Z","iopub.status.idle":"2022-12-09T11:12:37.59689Z","shell.execute_reply.started":"2022-12-09T11:12:36.800136Z","shell.execute_reply":"2022-12-09T11:12:37.595826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dcm1 = pydicom.dcmread(ROOT + '/train_images/10006/1459541791.dcm')\ndcm2 = pydicom.dcmread(ROOT + '/train_images/10226/1943220805.dcm')\nprint('1459541791.dcm:\\n', dcm1, '\\n\\n')\nprint('1943220805.dcm:\\n', dcm2)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T11:12:42.764085Z","iopub.execute_input":"2022-12-09T11:12:42.76543Z","iopub.status.idle":"2022-12-09T11:12:43.085751Z","shell.execute_reply.started":"2022-12-09T11:12:42.765387Z","shell.execute_reply":"2022-12-09T11:12:43.084528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dicom_meta_data(sampling_size=50):\n    \n    sample_index = df_train.sample(sampling_size, random_state=2022).index\n    patient_ids = df_train.loc[sample_index, 'patient_id'].to_list()\n    image_ids = df_train.loc[sample_index, 'image_id'].to_list()\n    dcm_paths = [ROOT + f'/train_images/{patient_id}/{image_id}.dcm'\n                 for patient_id, image_id in zip(patient_ids, image_ids)]\n    \n    \n    BodyPartThickness = []\n    CompressionForce = []\n    PhotometricInterpretation = []\n    PixelSpacing = []\n    BitsAllocated = []\n    RescaleIntercept = []\n    RescaleSlope = []\n    Rows = []\n    Columns = []\n    \n    for dcm_path in tqdm(dcm_paths):\n        dcm = pydicom.dcmread(dcm_path)\n        PhotometricInterpretation.append(dcm.PhotometricInterpretation)\n        try:\n            BodyPartThickness.append(dcm.BodyPartThickness)\n        except:\n             BodyPartThickness.append(None)\n        try:\n            CompressionForce.append(dcm.CompressionForce)\n        except:\n            CompressionForce.append(None)\n        try:\n            PixelSpacing.append(dcm.PixelSpacing)\n        except:\n            PixelSpacing.append(None)\n        BitsAllocated.append(dcm.BitsAllocated)    \n        RescaleIntercept.append(dcm.RescaleIntercept)\n        RescaleSlope.append(dcm.RescaleSlope)\n        Rows.append(dcm.Rows)\n        Columns.append(dcm.Columns)    \n\n    df_dcm = pd.DataFrame({\n        'patient_id':patient_ids,\n        'image_id':image_ids,\n        'PhotometricInterpretation': PhotometricInterpretation,\n        'BodyPartThickness': BodyPartThickness,\n        'CompressionForce': CompressionForce,\n        'PixelSpacing': PixelSpacing,\n        'BitsAllocated': BitsAllocated,\n        'RescaleIntercept': RescaleIntercept,\n        'RescaleSlope': RescaleSlope,\n        'Rows': Rows,\n        'Columns': Columns,\n        })\n    \n    return df_dcm\n\n\ndef plot_hist_dicom_meta(df):\n    fig, ax = plt.subplots(2, 4, figsize=(22,10))\n    ax = ax.flatten()\n    sns.set_palette(\"pastel\")\n    sns.histplot(df['PhotometricInterpretation'], ax=ax[0])\n    sns.histplot(df['BodyPartThickness'], ax=ax[1])\n    sns.histplot(df['CompressionForce'], ax=ax[2])\n    sns.countplot(df['BitsAllocated'], ax=ax[3], linewidth=1.0, edgecolor='black')\n    sns.countplot(df['RescaleIntercept'], ax=ax[4], linewidth=1.0, edgecolor='black')\n    sns.countplot(df['RescaleSlope'], ax=ax[5], linewidth=1.0, edgecolor='black')\n    sns.histplot(df['Rows'], ax=ax[6])\n    sns.histplot(df['Columns'], ax=ax[7])","metadata":{"execution":{"iopub.status.busy":"2022-12-09T08:56:22.403961Z","iopub.execute_input":"2022-12-09T08:56:22.40447Z","iopub.status.idle":"2022-12-09T08:56:22.423068Z","shell.execute_reply.started":"2022-12-09T08:56:22.404428Z","shell.execute_reply":"2022-12-09T08:56:22.421621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_dcm = get_dicom_meta_data(sampling_size=5000)\ndisplay(df_dcm.head(10).style.background_gradient(cmap=\"Pastel1\"))\nplot_hist_dicom_meta(df_dcm)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T08:56:57.800537Z","iopub.execute_input":"2022-12-09T08:56:57.801272Z","iopub.status.idle":"2022-12-09T09:01:43.794997Z","shell.execute_reply.started":"2022-12-09T08:56:57.801213Z","shell.execute_reply":"2022-12-09T09:01:43.793548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def processing_pixel_data(dcm, resize=512):\n    \n    img = dcm.pixel_array\n    img = img * dcm.RescaleSlope + dcm.RescaleIntercept\n    img = (img - img.min()) / (img.max() - img.min())\n    if dcm.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n    img = cv2.resize(img, (resize, resize))\n    img = (img * 255).astype(np.uint8)\n\n    return img\n\n\ndef show_dcm_images(patient_id, df):\n    \n    root = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n    dcm_paths = glob.glob(f'{root}/{patient_id}/*.dcm')\n    num_dcm_files = len(dcm_paths)\n     \n    # Sort dcm filed by id\n    def natural_keys(text):\n        keys = []\n        for t in re.split(r'(\\d+)', text):\n            if t.isdigit():\n                t = int(t)\n            keys.append(t)\n        return keys\n    dcm_paths.sort(key=natural_keys)\n            \n    # Plot images\n    ncols = 5\n    nrows = int(np.ceil(num_dcm_files/ncols))\n    fig, axes = plt.subplots(nrows=nrows, ncols=ncols, figsize=(25,5*nrows))\n    fig.suptitle(f'patient_id: {patient_id}', size=20)\n    axes = axes.flatten()\n    for i in range(ncols*nrows):\n        \n        axes[i].axis('off')\n        \n        if i >= num_dcm_files:\n            pass\n        \n        else:\n            dcm = pydicom.dcmread(dcm_paths[i])\n            img = processing_pixel_data(dcm, resize=512)\n            img_id = int(dcm_paths[i].split('/')[-1].split('.')[0])\n            diagnosis = df[df['image_id']==img_id]['cancer'].values[0]\n            axes[i].imshow(img, cmap=\"bone\")\n            axes[i].set_title(f'{\"+\" if diagnosis==1 else \"-\"}',\n                              fontsize=20,\n                              weight='bold')\n        \n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-09T11:13:31.291686Z","iopub.execute_input":"2022-12-09T11:13:31.292421Z","iopub.status.idle":"2022-12-09T11:13:31.308038Z","shell.execute_reply.started":"2022-12-09T11:13:31.292371Z","shell.execute_reply":"2022-12-09T11:13:31.30637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampling_size = 5\n\nprint('Positive cancer patients')\npositive_patient_ids = \\\n    df_train[df_train['cancer']==1]['patient_id'].unique()[:sampling_size]\nfor p_id in tqdm(positive_patient_ids):\n    show_dcm_images(p_id, df_train)\n    \nprint('Negative cancer patients')\nnegative_patient_ids = \\\n    df_train[df_train['cancer']==0]['patient_id'].unique()[:sampling_size]\nfor p_id in tqdm(negative_patient_ids):\n    show_dcm_images(p_id, df_train)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T11:13:35.976191Z","iopub.execute_input":"2022-12-09T11:13:35.976883Z","iopub.status.idle":"2022-12-09T11:14:25.169118Z","shell.execute_reply.started":"2022-12-09T11:13:35.976846Z","shell.execute_reply":"2022-12-09T11:14:25.167845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tensorflow\ndef pFScore(labels, preds, beta=1):\n    eps = 1e-5\n    preds = tf.clip_by_value(preds, 0, 1)\n    y_true_count = tf.reduce_sum(labels)\n    ctp = tf.reduce_sum(preds[labels==1])\n    cfp = tf.reduce_sum(preds[labels==0])\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp + eps)\n    c_recall = ctp / (y_true_count + eps)\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall + eps)\n        return result\n    else:\n        return 0.0\npFScore.__name__='pF1'\n\n# numpy\ndef pfbeta(labels, preds, beta=1):\n    eps = 1e-5\n    preds = preds.clip(0, 1)\n    y_true_count = labels.sum()\n    ctp = preds[labels==1].sum()\n    cfp = preds[labels==0].sum()\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp + eps)\n    c_recall = ctp / (y_true_count + eps)\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall + eps)\n        return result\n    else:\n        return 0.0","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:32:42.640117Z","iopub.execute_input":"2022-12-13T10:32:42.640638Z","iopub.status.idle":"2022-12-13T10:32:42.679138Z","shell.execute_reply.started":"2022-12-13T10:32:42.640536Z","shell.execute_reply":"2022-12-13T10:32:42.677999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q efficientnet >> /dev/null\n!pip install -qU wandb\n!pip install -qU scikit-learn","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:33:45.328848Z","iopub.execute_input":"2022-12-13T10:33:45.329399Z","iopub.status.idle":"2022-12-13T10:34:27.879518Z","shell.execute_reply.started":"2022-12-13T10:33:45.329351Z","shell.execute_reply":"2022-12-13T10:34:27.877594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q efficientnet >> /dev/null\n!pip install -qU wandb\n!pip install -qU scikit-learn","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:43:31.168833Z","iopub.execute_input":"2022-12-13T10:43:31.169854Z","iopub.status.idle":"2022-12-13T10:44:05.879354Z","shell.execute_reply.started":"2022-12-13T10:43:31.169804Z","shell.execute_reply":"2022-12-13T10:44:05.877799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'  # to avoid too many logging messages\nimport pandas as pd, numpy as np, random, shutil\nimport tensorflow as tf, re, math\nimport tensorflow.keras.backend as K\nimport sklearn\nimport matplotlib.pyplot as plt\nimport tensorflow_addons as tfa\nimport tensorflow_probability as tfp\nimport wandb\nimport yaml\n\nfrom IPython import display as ipd\nfrom glob import glob\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold, StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:37:13.665794Z","iopub.execute_input":"2022-12-13T10:37:13.666265Z","iopub.status.idle":"2022-12-13T10:37:16.363172Z","shell.execute_reply.started":"2022-12-13T10:37:13.666231Z","shell.execute_reply":"2022-12-13T10:37:16.361543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wandb\n\ntry:\n    from kaggle_secrets import UserSecretsClient\n    user_secrets = UserSecretsClient()\n    api_key = user_secrets.get_secret(\"WANDB\")\n\n    wandb.login(key=api_key)\n    anonymous = None\nexcept:\n    anonymous = \"must\"\n    print('To use your W&B account,\\nGo to Add-ons -> Secrets and provide your W&B access token. Use the Label name as WANDB. \\nGet your W&B access token from here: https://wandb.ai/authorize')","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:44:13.827456Z","iopub.execute_input":"2022-12-13T10:44:13.828404Z","iopub.status.idle":"2022-12-13T10:44:14.121549Z","shell.execute_reply.started":"2022-12-13T10:44:13.828352Z","shell.execute_reply":"2022-12-13T10:44:14.120366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_bins = 5\ndf[\"age_bin\"] = pd.cut(df['age'].values.reshape(-1), bins=num_bins, labels=False)\n\nstrat_cols = [\n    'laterality', 'view', 'biopsy','invasive', 'BIRADS', 'age_bin',\n    'implant', 'density','machine_id', 'difficult_negative_case',\n    'cancer',\n]\n\ndf['stratify'] = ''\nfor col in strat_cols:\n    df['stratify'] += df[col].astype(str)\n\nskf = StratifiedGroupKFold(n_splits=CFG.folds, shuffle=True, random_state=CFG.seed)\nfor fold, (train_idx, val_idx) in enumerate(skf.split(df, df['stratify'], df[\"patient_id\"])):\n    df.loc[val_idx, 'fold'] = fold\ndisplay(df.groupby(['fold', \"cancer\"]).size())","metadata":{"execution":{"iopub.status.busy":"2022-12-13T11:00:33.074357Z","iopub.execute_input":"2022-12-13T11:00:33.074875Z","iopub.status.idle":"2022-12-13T11:00:42.805343Z","shell.execute_reply.started":"2022-12-13T11:00:33.074833Z","shell.execute_reply":"2022-12-13T11:00:42.804187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    wandb         = True\n    competition   = 'rsna-bcd' \n    _wandb_kernel = 'awsaf49'\n    debug         = False\n    comment       = 'EfficientNetB4-1024x512-roi-up=10-lr4-focal-vflip'\n    exp_name      = 'roi-v2-fix' # name of the experiment, folds will be grouped using 'exp_name'\n    \n    # use verbose=0 for silent, vebose=1 for interactive,\n    verbose      = 1 if debug else 0\n    display_plot = True\n\n    # device\n    device = \"TPU-1VM\" #or \"GPU\"\n\n    model_name = 'EfficientNetB4'\n\n    # seed for data-split, layer init, augs\n    seed = 42\n\n    # number of folds for data-split\n    folds = 5\n    \n    # which folds to train\n    selected_folds = [0, 1, 2]\n\n    # size of the image\n    img_size = [1024, 512]\n\n    # batch_size and epochs\n    batch_size = 28\n    epochs = 10\n    \n    # upsample\n    upsample = 10\n\n    # loss and optimizer\n    loss      = 'Focal'  # BCE, Focal\n    optimizer = 'Adam'\n\n    # augmentation\n    augment   = True\n\n    # scale-shift-rotate-shear\n    transform = True\n    fill_mode = 'constant'\n    rot    = 2.0\n    shr    = 2.0\n    hzoom  = 50.0\n    wzoom  = 50.0\n    hshift = 10.0\n    wshift = 10.0\n\n    # flip\n    hflip = True\n    vflip = True\n\n    # clip\n    clip = False\n\n    # lr-scheduler\n    scheduler   = 'exp' # cosine\n\n    # dropout\n    drop_prob   = 0.6\n    drop_cnt    = 10\n    drop_size   = 0.08\n    \n    # cut-mix-up\n    mixup_prob = 0.0\n    mixup_alpha = 0.5\n    \n    cutmix_prob = 0.0\n    cutmix_alpha = 2.5\n\n    # pixel-augment\n    pixel_aug = True\n    sat  = [0.7, 1.3]\n    cont = [0.8, 1.2]\n    bri  = 0.15\n    hue  = 0.05\n\n    # test-time augs\n    tta = 1\n    \n    # target column\n    target_col  = ['cancer']","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:38:48.397385Z","iopub.execute_input":"2022-12-13T10:38:48.397862Z","iopub.status.idle":"2022-12-13T10:38:48.411234Z","shell.execute_reply.started":"2022-12-13T10:38:48.397826Z","shell.execute_reply":"2022-12-13T10:38:48.410128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import efficientnet.tfkeras as efn\n\ndef build_model(model_name=CFG.model_name,\n                loss_name=CFG.loss,\n                dim=CFG.img_size,\n                compile_model=True,\n                include_top=False):         \n    base = getattr(efn, model_name)(input_shape=(*dim,3),\n                                    weights='imagenet',\n                                    include_top=False) # get base model (efficientnet), use imgnet weights\n    inp = base.inputs\n    x = base.output\n    x = tf.keras.layers.GlobalAveragePooling2D()(x) # use GAP to get pooling result form conv outputs\n    x = tf.keras.layers.Dense(32, activation='silu')(x) # use activation to apply non-linearity\n    x = tf.keras.layers.Dense(1,activation='sigmoid')(x) # use sigmoid to convert predictions to [0-1]\n    model = tf.keras.Model(inputs=inp,outputs=x)\n    if compile_model:\n        # optimizer\n        opt = tf.keras.optimizers.Adam(learning_rate=0.0001)\n        # loss\n        if loss_name == 'BCE':\n            loss = tf.keras.losses.BinaryCrossentropy(label_smoothing=0.05)\n        elif loss_name == 'Focal':\n            loss = tfa.losses.SigmoidFocalCrossEntropy(alpha=0.80, gamma=2.0)\n        # metric\n        auc = tf.keras.metrics.AUC(name='auc')\n        pf1 = pFScore\n        metrics = [pf1, auc]\n        # compile\n        model.compile(optimizer=opt,\n                      loss=loss,\n                      metrics=metrics)\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:38:50.580787Z","iopub.execute_input":"2022-12-13T10:38:50.581213Z","iopub.status.idle":"2022-12-13T10:38:50.592438Z","shell.execute_reply.started":"2022-12-13T10:38:50.58118Z","shell.execute_reply":"2022-12-13T10:38:50.590958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = build_model(CFG.model_name, dim=CFG.img_size, compile_model=True)\n# tmp.summary()  # too long","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:38:54.325Z","iopub.execute_input":"2022-12-13T10:38:54.325958Z","iopub.status.idle":"2022-12-13T10:38:58.246086Z","shell.execute_reply.started":"2022-12-13T10:38:54.325913Z","shell.execute_reply":"2022-12-13T10:38:58.24453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lr_callback(batch_size=8, plot=False):\n    lr_start   = 0.000003\n    lr_max     = 0.00000095  * batch_size\n    lr_min     = 0.000001\n    lr_ramp_ep = 4\n    lr_sus_ep  = 0\n    lr_decay   = 0.8\n   \n    def lrfn(epoch):\n        if epoch < lr_ramp_ep:\n            lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n            \n        elif epoch < lr_ramp_ep + lr_sus_ep:\n            lr = lr_max\n            \n        elif CFG.scheduler=='exp':\n            lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n            \n        elif CFG.scheduler=='cosine':\n            decay_total_epochs = CFG.epochs - lr_ramp_ep - lr_sus_ep + 3\n            decay_epoch_index = epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            cosine_decay = 0.4 * (1 + math.cos(phase))\n            lr = (lr_max - lr_min) * cosine_decay + lr_min\n        return lr\n    if plot:\n        plt.figure(figsize=(10,5))\n        plt.plot(np.arange(CFG.epochs), [lrfn(epoch) for epoch in np.arange(CFG.epochs)], marker='o')\n        plt.xlabel('epoch'); plt.ylabel('learnig rate')\n        plt.title('Learning Rate Scheduler')\n        plt.show()\n\n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=False)\n    return lr_callback\n\n_=get_lr_callback(CFG.batch_size, plot=True )","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:44:36.749572Z","iopub.execute_input":"2022-12-13T10:44:36.750246Z","iopub.status.idle":"2022-12-13T10:44:37.025285Z","shell.execute_reply.started":"2022-12-13T10:44:36.750208Z","shell.execute_reply":"2022-12-13T10:44:37.024352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.cm as cm, cv2\nfrom tensorflow import keras\n\ndef gen_gradcam_heatmap(img, model, last_conv='top_conv', pred_index=0):\n    \n    grad_model = tf.keras.models.Model(\n        [model.inputs], [model.get_layer(last_conv).output, model.output]\n    )\n\n    \n    with tf.GradientTape() as tape:\n        features, preds = grad_model(img)\n        if pred_index is None:\n            pred_index = tf.argmax(preds[0])\n        class_channel = preds[:, pred_index]\n\n    \n    grads = tape.gradient(class_channel, features)\n\n    \n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n\n    \n    features = features[0]\n    heatmap = features @ pooled_grads[..., tf.newaxis]\n    heatmap = tf.squeeze(heatmap)\n\n   heatmap = tf.maximum(heatmap, 0) / tf.math.reduce_max(heatmap)\n    return heatmap.numpy()\n\ndef get_gradcam(img, model,  alpha=0.4, show=False):\n\n    heatmap = gen_gradcam_heatmap(img, model, last_conv='top_conv', pred_index=0)\n    img     = img[0]\n    heatmap = np.uint8(255 * heatmap)\n\n    jet = cm.get_cmap(\"jet\")\n\n    jet_colors  = jet(np.arange(256))[:, :3]\n    jet_heatmap = jet_colors[heatmap]\n\n    jet_heatmap = cv2.resize(jet_heatmap, dsize=(img.shape[1], img.shape[0]))\n\n    superimposed_img = jet_heatmap * alpha + img\n    superimposed_img = keras.preprocessing.image.array_to_img(superimposed_img)\n\n    # Display Grad CAM\n    if show:\n        plt.imshow(superimposed_img)\n        \n    return superimposed_img","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:45:59.904492Z","iopub.execute_input":"2022-12-13T10:45:59.905964Z","iopub.status.idle":"2022-12-13T10:46:00.204693Z","shell.execute_reply.started":"2022-12-13T10:45:59.905879Z","shell.execute_reply":"2022-12-13T10:46:00.203159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/rsna-breast-cancer-detection'\n\nif CFG.device==\"TPU\":\n    from kaggle_datasets import KaggleDatasets\n    GCS_PATH = KaggleDatasets().get_gcs_path(BASE_PATH.split('/')[-1])","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:58:28.846286Z","iopub.execute_input":"2022-12-13T10:58:28.846797Z","iopub.status.idle":"2022-12-13T10:58:28.853038Z","shell.execute_reply.started":"2022-12-13T10:58:28.846757Z","shell.execute_reply":"2022-12-13T10:58:28.851612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.device==\"TPU\":\n    BASE_PATH = GCS_PATH\n# ROOT = '/kaggle/input/rsna-breast-cancer-detection'\n# df_train = pd.read_csv(ROOT + '/train.csv')\n# df_test = pd.read_csv(ROOT + '/test.csv')\n# train\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf['image_path'] = f'{BASE_PATH}/train_images'\\\n                    + '/' + df.patient_id.astype(str)\\\n                    + '/' + df.image_id.astype(str)\\\n                    + '.dcm'\nprint('Train:')\ndisplay(df.head(2))\n\n# test\ntest_df = pd.read_csv(f'{BASE_PATH}/test.csv')\ntest_df['image_path'] = f'{BASE_PATH}/test_images'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.dcm'\nprint('\\nTest:')\ndisplay(test_df.head(2))","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:58:42.769563Z","iopub.execute_input":"2022-12-13T10:58:42.770013Z","iopub.status.idle":"2022-12-13T10:58:42.995207Z","shell.execute_reply.started":"2022-12-13T10:58:42.769981Z","shell.execute_reply":"2022-12-13T10:58:42.993967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.io.gfile.exists(df.image_path.iloc[0]), tf.io.gfile.exists(test_df.image_path.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2022-12-13T10:59:01.13225Z","iopub.execute_input":"2022-12-13T10:59:01.134994Z","iopub.status.idle":"2022-12-13T10:59:01.149713Z","shell.execute_reply.started":"2022-12-13T10:59:01.134944Z","shell.execute_reply":"2022-12-13T10:59:01.148301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p gradcam","metadata":{"execution":{"iopub.status.busy":"2022-12-13T11:13:20.376961Z","iopub.execute_input":"2022-12-13T11:13:20.378191Z","iopub.status.idle":"2022-12-13T11:13:21.499408Z","shell.execute_reply.started":"2022-12-13T11:13:20.378135Z","shell.execute_reply":"2022-12-13T11:13:21.497575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def wandb_init(fold):\n    config = {k:v for k,v in dict(vars(CFG)).items() if '__' not in k}\n    config.update({\"fold\":int(fold)})\n    yaml.dump(config, open(f'/kaggle/working/config fold-{fold}.yaml', 'w'),)\n    config = yaml.load(open(f'/kaggle/working/config fold-{fold}.yaml', 'r'), Loader=yaml.FullLoader)\n    run    = wandb.init(project=\"rsna-bcd-public\",\n               name=f\"fold-{fold}|dim-{CFG.img_size[0]}x{CFG.img_size[1]}|model-{CFG.model_name}\",\n               config=config,\n               anonymous=anonymous,\n               group=CFG.exp_name\n                    )\n    return run\n\ndef log_wandb(fold):\n    \"log best result for error analysis\"\n    valid_df = df.query(\"fold==@fold\").copy().reset_index(drop=True)\n    if CFG.debug:\n        valid_df = valid_df[:min_samples]\n    valid_df['pred'] = oof_pred[fold].reshape(-1)\n    valid_df['diff'] = abs(valid_df.cancer - valid_df.pred)\n    valid_df = valid_df.sort_values(by='diff', ascending=False)\n    \n    noimg_cols  = ['site_id', 'patient_id', 'image_id', 'laterality', 'view', 'age',\n                   'cancer', 'biopsy', 'invasive', 'BIRADS', 'implant', 'density',\n                   'machine_id', 'difficult_negative_case','width','height', 'fold'] + ['pred','diff']\n    \n    gradcam_df  = pd.concat((valid_df.groupby('cancer').tail(10), \n                             valid_df.groupby('cancer').head(10)), axis=0, ignore_index=True)\n    gradcam_ds  = build_dataset(gradcam_df.image_path, labels=None, cache=False, batch_size=1,\n                   repeat=False, shuffle=False, augment=False)\n    \n    data = []\n    for idx, img in enumerate(tqdm(gradcam_ds, total=40, desc='gradcam ', position=0, leave=True)):\n        gradcam = get_gradcam(img, model)\n        row = gradcam_df[noimg_cols].iloc[idx].tolist()\n        img = img.numpy()[0]\n        img = (img*255.0).astype('uint8')\n        data+=[[*row, wandb.Image(img), wandb.Image(gradcam)]]\n        if idx<10: # save best ones\n            cv2.imwrite(f'gradcam/fold{fold}_{idx:02d}.png', np.concatenate([img, gradcam], axis=1))\n    wandb_table = wandb.Table(data=data, columns=[*noimg_cols, 'image', 'gradcam'])\n    \n    best_epoch = np.argmax(history.history['val_pF1'])\n    best_auc = history.history['val_auc'][best_epoch]\n    wandb.log({\n               'best_pF1_batch':oof_val[-1], \n               'best_pF1':pF1,\n               'best_auc':best_auc,\n               'best_epoch':best_epoch,\n               'viz_table':wandb_table,\n              })","metadata":{"execution":{"iopub.status.busy":"2022-12-13T11:21:30.295172Z","iopub.execute_input":"2022-12-13T11:21:30.295732Z","iopub.status.idle":"2022-12-13T11:21:30.315606Z","shell.execute_reply.started":"2022-12-13T11:21:30.295689Z","shell.execute_reply":"2022-12-13T11:21:30.314064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_pred = []; oof_tar = []; oof_val = []; oof_ids = []; oof_folds = []\npreds = np.zeros((test_df.shape[0],1))\n\nfor fold in np.arange(CFG.folds):\n    \n    if fold not in CFG.selected_folds:\n        continue\n        \n    if CFG.wandb:\n        run = wandb_init(fold)\n        WandbCallback = wandb.keras.WandbCallback(save_model=False)\n            \n    train_df = df.query(\"fold!=@fold\")\n    valid_df = df.query(\"fold==@fold\")\n    \n    pos_df = train_df.query(\"cancer==1\").sample(frac=CFG.upsample, replace=True)\n    neg_df = train_df.query(\"cancer==0\")\n    train_df = pd.concat([pos_df, neg_df], axis=0, ignore_index=True)\n    \n    train_paths = train_df.image_path.values; train_labels = train_df[CFG.target_col].values.astype(np.float32)\n    valid_paths = valid_df.image_path.values; valid_labels = valid_df[CFG.target_col].values.astype(np.float32)\n    test_paths  = test_df.image_path.values\n    \n    index = np.arange(len(train_df))\n    np.random.shuffle(index)\n    train_paths  = train_paths[index]\n    train_labels = train_labels[index]\n    \n    min_samples = CFG.batch_size*2\n    \n    if CFG.debug:\n        train_paths = train_paths[:min_samples]; train_labels = train_labels[:min_samples]\n        valid_paths = valid_paths[:min_samples]; valid_labels = valid_labels[:min_samples]\n    \n    print('#'*40); print('#### FOLD: ',fold)\n    print('#### IMAGE_SIZE: (%i, %i) | MODEL_NAME: %s | BATCH_SIZE: %i'%\n          (CFG.img_size[0],CFG.img_size[1],CFG.model_name,CFG.batch_size))\n    num_train = len(train_paths)\n    num_valid = len(valid_paths)\n    if CFG.wandb:\n        wandb.log({'num_train':num_train,\n                   'num_valid':num_valid})\n    print('#### NUM_TRAIN: {:,} | NUM_VALID: {:,}'.format(num_train, num_valid))\n    \n    K.clear_session()\n    model = build_model(CFG.model_name, dim=CFG.img_size, compile_model=True)\n\n    cache = 'TPU' not in CFG.device\n    train_ds = build_dataset(train_paths, train_labels, cache=cache, batch_size=CFG.batch_size,\n                   repeat=True, shuffle=True, augment=CFG.augment)\n    val_ds = build_dataset(valid_paths, valid_labels, cache=cache, batch_size=CFG.batch_size,\n                   repeat=False, shuffle=False, augment=False)\n    print('#'*40)   \n    \n    callbacks = []\n    sv = tf.keras.callbacks.ModelCheckpoint(\n        'fold-%i.h5'%fold, monitor='val_pF1', verbose=CFG.verbose, save_best_only=True,\n        save_weights_only=False, mode='max', save_freq='epoch')\n    callbacks +=[sv]\n    callbacks += [get_lr_callback(CFG.batch_size)]\n    if CFG.wandb:\n        callbacks.append(WandbCallback)\n        \n    print('Training...')\n    history = model.fit(\n        train_ds, \n        epochs=CFG.epochs if not CFG.debug else 2, \n        callbacks = callbacks, \n        steps_per_epoch=len(train_paths)/CFG.batch_size,\n        validation_data=val_ds, \n        #class_weight = {0:1,1:2},\n        verbose=CFG.verbose\n    )\n    \n    print('Loading best model...')\n    model.load_weights('fold-%i.h5'%fold)  \n    \n    print('Predicting OOF with TTA...')\n    ds_valid = build_dataset(valid_paths, labels=None, cache=False, batch_size=CFG.batch_size*2,\n                   repeat=True, shuffle=False, augment=CFG.tta>1)\n    ct_valid = len(valid_paths); STEPS = CFG.tta * ct_valid/CFG.batch_size/2\n    pred = model.predict(ds_valid,steps=STEPS,verbose=CFG.verbose)[:CFG.tta*ct_valid,] \n    oof_pred.append(np.mean(pred.reshape((CFG.tta, ct_valid,-1)),axis=0))                 \n    \n    oof_tar.append(valid_df[CFG.target_col].values[:(min_samples if CFG.debug else len(valid_df))])\n    oof_folds.append(np.ones_like(oof_tar[-1],dtype='int8')*fold)\n    oof_ids.append(valid_df.image_path.tolist()[:(min_samples if CFG.debug else len(valid_df))])\n    \n    print('Predicting Test...')\n    ds_test = build_dataset(test_paths, labels=None, cache=False, \n                    batch_size=(CFG.batch_size*2 if len(test_df)>4 else 1),\n                   repeat=True, shuffle=False, augment=CFG.tta>1)\n    ct_test = len(test_paths); STEPS = 1 if len(test_df)<=4 else (CFG.tta * ct_test/CFG.batch_size/2)\n    pred = model.predict(ds_test,steps=STEPS,verbose=CFG.verbose)[:CFG.tta*ct_test,] \n    preds[:ct_test, :] += np.mean(pred.reshape((CFG.tta, ct_test,-1)),axis=0) / CFG.folds # not meaningful for DIBUG = True\n    \n    y_true = oof_tar[-1]; y_pred = oof_pred[-1]\n    pF1 = pfbeta(y_true.astype(np.float32), y_pred)\n    oof_val.append(np.max(history.history['val_pF1'] ))\n    print('>>>> FOLD %i OOF pF1_batch = %.3f, pF1 = %.3f\\n'%(fold,oof_val[-1], pF1))  # pF1_batch => pF1 batchwise canculated then aggregated\n    \n    if CFG.wandb:\n        log_wandb(fold) # log\n        wandb.run.finish() # finish the run\n        display(ipd.IFrame(run.url, width=1080, height=720)) # show wandb dashboard","metadata":{"execution":{"iopub.status.busy":"2022-12-13T11:21:39.167497Z","iopub.execute_input":"2022-12-13T11:21:39.168122Z","iopub.status.idle":"2022-12-13T11:22:29.464591Z","shell.execute_reply.started":"2022-12-13T11:21:39.168069Z","shell.execute_reply":"2022-12-13T11:22:29.46307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}