{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport torch\nimport pandas as pd\nimport matplotlib.pyplot as plt\n# import seaborn as sns\nimport os\nimport torch.nn as nn\nfrom tqdm import tqdm\nimport torch.nn.functional as F\nimport torch.optim.lr_scheduler as lr_scheduler\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-05T05:08:33.85713Z","iopub.execute_input":"2023-06-05T05:08:33.857595Z","iopub.status.idle":"2023-06-05T05:08:37.425489Z","shell.execute_reply.started":"2023-06-05T05:08:33.857566Z","shell.execute_reply":"2023-06-05T05:08:37.424239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main_directory = \"/kaggle/input/google-research-identify-contrails-reduce-global-warming\"\ntrain_directory = os.path.join(main_directory,\"train\")\ndir_list = os.listdir(train_directory)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:08:37.427414Z","iopub.execute_input":"2023-06-05T05:08:37.428255Z","iopub.status.idle":"2023-06-05T05:08:37.696528Z","shell.execute_reply.started":"2023-06-05T05:08:37.428224Z","shell.execute_reply":"2023-06-05T05:08:37.695436Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_T11_BOUNDS = (243, 303)\n_CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n_TDIFF_BOUNDS = (-4, 2)\n\ndef normalize_range(data, bounds):\n    \"\"\"Maps data to the range [0, 1].\"\"\"\n    return (data - bounds[0]) / (bounds[1] - bounds[0])\n\ndef normalize_std(spec):\n    return (spec- np.mean(spec))/np.std(spec)\n\nclass Dataset(torch.utils.data.Dataset):\n    def __init__(self, data_path, mode='train'):\n        self.data_path = data_path\n        self.file_name = os.listdir(data_path)\n        self.mode = mode\n        \n\n    def __len__(self):\n#         return 20\n#         print(f\" length is {len(self.file_name)}\")\n        return len(self.file_name)\n#         return 5000\n        \n\n    def __getitem__(self, i):\n        \n        band11 = np.load(os.path.join(self.data_path, self.file_name[i], 'band_11.npy'))\n        band14 = np.load(os.path.join(self.data_path, self.file_name[i], 'band_14.npy'))\n        band15 = np.load(os.path.join(self.data_path, self.file_name[i], 'band_15.npy'))\n        \n        r = normalize_range(band15 - band14, _TDIFF_BOUNDS)\n        g = normalize_range(band14 - band11, _CLOUD_TOP_TDIFF_BOUNDS)\n        b = normalize_range(band14, _T11_BOUNDS)\n        x = np.transpose(np.clip(np.stack([r, g, b], axis=2), 0, 1)[:,:,:,4],(2,0,1))\n        x = normalize_std(x)\n        \n        if self.mode == 'train':\n            y = np.load(os.path.join(self.data_path, self.file_name[i], 'human_pixel_masks.npy')).astype(np.float32).transpose(2,0,1)\n        elif self.mode == 'test':\n            y = self.file_name[i]\n        else:\n            y = None\n        \n        return x, y","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:08:37.697749Z","iopub.execute_input":"2023-06-05T05:08:37.69827Z","iopub.status.idle":"2023-06-05T05:08:37.708919Z","shell.execute_reply.started":"2023-06-05T05:08:37.698242Z","shell.execute_reply":"2023-06-05T05:08:37.707978Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 1\nnum_workers = 2\n\ntest_dataset = Dataset('/kaggle/input/google-research-identify-contrails-reduce-global-warming/test', mode='test')\ntrain_dataset = Dataset('/kaggle/input/google-research-identify-contrails-reduce-global-warming/train', mode='train')\n\ntest_loader = torch.utils.data.DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=num_workers, drop_last=True, pin_memory=True)\ntrain_loader = torch.utils.data.DataLoader(train_dataset, batch_size=batch_size, shuffle=False, num_workers=num_workers, drop_last=True, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:08:37.711608Z","iopub.execute_input":"2023-06-05T05:08:37.712148Z","iopub.status.idle":"2023-06-05T05:08:37.733572Z","shell.execute_reply.started":"2023-06-05T05:08:37.71212Z","shell.execute_reply":"2023-06-05T05:08:37.732866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# plotting some train valuess","metadata":{}},{"cell_type":"code","source":"for iterations,(X,y) in enumerate(train_loader):\n    if iterations in [3,4,5]:\n        print(X.shape)\n        X = np.transpose(X[0],(1,2,0))\n        y = np.transpose(y[0],(1,2,0))\n        plt.imshow(X)\n        plt.imshow(y, cmap='viridis', alpha=0.7) \n        plt.show()\n    elif iterations>5:\n        break\n    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-05T05:08:37.736Z","iopub.execute_input":"2023-06-05T05:08:37.736542Z","iopub.status.idle":"2023-06-05T05:08:39.182985Z","shell.execute_reply.started":"2023-06-05T05:08:37.736512Z","shell.execute_reply":"2023-06-05T05:08:39.181615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We need to Define UNet before we can load it ","metadata":{}},{"cell_type":"code","source":"# Designing the Unet Model\n\n\n        \nclass DoubleConv(nn.Module):\n    def __init__(self, in_channels, out_channels):\n        super(DoubleConv, self).__init__()\n        self.double_conv = nn.Sequential(\n            nn.Conv2d(in_channels, out_channels, kernel_size=3, padding=1),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(out_channels, out_channels, kernel_size=3, padding=1),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True)\n        )\n        \n    def forward(self, x):\n        return self.double_conv(x)\n\nclass Down(nn.Module):\n    def __init__(self, in_channels, out_channels):\n        super(Down, self).__init__()\n        self.maxpool_conv = nn.Sequential(\n            nn.MaxPool2d(2),\n            DoubleConv(in_channels, out_channels)\n        )\n\n    def forward(self, x):\n        return self.maxpool_conv(x)\n\nclass Up(nn.Module):\n    def __init__(self, in_channels, out_channels, bilinear=True):\n        super(Up, self).__init__()\n\n        if bilinear:\n            self.up = nn.Upsample(scale_factor=2, mode='bilinear', align_corners=True)\n        else:\n            self.up = nn.ConvTranspose2d(in_channels // 2, in_channels // 2, kernel_size=2, stride=2)\n\n        self.conv = DoubleConv(in_channels, out_channels)\n\n    def forward(self, x1, x2):\n        x1 = self.up(x1)\n\n        diffY = x2.size()[2] - x1.size()[2]\n        diffX = x2.size()[3] - x1.size()[3]\n\n        x1 = nn.functional.pad(x1, [diffX // 2, diffX - diffX // 2, diffY // 2, diffY - diffY // 2])\n        \n        x = torch.cat([x2, x1], dim=1)\n        return self.conv(x)\n    \n    \nclass UNet(nn.Module):\n    def __init__(self):\n        super(UNet, self).__init__()\n        # Define your layers\n        self.inc = DoubleConv(3, 64)\n        self.down1 = Down(64, 128)\n        self.down2 = Down(128, 256)\n        self.down3 = Down(256, 512)\n        self.down4 = Down(512, 512)\n        self.up1 = Up(1024, 256)\n        self.up2 = Up(512, 128)\n        self.up3 = Up(256, 64)\n        self.up4 = Up(128, 64)\n        self.outc = nn.Conv2d(64, 1, kernel_size=1)\n\n    def forward(self, x):\n        # Forward pass through the layers\n        x1 = self.inc(x)\n        x2 = self.down1(x1)\n        x3 = self.down2(x2)\n        x4 = self.down3(x3)\n        x5 = self.down4(x4)\n        x = self.up1(x5, x4)\n        x = self.up2(x, x3)\n        x = self.up3(x, x2)\n        x = self.up4(x, x1)\n        x = self.outc(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:08:39.18492Z","iopub.execute_input":"2023-06-05T05:08:39.185214Z","iopub.status.idle":"2023-06-05T05:08:39.203454Z","shell.execute_reply.started":"2023-06-05T05:08:39.185186Z","shell.execute_reply":"2023-06-05T05:08:39.202154Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = torch.load(\"/kaggle/input/predictiowith-unet/UnetBest.pth\",map_location=torch.device('cpu'))\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:09:08.584041Z","iopub.execute_input":"2023-06-05T05:09:08.584463Z","iopub.status.idle":"2023-06-05T05:09:09.773816Z","shell.execute_reply.started":"2023-06-05T05:09:08.584437Z","shell.execute_reply":"2023-06-05T05:09:09.772856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(x, fg_val=1):\n    \"\"\"\n    Args:\n        x:  numpy array of shape (height, width), 1 - mask, 0 - background\n    Returns: run length encoding as list\n    \"\"\"\n\n    dots = np.where(\n        x.T.flatten() == fg_val)[0]  # .T sets Fortran order down-then-right\n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return run_lengths\n\n\ndef list_to_string(x):\n    \"\"\"\n    Converts list to a string representation\n    Empty list returns '-'\n    \"\"\"\n    if x: # non-empty list\n        s = str(x).replace(\"[\", \"\").replace(\"]\", \"\").replace(\",\", \"\")\n    else:\n        s = '-'\n    return s\n\n\ndef rle_decode(mask_rle, shape=(256, 256)):\n    '''\n    mask_rle: run-length as string formatted (start length)\n              empty predictions need to be encoded with '-'\n    shape: (height, width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    if mask_rle != '-': \n        s = mask_rle.split()\n        starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n        starts -= 1\n        ends = starts + lengths\n        for lo, hi in zip(starts, ends):\n            img[lo:hi] = 1\n    return img.reshape(shape, order='F')  # Needed to align to RLE direction","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:09:12.187168Z","iopub.execute_input":"2023-06-05T05:09:12.187536Z","iopub.status.idle":"2023-06-05T05:09:12.197754Z","shell.execute_reply.started":"2023-06-05T05:09:12.187508Z","shell.execute_reply":"2023-06-05T05:09:12.196488Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tweaking the Threshold","metadata":{}},{"cell_type":"markdown","source":"Although the baseline for me was 0.34 when threshold was 0.5 , but the I read this article mentioned in the notebook : https://www.kaggle.com/code/mnokno/getting-started-eda-model-train-submit?kernelSessionId=131699711\n'' There is a significant imbalance in the number of pixels between the negative class (non-contrail) and the positive class (contrail). In the training and validation datasets, the ratio of negative to positive class pixels is 85:1 and 164:1 respectively. This severe class imbalance can lead to biased models that prioritize the majority class, ultimately reducing the overall prediction quality.\n\nTo address this issue, we can employ two strategies: optimizing the confidence threshold during post-processing or incorporating class weights into the loss function during training.\n\nOptimizing confidence threshold during post-processing: After training the model, we can adjust the confidence threshold used to determine class predictions. By carefully selecting the threshold, we can increase the sensitivity to the positive class, thereby improving the detection of contrails. This approach allows us to fine-tune the model's predictions without retraining it.''\n\n\n#### So , Now the thing revolves around which treshold is best for the test set , and from some tweaking , the best result so far is the threshold between (0.6-0.7) ( You can of course check for different values,as I have not done)","metadata":{}},{"cell_type":"code","source":"# making the dataset for the imbalance \nhave_mask = 0 \nno_mask = 0\n\ntrain_dataset =os.listdir(\"/kaggle/input/google-research-identify-contrails-reduce-global-warming/train\")\n\nfor items in train_dataset:\n    mask = np.load(os.path.join(\"/kaggle/input/google-research-identify-contrails-reduce-global-warming/train\",items,\"human_pixel_masks.npy\"))\n    if (np.sum(mask)) == 0:\n        no_mask =no_mask+ 1\n    else:\n        have_mask=have_mask+1\nprint(have_mask)","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:09:13.189348Z","iopub.execute_input":"2023-06-05T05:09:13.189772Z","iopub.status.idle":"2023-06-05T05:11:27.892131Z","shell.execute_reply.started":"2023-06-05T05:09:13.18974Z","shell.execute_reply":"2023-06-05T05:11:27.890928Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nsns.barplot(x = [\"Contrails\",\"No Contrails\"],y =[have_mask,no_mask] )\nplt.title(\"Imbalanced Training set\")","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:11:27.89416Z","iopub.execute_input":"2023-06-05T05:11:27.894579Z","iopub.status.idle":"2023-06-05T05:11:29.168257Z","shell.execute_reply.started":"2023-06-05T05:11:27.894543Z","shell.execute_reply":"2023-06-05T05:11:29.167218Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# predictions","metadata":{}},{"cell_type":"code","source":"m = nn.Sigmoid()\nsubmission = pd.read_csv('/kaggle/input/google-research-identify-contrails-reduce-global-warming/sample_submission.csv', index_col='record_id')\nmodel.eval()\nwith torch.no_grad():\n    for X, rec in test_loader:\n        X = X.to(device)\n#         pred = m(model.module(X)['out']).cpu().detach().numpy().copy()[0,0,:,:] \n        pred = m(model(X)).cpu().detach().numpy().copy()[0,0,:,:] \n        print(pred.shape)\n        mask = np.zeros((256, 256))\n        mask[pred<=0.6] = 0\n        mask[pred>0.6] = 1\n        \n        submission.loc[int(rec[0]), 'encoded_pixels'] = list_to_string(rle_encode(mask))\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:11:29.169634Z","iopub.execute_input":"2023-06-05T05:11:29.169986Z","iopub.status.idle":"2023-06-05T05:11:30.567595Z","shell.execute_reply.started":"2023-06-05T05:11:29.169953Z","shell.execute_reply":"2023-06-05T05:11:30.565536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-05T05:11:30.570255Z","iopub.execute_input":"2023-06-05T05:11:30.570662Z","iopub.status.idle":"2023-06-05T05:11:30.581461Z","shell.execute_reply.started":"2023-06-05T05:11:30.57063Z","shell.execute_reply":"2023-06-05T05:11:30.580235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# If you liked this notebook , please Upvote 😊","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}