{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!conda install ../input/pyvips-sub/*.tar.bz2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport PIL\nfrom PIL import Image\nimport torchvision\nfrom tqdm.auto import tqdm\nimport torch\nimport numpy as np\nfrom tqdm import tqdm, tqdm_notebook\nfrom pathlib import Path\nimport openslide\nfrom openslide import OpenSlide\n\nfrom torchvision import transforms\n#from multiprocessing.pool import ThreadPool\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn as nn\nimport gc\nimport pyvips\n#import cv2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\ntest_dir = './test_images'\nDATASET_FOLDER = '../input/mayo-clinic-strip-ai/'\nImage.MAX_IMAGE_PIXELS = None\nDEVICE = torch.device(\"cpu\")\nDATA_MODES = ['train', 'val', 'test']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prune_image_rows_cols(im, mask, thr=0.990):\n    # delete empty columns\n    for l in reversed(range(im.shape[1])):\n        if (np.sum(mask[:, l]) / float(mask.shape[0])) > thr:\n            im = np.delete(im, l, 1)\n    # delete empty rows\n    for l in reversed(range(im.shape[0])):\n        if (np.sum(mask[l, :]) / float(mask.shape[1])) > thr:\n            im = np.delete(im, l, 0)\n    return im\n\n\ndef mask_median(im, val=255):\n    masks = [None] * 3\n    for c in range(3):\n        masks[c] = im[..., c] >= np.median(im[:, :, c]) - 5\n    mask = np.logical_and(*masks)\n    im[mask, :] = val\n    return im, mask","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def vips2numpy(vi):\n    # map vips formats to np dtypes\n    format_to_dtype = {\n        'uchar': np.uint8,\n        'char': np.int8,\n        'ushort': np.uint16,\n        'short': np.int16,\n        'uint': np.uint32,\n        'int': np.int32,\n        'float': np.float32,\n        'double': np.float64,\n        'complex': np.complex64,\n        'dpcomplex': np.complex128,\n    }\n    return np.ndarray(buffer=vi.write_to_memory(),\n                      dtype=format_to_dtype[vi.format],\n                      shape=[vi.height, vi.width, vi.bands])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p test_images\n\ndef image_load_scale_norm(img_path, as_numpy=True):\n    #tmp_img = pyvips.Image.new_from_file(img_path)\n    # Resize the image\n    tmp_img = pyvips.Image.thumbnail(img_path, 4096)\n    tmp_img = vips2numpy(tmp_img) if as_numpy else tmp_img\n    im, mask = mask_median(tmp_img, val=255)\n    im = prune_image_rows_cols(im, mask, thr=0.990)\n    img = Image.fromarray(im)\n    scale = min(img.height / 1e3, img.width / 1e3)\n    if scale > 1:\n        img = img.resize((int(img.width / scale), int(img.height / scale)), Image.Resampling.LANCZOS)\n    gc.collect(); gc.collect();\n    return img","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for name in tqdm(df_test[\"image_id\"]):\n    img_path = os.path.join(DATASET_FOLDER, \"test\", f\"{name}.tif\")\n    img = image_load_scale_norm(img_path, as_numpy=True)\n    img.save(os.path.join(\"test_images\", f\"{name}.png\"))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RESCALE_WIDTH = 1200\nRESCALE_HEIGHT = 1500","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ClustDataset(Dataset):\n    \"\"\"\n    Датасет с обычными картинками, который паралельно подгружает их из папки\n    производит скалирование и превращение в торчевые тензоры\n    \"\"\"\n    def __init__(self, files, mode):\n        super().__init__()\n        # список файлов для загрузки\n        self.files = files\n        # режим работы\n        self.mode = mode\n\n        if self.mode not in DATA_MODES:\n            print(f\"{self.mode} is not correct; correct modes: {DATA_MODES}\")\n            raise NameError\n\n        self.len_ = len(self.files)\n                      \n    def __len__(self):\n        return self.len_\n      \n    def load_sample(self, file):\n        image = Image.open(os.path.join(test_dir, f'{file}'))\n        image.load()\n        return image\n  \n    def __getitem__(self, index):\n        # для преобразования изображений в тензоры PyTorch и нормализации входа\n        transform = transforms.Compose([transforms.ToTensor(), \n                                       transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                                                             std=[0.229, 0.224, 0.225]),])\n        x = self.load_sample(self.files[index])\n        x = self._prepare_sample(x)\n        x = np.array(x / 255, dtype='float32')\n        x = transform(x)\n        x = x.unsqueeze(0)\n        return x\n        \n    def _prepare_sample(self, image):\n        image = image.resize((RESCALE_WIDTH, RESCALE_HEIGHT), Image.Resampling.LANCZOS)\n        return np.array(image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SimpleCnn(nn.Module):\n  \n    def __init__(self, n_classes):\n        super().__init__()\n\n        self.conv1 = nn.Sequential(\n            nn.Conv2d(in_channels=3, out_channels=16, kernel_size=3),\n            nn.BatchNorm2d(16),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=2),\n            nn.Dropout(p=0.1)\n        )\n        self.conv2 = nn.Sequential(\n            nn.Conv2d(in_channels=16, out_channels=32, kernel_size=3, padding=1),\n            nn.BatchNorm2d(32),\n            nn.ReLU(),\n            nn.Conv2d(in_channels=32, out_channels=32, kernel_size=3, padding=1),\n            nn.BatchNorm2d(32),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=2),\n            nn.Dropout(p=0.2)\n        )\n        self.conv3 = nn.Sequential(\n            nn.Conv2d(in_channels=32, out_channels=64, kernel_size=3, padding=1),\n            nn.BatchNorm2d(64),\n            nn.ReLU(),\n            nn.Conv2d(in_channels=64, out_channels=64, kernel_size=3, padding=1),\n            nn.BatchNorm2d(64),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=2),\n            nn.Dropout(p=0.1)\n        )\n        self.conv4 = nn.Sequential(\n            nn.Conv2d(in_channels=64, out_channels=128, kernel_size=3, padding=1),\n            nn.BatchNorm2d(128),\n            nn.ReLU(),\n            nn.Conv2d(in_channels=128, out_channels=128, kernel_size=3, padding=1),\n            nn.BatchNorm2d(128),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=2),\n            nn.Dropout(p=0.1)\n        )\n        self.conv5 = nn.Sequential(\n            nn.Conv2d(in_channels=128, out_channels=256, kernel_size=3),\n            nn.BatchNorm2d(256),\n            nn.ReLU(),\n            nn.MaxPool2d(kernel_size=2),\n            nn.Dropout(p=0.2)\n        )\n        self.out = nn.Linear(256*45*36, n_classes)\n  \n  \n    def forward(self, x):\n        x = self.conv1(x)\n        x = self.conv2(x)\n        x = self.conv3(x)\n        x = self.conv4(x)\n        x = self.conv5(x)\n        #print(x.shape)\n        x = x.view(x.size(0), -1)\n        x = self.out(x)\n        return x","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_array_test = [filename for filename in os.listdir('./test_images')]\nfiles_array_test = sorted(files_array_test)\nimage_data = {'image_id':files_array_test}\ntest_data = pd.DataFrame(image_data)\ntest_data['image_id'] = test_data['image_id'].map(lambda x: x.rstrip('.png'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = SimpleCnn(n_classes=4)\nmodel.load_state_dict(torch.load('../input/model-6/model (5).pth', map_location='cpu'))\ntest_dataset = ClustDataset(files_array_test, mode='test')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(model, test_loader):\n    with torch.no_grad():\n        logits = []\n    \n        for inputs in test_loader:\n            inputs = inputs.to(DEVICE)\n            model.eval()\n            outputs = model(inputs).cpu()\n            logits.append(outputs)\n            \n    probs = nn.functional.softmax(torch.cat(logits), dim=-1).numpy()\n    return probs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ans = predict(model, test_dataset)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for file in os.listdir(test_dir):\n    file_path = os.path.join(test_dir, file)\n    os.remove(file_path)\n\nos.rmdir(test_dir)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ans = ans[:, :2]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['CE'] = ans[:, 0].round(6)\ndf_test['LAA'] = ans[:, 1].round(6)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub =  df_test[['patient_id', 'CE', 'LAA']]\nsub = sub.groupby('patient_id').mean()\nsub = sub[['CE', 'LAA']].reset_index()\nsub.to_csv('submission.csv', index=False)\n#display(sub)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}