{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"try :\n    import timm\nexcept :\n    ! pip install timm==0.6.12","metadata":{"execution":{"iopub.status.busy":"2023-01-06T15:11:32.468563Z","iopub.execute_input":"2023-01-06T15:11:32.469173Z","iopub.status.idle":"2023-01-06T15:11:44.573943Z","shell.execute_reply.started":"2023-01-06T15:11:32.469051Z","shell.execute_reply":"2023-01-06T15:11:44.572759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import classification_report\nimport timm.optim.optim_factory as optim_factory\nfrom timm.data import create_transform\nfrom torch.utils.data import Dataset\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom sklearn import metrics\nimport glob\n%matplotlib inline\nimport torch.nn as nn\nfrom PIL import Image\nimport pandas as pd\nimport numpy as np\nimport torch\nimport timm\nimport cv2\nimport torch.nn.functional as F","metadata":{"execution":{"iopub.status.busy":"2023-01-06T15:11:44.576485Z","iopub.execute_input":"2023-01-06T15:11:44.577211Z","iopub.status.idle":"2023-01-06T15:11:47.703396Z","shell.execute_reply.started":"2023-01-06T15:11:44.577167Z","shell.execute_reply":"2023-01-06T15:11:47.702314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_EPOCHS = 5\nNUM_SPLITS = 4\n\nRESIZE_TO = (512, 512)\n\nDATA_PATH = '/kaggle/input/rsna-breast-cancer-detection/'\nTRAIN_IMAGE_DIR = '/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_vl_asp_1024/train_images_processed_cv2_vl_asp_1024/'\nMODEL_PATH = '/kaggle/working/'","metadata":{"execution":{"iopub.status.busy":"2023-01-06T15:11:47.705075Z","iopub.execute_input":"2023-01-06T15:11:47.705667Z","iopub.status.idle":"2023-01-06T15:11:47.712307Z","shell.execute_reply.started":"2023-01-06T15:11:47.705622Z","shell.execute_reply":"2023-01-06T15:11:47.710787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv(f'{DATA_PATH}/train.csv')\ntrain_csv['image_path'] = TRAIN_IMAGE_DIR + train_csv[\"patient_id\"].astype(str) + \"/\" + train_csv[\"image_id\"].astype(str) + \".png\"\ntrain_csv = train_csv.sample(frac=0.1, random_state =0)\ntrain_csv.reset_index(drop = True, inplace = True)\nprint(train_csv.shape)\nskf = StratifiedKFold(NUM_SPLITS, shuffle=True, random_state=7)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-06T15:11:47.714984Z","iopub.execute_input":"2023-01-06T15:11:47.715603Z","iopub.status.idle":"2023-01-06T15:11:47.946482Z","shell.execute_reply.started":"2023-01-06T15:11:47.715566Z","shell.execute_reply":"2023-01-06T15:11:47.945476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(train_csv['cancer']==1).sum()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T15:11:47.947793Z","iopub.execute_input":"2023-01-06T15:11:47.948611Z","iopub.status.idle":"2023-01-06T15:11:47.959691Z","shell.execute_reply.started":"2023-01-06T15:11:47.948582Z","shell.execute_reply":"2023-01-06T15:11:47.958225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-01-06T15:11:47.961455Z","iopub.execute_input":"2023-01-06T15:11:47.961794Z","iopub.status.idle":"2023-01-06T15:11:47.989331Z","shell.execute_reply.started":"2023-01-06T15:11:47.961761Z","shell.execute_reply":"2023-01-06T15:11:47.981106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pfbeta_torch(preds, labels, beta=1):\n    preds = preds.clip(0, 1)\n\n    y_true_count = labels.sum()\n    ctp = preds[labels == 1].sum()\n    cfp = preds[labels == 0].sum()\n\n    beta_squared = beta * beta\n\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n\n    if c_precision > 0 and c_recall > 0:\n        return ((1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)).item()\n    else:\n        return 0.0\n\ndef pfbeta_thresh(preds, labels):\n    optimized_preds = optimize_preds(preds, labels)\n    return pfbeta_torch(optimized_preds, labels)\n\n\ndef optimize_preds(preds, labels, return_thresh=False, print_results=False):\n    preds = preds.clone()\n\n    without_thresh = pfbeta_torch(preds, labels)\n\n    threshs = np.linspace(0, 1, 101)\n    f1s = [pfbeta_torch((preds > thr).float(), labels) for thr in threshs]\n    idx = np.argmax(f1s)\n    thresh, best_pfbeta = threshs[idx], f1s[idx]\n\n    preds = (preds > thresh).float()\n\n    if print_results:\n        print(f\"without optimization: {without_thresh:.3f}\")\n        pfbeta = pfbeta_torch(preds, labels)\n        print(f\"with optimization: {pfbeta:.3f}\")\n        print(f\"best_thresh: {thresh}\")\n\n    if return_thresh:\n        return thresh\n\n    return preds","metadata":{"execution":{"iopub.status.busy":"2023-01-06T15:11:47.992904Z","iopub.execute_input":"2023-01-06T15:11:47.993625Z","iopub.status.idle":"2023-01-06T15:11:48.012527Z","shell.execute_reply.started":"2023-01-06T15:11:47.993591Z","shell.execute_reply":"2023-01-06T15:11:48.010055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tf_efficientnetv2_s():\n    model = timm.create_model(\n        'tf_efficientnetv2_s', pretrained=True, in_chans=3, num_classes=1)\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-06T15:11:48.01945Z","iopub.execute_input":"2023-01-06T15:11:48.022238Z","iopub.status.idle":"2023-01-06T15:11:48.031505Z","shell.execute_reply.started":"2023-01-06T15:11:48.022179Z","shell.execute_reply":"2023-01-06T15:11:48.030423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BalanceSampler(torch.utils.data.sampler.Sampler):\n    def __init__(self, dataset, ratio=4):\n        self.r = ratio-1\n        self.dataset = dataset\n        self.pos_index = np.where(dataset.label>0)[0]\n        self.neg_index = np.where(dataset.label==0)[0]\n\n        self.length = self.r*int(np.floor(len(self.neg_index)/self.r))\n\n    def __iter__(self):\n        pos_index = self.pos_index.copy()\n        neg_index = self.neg_index.copy()\n        np.random.shuffle(pos_index)\n        np.random.shuffle(neg_index)\n\n        neg_index = neg_index[:self.length].reshape(-1,self.r)\n        pos_index = np.random.choice(pos_index, self.length//self.r).reshape(-1,1)\n\n        index = np.concatenate([pos_index,neg_index],-1).reshape(-1)\n        return iter(index)\n\n    def __len__(self):\n        return self.length","metadata":{"execution":{"iopub.status.busy":"2023-01-06T06:56:25.144961Z","iopub.execute_input":"2023-01-06T06:56:25.14556Z","iopub.status.idle":"2023-01-06T06:56:25.158271Z","shell.execute_reply.started":"2023-01-06T06:56:25.14545Z","shell.execute_reply":"2023-01-06T06:56:25.157005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ************************* AVEC ROI *******************************\n    \ndef img2roi(img):\n    # Binarize the image\n    bin_img = cv2.threshold(img, 20, 255, cv2.THRESH_BINARY)[1]\n\n    # Make contours around the binarized image, keep only the largest contour\n    contours, _ = cv2.findContours(bin_img, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n    contour = max(contours, key=cv2.contourArea)\n\n    # Find ROI from largest contour\n    ys = contour.squeeze()[:, 0]\n    xs = contour.squeeze()[:, 1]\n    roi =  img[np.min(xs):np.max(xs), np.min(ys):np.max(ys)]\n    \n    return roi\n\nclass GetLoader(Dataset):\n    def __init__(self, transform, df, is_train=True):\n        self.data = df['image_path'].values\n        self.targets = df['cancer'].values\n        self.trans = transform\n\n    def __getitem__(self, index):\n        #data = Image.open(self.data[index]).convert('RGB')\n        data = cv2.imread(self.data[index])\n        data = cv2.cvtColor(data, cv2.COLOR_BGR2GRAY)\n        data = img2roi(data)\n        data = cv2.cvtColor(data, cv2.COLOR_BGR2RGB)\n        data = cv2.resize(data, (512,512))\n        data = Image.fromarray(data)\n        \n        data = torch.as_tensor(self.trans(data),dtype=torch.float32).cuda()\n        targets = torch.as_tensor(self.targets[index],dtype=torch.float32).cuda()\n        return data, targets\n\n    def __len__(self):\n        return len(self.data)\n\n\ndef build_transform(is_train):\n    \n    if is_train:\n        transform = create_transform(\n            input_size=(512,512),\n            is_training=True,\n            scale=(0.75, 1.33),\n            ratio=(0.08, 1.0),\n            hflip=0.5,\n            vflip=0.5,\n            color_jitter=0.4,\n            interpolation=\"random\",\n        )\n    else:\n        transform = create_transform(\n            input_size=(512,512),\n            is_training=False,\n            interpolation=\"bilinear\",\n        )\n    return transform\n\ndef build_dataset(df, is_train=True):\n    transform = build_transform(is_train)\n    dataset = GetLoader(transform,df, is_train=True)\n    return dataset\n\nclass BalanceSampler(torch.utils.data.sampler.Sampler):\n    def __init__(self, dataset, ratio=4):\n        self.r = ratio-1\n        self.dataset = dataset\n        self.pos_index = np.where(dataset.targets>0)[0]\n        self.neg_index = np.where(dataset.targets==0)[0]\n\n        self.length = self.r*int(np.floor(len(self.neg_index)/self.r)) # = nombre de row négatifs \n\n    def __iter__(self):\n        pos_index = self.pos_index.copy()\n        neg_index = self.neg_index.copy()\n        np.random.shuffle(pos_index)\n        np.random.shuffle(neg_index)\n\n        neg_index = neg_index[:self.length].reshape(-1,self.r)\n        pos_index = np.random.choice(pos_index, self.length//self.r).reshape(-1,1)\n\n        index = np.concatenate([pos_index,neg_index],-1).reshape(-1)\n        return iter(index)\n\n    def __len__(self):\n        return self.length","metadata":{"execution":{"iopub.status.busy":"2023-01-06T09:44:43.320288Z","iopub.execute_input":"2023-01-06T09:44:43.320644Z","iopub.status.idle":"2023-01-06T09:44:43.340744Z","shell.execute_reply.started":"2023-01-06T09:44:43.320612Z","shell.execute_reply":"2023-01-06T09:44:43.339525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = np.zeros((train_csv.shape[0],1))\n\nskf = StratifiedKFold(5, shuffle=True, random_state=7)\n\nfor fold, (train_index, test_index) in enumerate((skf.split(train_csv,train_csv['cancer']))):\n\n        print(f'---------- training fold {fold} -------------')\n        X_train, X_test = train_csv.loc[train_index], train_csv.loc[test_index]\n        y_train, y_test = train_csv.loc[train_index]['cancer'], train_csv.loc[test_index]['cancer']\n        print('nombre de cas de cancer dans train :',(X_train['cancer']==1).sum())\n        print('nombre de cas de cancer dans test :',(X_test['cancer']==1).sum())\n        \n        if fold > 0 :\n            break\n            \n        train_dataset = build_dataset(df = X_train,is_train=True)\n        sampler = BalanceSampler(train_dataset, ratio = 4)\n        train_loader = torch.utils.data.DataLoader(\n                                                    train_dataset,\n                                                    batch_size=4,\n                                                    sampler=sampler,\n                                                    drop_last=True,\n                                                    )\n        test_dataset = build_dataset(df = X_test,is_train=False)\n        test_loader = torch.utils.data.DataLoader(\n                                                    test_dataset,\n                                                    batch_size=4,\n                                                    drop_last=False\n                                                    )\n    \n  \n        model = tf_efficientnetv2_s().cuda()\n        model_without_ddp = model\n        param_groups = optim_factory.param_groups_weight_decay(model_without_ddp,0.05)\n        optimizer = torch.optim.AdamW(param_groups, lr=2.25e-4, betas=(0.9, 0.999))\n        loss = nn.BCEWithLogitsLoss(pos_weight=torch.tensor([3])).cuda()\n        \n        model.train()\n        early_stopping = 0  \n        best_metric = 0\n        best_epoch = 0\n        \n        for epoch in range(NUM_EPOCHS):\n            model.train()\n            train_loss = []\n            stream = tqdm(train_loader) \n            for i, (img, label) in enumerate(stream, start=0): \n                label = label.unsqueeze(1)\n                l = loss(model(img),label.float())\n                train_loss.append(l.item())\n                optimizer.zero_grad()\n                l.backward()\n                optimizer.step()\n                \n                loss_mean = np.mean(train_loss)\n                stream.set_description(\n                    \"Training   | Epoch: {epoch} | loss : {loss:.5f} | \".format(epoch=epoch, loss = loss_mean )\n                )\n            # ----------------- evaluation de l'epoch -----------------------\n            \n            preds1 = []\n            model.eval()\n            \n            for i, (img, label) in enumerate(tqdm(test_loader)):\n                output = torch.sigmoid(model(img)).detach().cpu()   \n                \n                if i ==0 :\n                    preds1=output.numpy()\n                else :\n                    preds1 = np.vstack([preds1,output.numpy()])\n            \n            score_epoch = np.round(pfbeta_torch(preds1,y_test),3)         \n            print(f'Epoch {epoch} pfbeta {score_epoch} ')\n            \n            if score_epoch > best_metric:\n                best_metric = score_epoch\n                best_epoch = epoch\n                best_model = f'tf_efficient_{fold}_ckpt_pytorch'\n                torch.save(model.state_dict(),best_model)\n                print('model saved')\n                oof[test_index] = preds1\n                          \n            else :\n                early_stopping +=1\n                if early_stopping == 2 :\n                    print('early stopping stop')\n                    break\n                else :\n                    print('early stopping =',early_stopping)\n                    \n        \n        print(f'===> Best epoch = {best_epoch} OOF pfbeta = {best_metric}')\n        #threshold = optimize_preds(oof[test_index], y_test, return_thresh=True, print_results=True)\n\nscore_oof = np.round(pfbeta_torch(oof,labels_train),3)\nprint(f'\\n******** OOF pfbeta {score_oof} ***************')","metadata":{"execution":{"iopub.status.busy":"2023-01-06T09:44:44.040889Z","iopub.execute_input":"2023-01-06T09:44:44.041338Z","iopub.status.idle":"2023-01-06T10:15:03.35422Z","shell.execute_reply.started":"2023-01-06T09:44:44.041303Z","shell.execute_reply":"2023-01-06T10:15:03.352806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ajout de tabular data","metadata":{}},{"cell_type":"code","source":"# Keep only columns in test + target variable\ntrain = train[[\"patient_id\", \"image_id\", \"laterality\", \"view\", \"age\", \"implant\", \"path\", \"cancer\"]]\n\n# Encode categorical variables\nle_laterality = LabelEncoder()\nle_view = LabelEncoder()\n\ntrain['laterality'] = le_laterality.fit_transform(train['laterality'])\ntrain['view'] = le_view.fit_transform(train['view'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\n\ntrain_csv['age'].fillna(60, inplace = True)\n\nst = preprocessing.StandardScaler()\ntrain_csv['age']= st.fit_transform(train_csv[['age']])\n\nst1 = preprocessing.StandardScaler()\ntrain_csv['site_id']= st1.fit_transform(train_csv[['site_id']])\n\nst2 = preprocessing.StandardScaler()\ntrain_csv['machine_id']= st2.fit_transform(train_csv[['machine_id']])\n\n# Encode categorical variables\nle_laterality = preprocessing.LabelEncoder()\nle_view = preprocessing.LabelEncoder()\n\ntrain_csv['laterality'] = le_laterality.fit_transform(train_csv['laterality'])\ntrain_csv['view'] = le_view.fit_transform(train_csv['view'])\n\ntrain_csv","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ************************* AVEC ROI *******************************\n    \ndef img2roi(img):\n    # Binarize the image\n    bin_img = cv2.threshold(img, 20, 255, cv2.THRESH_BINARY)[1]\n\n    # Make contours around the binarized image, keep only the largest contour\n    contours, _ = cv2.findContours(bin_img, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n    contour = max(contours, key=cv2.contourArea)\n\n    # Find ROI from largest contour\n    ys = contour.squeeze()[:, 0]\n    xs = contour.squeeze()[:, 1]\n    roi =  img[np.min(xs):np.max(xs), np.min(ys):np.max(ys)]\n    \n    return roi\n\nclass GetLoader(Dataset):\n    def __init__(self, transform, df, is_train=True):\n        self.data = df['image_path'].values\n        self.targets = df['cancer'].values\n        self.tabular_data = np.array(df[['laterality', 'view', 'age', 'implant']])\n        self.trans = transform\n\n    def __getitem__(self, index):\n        #data = Image.open(self.data[index]).convert('RGB')\n        data = cv2.imread(self.data[index])\n        data = cv2.cvtColor(data, cv2.COLOR_BGR2GRAY)\n        data = img2roi(data)\n        data = cv2.cvtColor(data, cv2.COLOR_BGR2RGB)\n        data = cv2.resize(data, (512,512))\n        data = Image.fromarray(data)\n        \n        data = torch.as_tensor(self.trans(data),dtype=torch.float32).cuda()\n        targets = torch.as_tensor(self.targets[index],dtype=torch.float32).cuda()\n        z = torch.from_numpy(self.tabular_data[index])\n        return data, targets, z\n\n    def __len__(self):\n        return len(self.data)\n\n\ndef build_transform(is_train):\n    \n    if is_train:\n        transform = create_transform(\n            input_size=(512,512),\n            is_training=True,\n            scale=(0.75, 1.33),\n            ratio=(0.08, 1.0),\n            hflip=0.5,\n            vflip=0.5,\n            color_jitter=0.4,\n            interpolation=\"random\",\n        )\n    else:\n        transform = create_transform(\n            input_size=(512,512),\n            is_training=False,\n            interpolation=\"bilinear\",\n        )\n    return transform\n\ndef build_dataset(df, is_train=True):\n    transform = build_transform(is_train)\n    dataset = GetLoader(transform,df, is_train=True)\n    return dataset\n\nclass BalanceSampler(torch.utils.data.sampler.Sampler):\n    def __init__(self, dataset, ratio=4):\n        self.r = ratio-1\n        self.dataset = dataset\n        self.pos_index = np.where(dataset.targets>0)[0]\n        self.neg_index = np.where(dataset.targets==0)[0]\n\n        self.length = self.r*int(np.floor(len(self.neg_index)/self.r)) # = nombre de row négatifs \n\n    def __iter__(self):\n        pos_index = self.pos_index.copy()\n        neg_index = self.neg_index.copy()\n        np.random.shuffle(pos_index)\n        np.random.shuffle(neg_index)\n\n        neg_index = neg_index[:self.length].reshape(-1,self.r)\n        pos_index = np.random.choice(pos_index, self.length//self.r).reshape(-1,1)\n\n        index = np.concatenate([pos_index,neg_index],-1).reshape(-1)\n        return iter(index)\n\n    def __len__(self):\n        return self.length","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tf_efficientnetv2_s():\n    model = timm.create_model(\n        'tf_efficientnetv2_s', pretrained=True, in_chans=3, num_classes=1)\n    return model\n\n\nclass MyModel(nn.Module):\n    \n    def __init__(self, timm_model):\n        super().__init__()\n        self.timm_model = timm_model\n        self.fc1 = nn.Linear(1, 1)\n        self.fc2 = nn.Linear(1, 1)\n        self.fc3 = nn.Linear(1+4, 16) #x.size(1) =4, in_features = nombre de features en entrée (colonnes)\n        self.fc4 = nn.Linear(16, 1)\n    \n    def forward(self, x,z):\n        x = self.timm_model(x)\n        x = F.relu(self.fc1(x))\n        #x = self.fc2(x)\n        x = torch.cat((x,z),1)\n        x = self.fc3(x) # x.size(1) in_features = nombre de features en entrée (colonnes)\n        x = self.fc4(x)\n        output = x #torch.sigmoid(x)\n        \n        return output\n    \nmodel1 = tf_efficientnetv2_s()\nmodel1 = model1.to('cuda')\nmodel = MyModel(model1).to('cuda')\n\n\nimage_test = torch.rand(*(3,3,256,256))\nmodel((image_test).to('cuda'),torch.tensor(np.ones((3,4))).to('cuda').float())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = np.zeros((train_csv.shape[0],1))\n\nskf = StratifiedKFold(5, shuffle=True, random_state=7)\n\nfor fold, (train_index, test_index) in enumerate((skf.split(train_csv,train_csv['cancer']))):\n\n        print(f'---------- training fold {fold} -------------')\n        X_train, X_test = train_csv.loc[train_index], train_csv.loc[test_index]\n        y_train, y_test = train_csv.loc[train_index]['cancer'], train_csv.loc[test_index]['cancer']\n        print('nombre de cas de cancer dans train :',(X_train['cancer']==1).sum())\n        print('nombre de cas de cancer dans test :',(X_test['cancer']==1).sum())\n        \n        if fold > 0 :\n            break\n            \n        train_dataset = build_dataset(df = X_train,is_train=True)\n        sampler = BalanceSampler(train_dataset, ratio = 4)\n        train_loader = torch.utils.data.DataLoader(\n                                                    train_dataset,\n                                                    batch_size=4,\n                                                    sampler=sampler,\n                                                    drop_last=True,\n                                                    )\n        test_dataset = build_dataset(df = X_test,is_train=False)\n        test_loader = torch.utils.data.DataLoader(\n                                                    test_dataset,\n                                                    batch_size=4,\n                                                    drop_last=False\n                                                    )\n    \n  \n        model1 = tf_efficientnetv2_s()\n        model1 = model1.to('cuda')\n        model = MyModel(model1).to('cuda')\n        \n        model_without_ddp = model\n        param_groups = optim_factory.param_groups_weight_decay(model_without_ddp,0.05)\n        optimizer = torch.optim.AdamW(param_groups, lr=2.25e-4, betas=(0.9, 0.999))\n        loss = nn.BCEWithLogitsLoss(pos_weight=torch.tensor([3])).cuda()\n        \n        model.train()\n        early_stopping = 0  \n        best_metric = 0\n        best_epoch = 0\n        \n        for epoch in range(NUM_EPOCHS):\n            model.train()\n            train_loss = []\n            stream = tqdm(train_loader) \n            for i, (img, label,z) in enumerate(stream, start=0): \n                label = label.unsqueeze(1)\n                z = z.to('cuda', non_blocking=True).float()\n                l = loss(model(img,z),label.float())\n                train_loss.append(l.item())\n                optimizer.zero_grad()\n                l.backward()\n                optimizer.step()\n                \n                loss_mean = np.mean(train_loss)\n                stream.set_description(\n                    \"Training   | Epoch: {epoch} | loss : {loss:.5f} | \".format(epoch=epoch, loss = loss_mean )\n                )\n            # ----------------- evaluation de l'epoch -----------------------\n            \n            preds1 = []\n            model.eval()\n            \n            for i, (img, label,z) in enumerate(tqdm(test_loader)):\n                z = z.to('cuda', non_blocking=True).float()\n                output = torch.sigmoid(model(img,z)).detach().cpu()   \n                \n                if i ==0 :\n                    preds1=output.numpy()\n                else :\n                    preds1 = np.vstack([preds1,output.numpy()])\n            \n            score_epoch = np.round(pfbeta_torch(preds1,y_test),3)         \n            print(f'Epoch {epoch} pfbeta {score_epoch} ')\n            \n            if score_epoch > best_metric:\n                best_metric = score_epoch\n                best_epoch = epoch\n                best_model = f'tf_efficient_{fold}_ckpt_pytorch'\n                torch.save(model.state_dict(),best_model)\n                print('model saved')\n                oof[test_index] = preds1\n                          \n            else :\n                early_stopping +=1\n                if early_stopping == 2 :\n                    print('early stopping stop')\n                    break\n                else :\n                    print('early stopping =',early_stopping)\n                    \n        \n        print(f'===> Best epoch = {best_epoch} OOF pfbeta = {best_metric}')\n        #threshold = optimize_preds(oof[test_index], y_test, return_thresh=True, print_results=True)\n\nscore_oof = np.round(pfbeta_torch(oof,labels_train),3)\nprint(f'\\n******** OOF pfbeta {score_oof} ***************')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ajout de 2 models","metadata":{}},{"cell_type":"code","source":"class Multimodels(nn.Module):\n    \n    def __init__(self, timm_model1,\n                 timm_model2,\n                 #timm_model3\n                ):\n        super().__init__()\n        self.timm_model1 = timm_model1\n        self.timm_model2 = timm_model2\n        #self.timm_model3 = timm_model3\n        self.fc1 = nn.Linear(1, 1)\n        self.fc2 = nn.Linear(1, 1)\n        #self.fc3 = nn.Linear(2+2, 16) #x.size(1) =4, in_features = nombre de features en entrée (colonnes)\n        self.fc3 = nn.Linear(2+0, 1)\n        #self.fc4 = nn.Linear(8, 1)\n    \n    def forward(self,\n                x1,\n                x2,\n                #x3,\n                #z\n               ):\n        x1 = self.timm_model1(x1)\n        x1 = F.relu(self.fc1(x1))\n        \n        x2 = self.timm_model2(x2)\n        x2 = F.relu(self.fc1(x2))\n        \n        #x3 = self.timm_model3(x3)\n        #x3 = F.relu(self.fc1(x3))\n        \n        x = torch.cat((x1, \n                       x2,\n                       #x3\n                     ),1)\n        x = torch.max(x,1,True)[0]\n   \n        #x = torch.cat((x1,\n         #              x2, \n                       #x3,\n                       #z\n               #       ),1)\n        #x = torch.cat((x[0], z),1)\n        #x = torch.cat((x1, z),1)\n        #x = self.fc3(x) # x.size(1) in_features = nombre de features en entrée (colonnes)\n        #x = self.fc4(x)\n        output = torch.sigmoid(x)\n       \n        \n        return output\n    \ndef create_multimodels() :\n    \n    model1 = timm.create_model('efficientnetv2_s', pretrained=True, in_chans=3, num_classes=1)\n    model1 = model1.to('cuda')\n\n\n    model2 = timm.create_model('resnet50', pretrained=True, in_chans=3, num_classes=1)\n    model2 = model2.to('cuda')\n\n    #model3 = timm.create_model('tf_efficientnetv2_b3', pretrained=False, in_chans=3, num_classes=1)\n    #model3 = model2.to(params['device'])\n\n    model = Multimodels(\n                        model1,\n                        model2,\n                        #model3\n                       ).to('cuda')\n    \n    return model\n\nimage_test = torch.rand(*(3,3,256,256))\n\nmodel4 = create_multimodels() \n\nmodel4((image_test).to('cuda'),\n       (image_test).to('cuda'),\n       #(image_test).to(params['device']),\n       #torch.tensor(np.ones((3,2))).to(params['device']).float()\n      )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = torch.randn(3, 2)\ndisplay(pd.DataFrame(a.numpy()))\nprint(a.numpy().shape)\n\nb = torch.max(a, 1)[0]\nc = torch.sigmoid(b)\n\nprint(b)\nprint(c)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ************************* AVEC ROI *******************************\n    \ndef img2roi(img):\n    # Binarize the image\n    bin_img = cv2.threshold(img, 20, 255, cv2.THRESH_BINARY)[1]\n\n    # Make contours around the binarized image, keep only the largest contour\n    contours, _ = cv2.findContours(bin_img, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n    contour = max(contours, key=cv2.contourArea)\n\n    # Find ROI from largest contour\n    ys = contour.squeeze()[:, 0]\n    xs = contour.squeeze()[:, 1]\n    roi =  img[np.min(xs):np.max(xs), np.min(ys):np.max(ys)]\n    \n    return roi\n\nclass GetLoader(Dataset):\n    def __init__(self, transform, df, is_train=True):\n        self.data = df['image_path'].values\n        self.targets = df['cancer'].values\n        self.trans = transform\n\n    def __getitem__(self, index):\n        #data = Image.open(self.data[index]).convert('RGB')\n        data = cv2.imread(self.data[index])\n        data = cv2.cvtColor(data, cv2.COLOR_BGR2GRAY)\n        data = img2roi(data)\n        data = cv2.cvtColor(data, cv2.COLOR_BGR2RGB)\n        data = cv2.resize(data, (512,512))\n        data = Image.fromarray(data)\n        \n        data = torch.as_tensor(self.trans(data),dtype=torch.float32).cuda()\n        targets = torch.as_tensor(self.targets[index],dtype=torch.float32).cuda()\n        return data, targets\n\n    def __len__(self):\n        return len(self.data)\n\n\ndef build_transform(is_train):\n    \n    if is_train:\n        transform = create_transform(\n            input_size=(512,512),\n            is_training=True,\n            scale=(0.75, 1.33),\n            ratio=(0.08, 1.0),\n            hflip=0.5,\n            vflip=0.5,\n            color_jitter=0.4,\n            interpolation=\"random\",\n        )\n    else:\n        transform = create_transform(\n            input_size=(512,512),\n            is_training=False,\n            interpolation=\"bilinear\",\n        )\n    return transform\n\ndef build_dataset(df, is_train=True):\n    transform = build_transform(is_train)\n    dataset = GetLoader(transform,df, is_train=True)\n    return dataset\n\nclass BalanceSampler(torch.utils.data.sampler.Sampler):\n    def __init__(self, dataset, ratio=4):\n        self.r = ratio-1\n        self.dataset = dataset\n        self.pos_index = np.where(dataset.targets>0)[0]\n        self.neg_index = np.where(dataset.targets==0)[0]\n\n        self.length = self.r*int(np.floor(len(self.neg_index)/self.r)) # = nombre de row négatifs \n\n    def __iter__(self):\n        pos_index = self.pos_index.copy()\n        neg_index = self.neg_index.copy()\n        np.random.shuffle(pos_index)\n        np.random.shuffle(neg_index)\n\n        neg_index = neg_index[:self.length].reshape(-1,self.r)\n        pos_index = np.random.choice(pos_index, self.length//self.r).reshape(-1,1)\n\n        index = np.concatenate([pos_index,neg_index],-1).reshape(-1)\n        return iter(index)\n\n    def __len__(self):\n        return self.length","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = np.zeros((train_csv.shape[0],1))\n\nskf = StratifiedKFold(5, shuffle=True, random_state=7)\n\nfor fold, (train_index, test_index) in enumerate((skf.split(train_csv,train_csv['cancer']))):\n\n        print(f'---------- training fold {fold} -------------')\n        X_train, X_test = train_csv.loc[train_index], train_csv.loc[test_index]\n        y_train, y_test = train_csv.loc[train_index]['cancer'], train_csv.loc[test_index]['cancer']\n        print('nombre de cas de cancer dans train :',(X_train['cancer']==1).sum())\n        print('nombre de cas de cancer dans test :',(X_test['cancer']==1).sum())\n        \n        if fold > 0 :\n            break\n            \n        train_dataset = build_dataset(df = X_train,is_train=True)\n        sampler = BalanceSampler(train_dataset, ratio = 4)\n        train_loader = torch.utils.data.DataLoader(\n                                                    train_dataset,\n                                                    batch_size=4,\n                                                    sampler=sampler,\n                                                    drop_last=True,\n                                                    )\n        test_dataset = build_dataset(df = X_test,is_train=False)\n        test_loader = torch.utils.data.DataLoader(\n                                                    test_dataset,\n                                                    batch_size=4,\n                                                    drop_last=False\n                                                    )\n    \n  \n       \n        model = create_multimodels().to('cuda')\n        \n        model_without_ddp = model\n        param_groups = optim_factory.param_groups_weight_decay(model_without_ddp,0.05)\n        optimizer = torch.optim.AdamW(param_groups, lr=2.25e-4, betas=(0.9, 0.999))\n        loss = nn.BCEWithLogitsLoss(pos_weight=torch.tensor([3])).cuda()\n        \n        model.train()\n        early_stopping = 0  \n        best_metric = 0\n        best_epoch = 0\n        \n        for epoch in range(NUM_EPOCHS):\n            model.train()\n            train_loss = []\n            stream = tqdm(train_loader) \n            for i, (img, label) in enumerate(stream, start=0): \n                label = label.unsqueeze(1)\n                l = loss(model(img,img),label.float())\n                train_loss.append(l.item())\n                optimizer.zero_grad()\n                l.backward()\n                optimizer.step()\n                \n                loss_mean = np.mean(train_loss)\n                stream.set_description(\n                    \"Training   | Epoch: {epoch} | loss : {loss:.5f} | \".format(epoch=epoch, loss = loss_mean )\n                )\n            # ----------------- evaluation de l'epoch -----------------------\n            \n            preds1 = []\n            model.eval()\n            \n            for i, (img, label) in enumerate(tqdm(test_loader)):\n                #output = torch.sigmoid(model(img,img)).detach().cpu()   \n                output = (model(img,img)).detach().cpu()   \n                if i ==0 :\n                    preds1=output.numpy()\n                else :\n                    preds1 = np.vstack([preds1,output.numpy()])\n            \n            score_epoch = np.round(pfbeta_torch(preds1,y_test),3)         \n            print(f'Epoch {epoch} pfbeta {score_epoch} ')\n            \n            if score_epoch > best_metric:\n                best_metric = score_epoch\n                best_epoch = epoch\n                best_model = f'tf_efficient_{fold}_ckpt_pytorch'\n                torch.save(model.state_dict(),best_model)\n                print('model saved')\n                oof[test_index] = preds1\n                          \n            else :\n                early_stopping +=1\n                if early_stopping == 2 :\n                    print('early stopping stop')\n                    break\n                else :\n                    print('early stopping =',early_stopping)\n                    \n        \n        print(f'===> Best epoch = {best_epoch} OOF pfbeta = {best_metric}')\n        #threshold = optimize_preds(oof[test_index], y_test, return_thresh=True, print_results=True)\n\nscore_oof = np.round(pfbeta_torch(oof,labels_train),3)\nprint(f'\\n******** OOF pfbeta {score_oof} ***************')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Amelioration du ROI selon MLO ou pas","metadata":{}},{"cell_type":"code","source":"# ************************* AVEC ROI *******************************\n\"\"\"\ndef img2roi(img):\n    # Binarize the image\n    bin_img = cv2.threshold(img, 20, 255, cv2.THRESH_BINARY)[1]\n\n    # Make contours around the binarized image, keep only the largest contour\n    contours, _ = cv2.findContours(bin_img, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n    contour = max(contours, key=cv2.contourArea)\n\n    # Find ROI from largest contour\n    ys = contour.squeeze()[:, 0]\n    xs = contour.squeeze()[:, 1]\n    roi =  img[np.min(xs):np.max(xs), np.min(ys):np.max(ys)]\n    \n    return roi\n\"\"\"\n\ndef slice_image(img):\n    _, width = img.shape\n    new_width = int(width / 2)\n    left_img = img[:, 0:new_width]\n    right_img = img[:, new_width:]\n    return left_img, right_img\n\ndef mask_pixels(img):\n    img[img > 1] = 1\n    return img\n\ndef count_pixels(img):\n    return np.sum(img)\n\ndef normalize_horiz_orientation(img):\n    left_img, right_img = slice_image(mask_pixels(img.copy()))\n    if count_pixels(right_img) > count_pixels(left_img):\n        return cv2.flip(img, 1)\n    return img\n\nclass CFG:\n    resize_dim = 1024\n    aspect_ratio = True\n    #img_size = [1024, 512]\n\ndef img2roi(img):\n    # Binarize the image\n    bin_img = cv2.threshold(img, 20, 255, cv2.THRESH_BINARY)[1]\n\n    # Make contours around the binarized image, keep only the largest contour\n    contours, _ = cv2.findContours(bin_img, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n    contour = max(contours, key=cv2.contourArea)\n\n    # Find ROI from largest contour\n    ys = contour.squeeze()[:, 0]\n    xs = contour.squeeze()[:, 1]\n    roi =  img[np.min(xs):np.max(xs), np.min(ys):np.max(ys)]\n    \n    return roi\n\ndef resize_and_save(data,VIEW):\n    #img = read_xray(file_path)\n    #img = (data * 255).astype(np.uint8)\n    img = normalize_horiz_orientation(data)\n    img = img2roi(img)\n    img = cv2.resize(img, (512,512), cv2.INTER_NEAREST)\n    \n    ratio = False\n    #VIEW = VIEW\n    \n    y,x = img.shape\n    if y/x > 2 :\n        ratio = True\n\n    img = (img * 255).astype(np.uint8)\n    img = normalize_horiz_orientation(img)\n    \n    # ------------ truncation car difference importante entre y et x -----------\n    \n    if VIEW == 'MLO':\n        y,x = img.shape\n        y_truncated, x_truncated = int(y*0.2),0\n        img = img[y_truncated:y,x_truncated:x]\n        img = cv2.resize(img, (512,512), 0, 0, interpolation = cv2.INTER_NEAREST)\n        img = cv2.equalizeHist(img)\n      \n            \n    # ------------ pas de truncation -----------    \n    else :\n        if ratio :\n            img = img[int(y*0.1):int(y*0.9),0:x]          \n        img = cv2.resize(img, (512,512), 0, 0, interpolation = cv2.INTER_NEAREST) \n        img = cv2.equalizeHist(img)\n    \n    return img\n\n\nclass GetLoader(Dataset):\n    def __init__(self, transform, df, is_train=True):\n        self.data = df['image_path'].values\n        self.targets = df['cancer'].values\n        self.view = df['view'].values\n        self.trans = transform\n\n    def __getitem__(self, index):\n        #data = Image.open(self.data[index]).convert('RGB')\n        data = cv2.imread(self.data[index])\n        data = cv2.cvtColor(data, cv2.COLOR_BGR2GRAY)\n        VIEW =  self.view[index]\n        data = resize_and_save(data,VIEW)\n        data = cv2.cvtColor(data, cv2.COLOR_BGR2RGB)\n        data = cv2.resize(data, (512,512))\n        data = Image.fromarray(data)\n        \n        data = torch.as_tensor(self.trans(data),dtype=torch.float32).cuda()\n        targets = torch.as_tensor(self.targets[index],dtype=torch.float32).cuda()\n        return data, targets\n\n    def __len__(self):\n        return len(self.data)\n\n\ndef build_transform(is_train):\n    \n    if is_train:\n        transform = create_transform(\n            input_size=(512,512),\n            is_training=True,\n            scale=(0.75, 1.33),\n            ratio=(0.08, 1.0),\n            hflip=0.5,\n            vflip=0.5,\n            color_jitter=0.4,\n            interpolation=\"random\",\n        )\n    else:\n        transform = create_transform(\n            input_size=(512,512),\n            is_training=False,\n            interpolation=\"bilinear\",\n        )\n    return transform\n\ndef build_dataset(df, is_train=True):\n    transform = build_transform(is_train)\n    dataset = GetLoader(transform,df, is_train=True)\n    return dataset\n\nclass BalanceSampler(torch.utils.data.sampler.Sampler):\n    def __init__(self, dataset, ratio=4):\n        self.r = ratio-1\n        self.dataset = dataset\n        self.pos_index = np.where(dataset.targets>0)[0]\n        self.neg_index = np.where(dataset.targets==0)[0]\n\n        self.length = self.r*int(np.floor(len(self.neg_index)/self.r)) # = nombre de row négatifs \n\n    def __iter__(self):\n        pos_index = self.pos_index.copy()\n        neg_index = self.neg_index.copy()\n        np.random.shuffle(pos_index)\n        np.random.shuffle(neg_index)\n\n        neg_index = neg_index[:self.length].reshape(-1,self.r)\n        pos_index = np.random.choice(pos_index, self.length//self.r).reshape(-1,1)\n\n        index = np.concatenate([pos_index,neg_index],-1).reshape(-1)\n        return iter(index)\n\n    def __len__(self):\n        return self.length","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:23:46.136134Z","iopub.execute_input":"2023-01-06T18:23:46.136492Z","iopub.status.idle":"2023-01-06T18:23:46.161696Z","shell.execute_reply.started":"2023-01-06T18:23:46.136461Z","shell.execute_reply":"2023-01-06T18:23:46.160619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof = np.zeros((train_csv.shape[0],1))\n\nskf = StratifiedKFold(5, shuffle=True, random_state=7)\n\nfor fold, (train_index, test_index) in enumerate((skf.split(train_csv,train_csv['cancer']))):\n\n        print(f'---------- training fold {fold} -------------')\n        X_train, X_test = train_csv.loc[train_index], train_csv.loc[test_index]\n        y_train, y_test = train_csv.loc[train_index]['cancer'], train_csv.loc[test_index]['cancer']\n        print('nombre de cas de cancer dans train :',(X_train['cancer']==1).sum())\n        print('nombre de cas de cancer dans test :',(X_test['cancer']==1).sum())\n        \n        if fold > 0 :\n            break\n            \n        train_dataset = build_dataset(df = X_train,is_train=True)\n        sampler = BalanceSampler(train_dataset, ratio = 4)\n        train_loader = torch.utils.data.DataLoader(\n                                                    train_dataset,\n                                                    batch_size=4,\n                                                    sampler=sampler,\n                                                    drop_last=True,\n                                                    )\n        test_dataset = build_dataset(df = X_test,is_train=False)\n        test_loader = torch.utils.data.DataLoader(\n                                                    test_dataset,\n                                                    batch_size=4,\n                                                    drop_last=False\n                                                    )\n    \n  \n        model = tf_efficientnetv2_s().cuda()\n        model_without_ddp = model\n        param_groups = optim_factory.param_groups_weight_decay(model_without_ddp,0.05)\n        optimizer = torch.optim.AdamW(param_groups, lr=2.25e-4, betas=(0.9, 0.999))\n        loss = nn.BCEWithLogitsLoss(pos_weight=torch.tensor([3])).cuda()\n        \n        model.train()\n        early_stopping = 0  \n        best_metric = 0\n        best_epoch = 0\n        \n        for epoch in range(NUM_EPOCHS):\n            model.train()\n            train_loss = []\n            stream = tqdm(train_loader) \n            for i, (img, label) in enumerate(stream, start=0): \n                label = label.unsqueeze(1)\n                l = loss(model(img),label.float())\n                train_loss.append(l.item())\n                optimizer.zero_grad()\n                l.backward()\n                optimizer.step()\n                \n                loss_mean = np.mean(train_loss)\n                stream.set_description(\n                    \"Training   | Epoch: {epoch} | loss : {loss:.5f} | \".format(epoch=epoch, loss = loss_mean )\n                )\n            # ----------------- evaluation de l'epoch -----------------------\n            \n            preds1 = []\n            model.eval()\n            \n            for i, (img, label) in enumerate(tqdm(test_loader)):\n                output = torch.sigmoid(model(img)).detach().cpu()   \n                \n                if i ==0 :\n                    preds1=output.numpy()\n                else :\n                    preds1 = np.vstack([preds1,output.numpy()])\n            \n            score_epoch = np.round(pfbeta_torch(preds1,y_test),3)         \n            print(f'Epoch {epoch} pfbeta {score_epoch} ')\n            \n            if score_epoch > best_metric:\n                best_metric = score_epoch\n                best_epoch = epoch\n                best_model = f'tf_efficient_{fold}_ckpt_pytorch'\n                torch.save(model.state_dict(),best_model)\n                print('model saved')\n                oof[test_index] = preds1\n                          \n            else :\n                early_stopping +=1\n                if early_stopping == 2 :\n                    print('early stopping stop')\n                    break\n                else :\n                    print('early stopping =',early_stopping)\n                    \n        \n        print(f'===> Best epoch = {best_epoch} OOF pfbeta = {best_metric}')\n        #threshold = optimize_preds(oof[test_index], y_test, return_thresh=True, print_results=True)\n\nscore_oof = np.round(pfbeta_torch(oof,labels_train),3)\nprint(f'\\n******** OOF pfbeta {score_oof} ***************')","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:23:46.901046Z","iopub.execute_input":"2023-01-06T18:23:46.901387Z","iopub.status.idle":"2023-01-06T18:54:54.210738Z","shell.execute_reply.started":"2023-01-06T18:23:46.901357Z","shell.execute_reply":"2023-01-06T18:54:54.209347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"===> Best epoch = 4 OOF pfbeta = 0.106\n===> Best epoch = 1 OOF pfbeta = 0.103\n===> Best epoch = 0 OOF pfbeta = 0.084","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}