{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# About this kernel\n\n- This is forked from https://www.kaggle.com/appian/panda-imagehash-to-detect-duplicate-images\n- I did a kfold split with this, grouping similar images\n    - And I used `networkx` when grouping","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport torch\n# import pytorch_lightning as pl\nfrom torch import nn\nimport os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport torchvision\nimport torch.nn.functional as F\n# import timm\nimport glob\nfrom pathlib import Path\nfrom torchvision import transforms\nimport openslide","metadata":{"execution":{"iopub.status.busy":"2023-06-05T14:52:27.126369Z","iopub.execute_input":"2023-06-05T14:52:27.126789Z","iopub.status.idle":"2023-06-05T14:52:27.731884Z","shell.execute_reply.started":"2023-06-05T14:52:27.126746Z","shell.execute_reply":"2023-06-05T14:52:27.730543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# Use Lopuhin's dataset for faster image loading.\n# https://www.kaggle.com/lopuhin/panda-2020-level-1-2\n\nimport glob\nfrom pathlib import Path\n\npaths = sorted(glob.glob('/kaggle/input/prostate-cancer-grade-assessment/train_images/*.tiff'))\nprint(len(paths))\n\nimgids = [Path(p).stem.split('_')[0] for p in paths]\n\nprint(len(imgids))\nprint(len(set(imgids)))\n\nimport torch","metadata":{"execution":{"iopub.status.busy":"2023-06-05T14:52:22.899292Z","iopub.execute_input":"2023-06-05T14:52:22.899874Z","iopub.status.idle":"2023-06-05T14:52:24.395619Z","shell.execute_reply.started":"2023-06-05T14:52:22.899826Z","shell.execute_reply":"2023-06-05T14:52:24.394143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use only 4000 images for demonstration.\n\n# paths = paths[:4000]\n\n# Here comes imagehash\n# https://github.com/JohannesBuchner/imagehash\n\nimport cv2\nimport imagehash\nfrom tqdm.notebook import tqdm\nfrom PIL import Image\nimport skimage.io\nimport openslide\n\nfuncs = [\n    imagehash.average_hash,\n    imagehash.phash,\n    imagehash.dhash,\n    imagehash.whash,\n    #lambda x: imagehash.whash(x, mode='db4'),\n]\n\nhashes = []\n\nfor path in tqdm(paths, total=len(paths)):\n    image = skimage.io.MultiImage(path)[2] # Load the numpy array in lowest resolution --> for secondlowest it would already take 8+ hours\n#     image = openslide.OpenSlide(path)\n    image = Image.fromarray(image)\n    hashes.append(np.array([f(image).hash for f in funcs]).reshape(256))","metadata":{"execution":{"iopub.status.busy":"2023-05-12T14:00:00.535941Z","iopub.execute_input":"2023-05-12T14:00:00.536377Z","iopub.status.idle":"2023-05-12T14:00:01.27246Z","shell.execute_reply.started":"2023-05-12T14:00:00.536335Z","shell.execute_reply":"2023-05-12T14:00:01.271537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use cuda to speed up\nhashes = torch.Tensor(np.array(hashes).astype(int))\n\n# calc similarity scores\nsims = np.array([(hashes[i] == hashes).sum(dim=1).cpu().numpy()/256 for i in range(hashes.shape[0])])","metadata":{"execution":{"iopub.status.busy":"2023-05-04T19:30:15.501748Z","iopub.status.idle":"2023-05-04T19:30:15.502276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sims.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-04T19:30:15.508295Z","iopub.status.idle":"2023-05-04T19:30:15.509203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sims2 = sims.copy()\nnp.fill_diagonal(sims2, 0)\n\nthreshold = 0.92\nduplicates = np.where(sims2 > threshold)\n# duplicates = np.where((sims2 > threshold))  (sims2 < (threshold + 0.1))\nprint(len(duplicates[0]))","metadata":{"execution":{"iopub.status.busy":"2023-05-04T19:30:15.510723Z","iopub.status.idle":"2023-05-04T19:30:15.511813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's check image pairs with similarity larget than threshold.\n# You can lower threshold to find more duplicates (and more false positives).\n\nimport matplotlib.pyplot as plt\n\ncount = 20\ntmp = 0\n\npairs = {}\nfor i,j in zip(*duplicates):\n    if i == j:\n        continue\n\n    path1 = paths[i]\n    path2 = paths[j]\n    print(path1)\n    print(path2)\n    print(sims2[i, j])\n\n    image1 = skimage.io.MultiImage(path1)[2]\n    image2 = skimage.io.MultiImage(path2)[2]\n\n    if image1.shape[0] > image1.shape[1] / 2:\n        fig,ax = plt.subplots(figsize=(20,20), ncols=2)\n    elif image1.shape[1] > image1.shape[0] / 2:\n        fig,ax = plt.subplots(figsize=(20,20), nrows=2)\n    else:\n        fig,ax = plt.subplots(figsize=(20,30), nrows=2)\n    ax[0].imshow(image1)\n    ax[1].imshow(image2)\n    plt.show()\n    \n    tmp += 1\n    if tmp > count:\n        break","metadata":{"execution":{"iopub.status.busy":"2023-05-04T19:30:15.513278Z","iopub.status.idle":"2023-05-04T19:30:15.513733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"duplicates","metadata":{"execution":{"iopub.status.busy":"2023-05-04T19:30:15.514803Z","iopub.status.idle":"2023-05-04T19:30:15.5153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import networkx as nx\n\ng1 = nx.Graph()\nfor i, j in tqdm(zip(*duplicates)):\n    g1.add_edge(i, j)\n\nduplicates_groups = list(list(x) for x in nx.connected_components(g1))\n\nprint(len(duplicates_groups))\nlen_id = len(\"004dd32d9cd167d9cc31c13b704498af\")\n\ndf_dict = {\n    \"image_id\": list(),\n    \"group_id\": list(),\n    \"index_in_group\": list(),\n}\n\nfor group_idx, group in enumerate(duplicates_groups):\n    for indx, indx_path in enumerate(group):\n        p = Path(paths[indx_path])\n        img_id = p.stem.split('_')[0]\n        assert len(img_id) == 32\n        \n        df_dict[\"image_id\"].append(img_id)\n        df_dict[\"group_id\"].append(group_idx)\n        df_dict[\"index_in_group\"].append(indx)\n    \ndf = pd.DataFrame(df_dict)\ndisplay(df.head())\n\nprint(len(df))\nprint(len(df.image_id.unique()))","metadata":{"execution":{"iopub.status.busy":"2023-05-04T19:30:15.516186Z","iopub.status.idle":"2023-05-04T19:30:15.516602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"duplicate_imgids_imghash_thres_092.csv\", index=False) # has been made a public dataset","metadata":{"execution":{"iopub.status.busy":"2023-05-04T19:30:15.517483Z","iopub.status.idle":"2023-05-04T19:30:15.517901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Making folds ","metadata":{}},{"cell_type":"code","source":"def RandomGroupKFold_split(groups, n,df, seed=None): \n    #modified from https://stackoverflow.com/a/54254277/11808186\n    # now taken from https://stats.stackexchange.com/questions/418254/k-fold-cv-scheme-stratifying-response-and-considering-groups and adapted to work with varying size of groups\n    \"\"\"\n    Takes a dataframe with groups (from groups) and places them randomly in n folds, fills these with individual samples without groups.\n    Random analogous of sklearn.model_selection.GroupKFold.split.\n\n    :return: list of (train, test) indices based on df index\n    \"\"\"\n    groups = pd.Series(groups)\n    remaining_samples = df['image_id'][df['group_id'].isnull()]\n    print(len(remaining_samples))\n    \n    n_samples = len(groups)+len(remaining_samples)\n    ix = np.arange(n_samples)\n    unique = np.unique(groups)\n    np.random.RandomState(seed).shuffle(unique)\n    result = []\n    target_size=np.round(n_samples/n) # test target size \n    \n    samples_left = np.full_like(remaining_samples, True)\n    test_full = []\n    for split in np.array_split(unique, n): \n#         mask = groups.isin(split)\n        mask = df['group_id'].isin(split) # put these in test \n\n        n_missing = int(target_size-sum(mask))\n        sample_from = remaining_samples[samples_left] # only use samples which have not been used for test \n        if n_missing <= len(sample_from):\n            chosen = sample_from.sample(n=n_missing, replace=False, random_state=seed) # chose n_missing non-used \n            samples_left = np.logical_not(remaining_samples.isin(chosen)) & samples_left # remember which have been used before and add the newly used ones \n        else:\n            chosen = sample_from\n#         print(df.isin(chosen))\n        mask = mask | df.isin(chosen).image_id # I need to verify whether the stuff in chosen can actually be undertsood by the df.isin stuff because it kinda come from a new df and is not a boolean \n#         mask = mask.append(remaining_samples.isin(chosen)) # filling up mask with individual samples \n        \n        train, test = ix[~mask], ix[mask] # select train and test for this split \n\n        result.append((train, test))\n    \n    return result\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-05T14:52:34.849594Z","iopub.execute_input":"2023-06-05T14:52:34.850155Z","iopub.status.idle":"2023-06-05T14:52:34.867855Z","shell.execute_reply.started":"2023-06-05T14:52:34.850088Z","shell.execute_reply":"2023-06-05T14:52:34.866658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/duplicate-imgids-imghash-thres-092csv/duplicate_imgids_imghash_thres_092.csv') # all images with their group\npaths = sorted(glob.glob('/kaggle/input/prostate-cancer-grade-assessment/train_images/*.tiff'))\n\nimgids = [Path(p).stem.split('_')[0] for p in paths] # all imgids\nunique_samples = [id for id in imgids if id not in list(df['image_id'])] # all imgids which are not in a group\n\ndf2 = pd.DataFrame({'image_id':unique_samples}) # make dataframe with non-group images\ndf2 = df2.reindex(df.columns.tolist(), axis=1)\n\ndf = pd.concat([df, df2]).reset_index(drop=True) # join the dfs \n\ngroups= df['group_id'].dropna(inplace=False) # getting individual group ids \n\ntrain_df = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv')\n\n\ndf = train_df.join(other=df.set_index('image_id'), on='image_id', how='left')\n\n\nfolds = RandomGroupKFold_split(groups=groups, n=10, df=df,seed=42) ","metadata":{"execution":{"iopub.status.busy":"2023-06-05T14:52:39.29504Z","iopub.execute_input":"2023-06-05T14:52:39.29577Z","iopub.status.idle":"2023-06-05T14:52:43.782352Z","shell.execute_reply.started":"2023-06-05T14:52:39.295726Z","shell.execute_reply":"2023-06-05T14:52:43.781341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_folds = False\n\nif check_folds:\n    \n    print(len(df))\n    \n    train_d = []\n    test_d = []\n    for fold, (train, test) in enumerate(folds): \n        df.loc[train,:].to_csv(f'train_{fold}.csv')\n        df.loc[test,:].to_csv(f'test_{fold}.csv')\n        train_d += list(df.loc[train,:]['image_id'])\n        test_d += list(df.loc[test,:]['image_id'])\n    print(len(np.unique(train_d)))\n    print(len(np.unique(test_d))) # checking if all or almost all are included (depending on rounding some might be left out)\n\n    print(np.unique(df.group_id[df.index.isin(folds[1][1])])) # checking if a given fold has low group id --> these are big, should avoid this if this is the only validation they are since they are similar ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold, (train, test) in enumerate(folds): \n    train_data = df.loc[train,:]\n    test_data = df.loc[test,:]\n    counts_train = train_data['isup_grade'].value_counts()\n    counts_test = test_data['isup_grade'].value_counts()\n    \n    fig, (ax1, ax2) = plt.subplots(2)\n    fig.suptitle(f'Fold {fold}', size=15)\n    ax1.bar(height=np.array(counts_train), x=list(range(len(counts_train))))\n    ax1.set_title('Train')\n    ax2.bar(height=np.array(counts_test), x=list(range(len(counts_test))))\n    ax2.set_title('Test')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-05T14:52:47.48524Z","iopub.execute_input":"2023-06-05T14:52:47.485675Z","iopub.status.idle":"2023-06-05T14:52:50.600991Z","shell.execute_reply.started":"2023-06-05T14:52:47.485634Z","shell.execute_reply":"2023-06-05T14:52:50.599468Z"},"trusted":true},"execution_count":null,"outputs":[]}]}