{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\nimport tifffile\nfrom tqdm import tqdm\n\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"IMAGE_PATH = \"/kaggle/input/prostate-cancer-grade-assessment/train_images\"\n\nTILE_SIZE = 1024\nMIN_TILES_PER_FILE = 36  # deprecated\nTRIES = 400              # faster to use for loop with fixed number of tries","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"def maybe_get_offset(img, size=TILE_SIZE//16, threshold=.95):\n    \"\"\"\n    Load a tile from a random location and return its position if its \n    \"energy\" exceeds a threshold; otherwise return None. \n    \n    To be used in a loop that generates a list of offsets.\n    \"\"\" \n    \n    threshold = threshold * 765 * size**2\n    try:\n        high = (img.shape[0] - size - 1, img.shape[1] - size - 1)\n        offset = np.random.randint(0, high=high, size=2)\n    except ValueError:\n        return None\n    score = get_score(get_tile(img, offset))\n    if score < threshold:\n        return offset.tolist()\n\ndef get_tile(img, offset, size=TILE_SIZE//16):\n    return img[offset[0]:offset[0]+size, offset[1]:offset[1]+size]\n\ndef get_score(tile):\n    # Could be replaced with a more intelligent \"energy\" function\n    return tile.sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Build list of tuples (id, offset) with TRIES tries per id.\n# Record the start and end index of the offsets for a fixed id.\n\n# Threshold = .95 seems to hit about 50% of the time.\n\noffsets = []\noffset_addresses = []\n\ncount = 0\n\nfor path in tqdm(os.listdir(IMAGE_PATH)):\n    \n    idx = path[:-5]    # shaves off '.tiff'\n    address = [idx, len(offsets)]\n    with tifffile.TiffFile(os.path.join(IMAGE_PATH, path)) as handle:\n        img = handle.asarray(2)    \n        # Using the lowest resolution image should be good enough.\n        \n        for _ in range(TRIES):\n            offset = maybe_get_offset(img, threshold=.95)\n            if offset is not None:\n                offsets.append((idx, offset))\n        del(img)\n        address.append(len(offsets))\n        offset_addresses.append(address)\n    \n    count = count + 1\n    if count > 100:    # for testing. Delete for full run.\n        break\n\nprint(f\"retrieved {len(offsets):d} offsets\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"### Visualize sample tiles ###\n\nfig = make_subplots(1,5)\n\nsample = [offset_addresses[i] for i in np.random.randint(low=0, high=100, size=5)]\nsample_id = [entry[0] for entry in sample]\nsample_offsets = [offsets[np.random.randint(low=entry[1], \n                                            high=entry[2])][1:][0] for entry in sample]\n\nfor i in range(5):\n    handle = tifffile.TiffFile(os.path.join(IMAGE_PATH, sample_id[i] + '.tiff'))\n    img = get_tile(handle.asarray(2), sample_offsets[i])\n    fig.add_trace(go.Image(z=img), 1, i + 1)\n\n    \nfig.update_layout(height=300, width=1000, title_text=\"Check out my tiles\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"### Write output tables ###\n\nimport csv\noffsets_flat = [[row[0], row[1][0]*16, row[1][1]*16] for row in offsets]\ncolumn_names = [\"file_id\", \"y_offset\", \"x_offset\"]\nwith open(f'offsets_{TILE_SIZE:d}.csv', 'w') as out:\n    w = csv.writer(out)\n    w.writerow(column_names)\n    w.writerows(offsets_flat)\n    \naddress_column_names = [\"file_id\", \"offsets_start\", \"offsets_end\"]\nwith open(f'offset_addresses_{TILE_SIZE:d}.csv', 'w') as out:\n    w = csv.writer(out)\n    w.writerow(address_column_names)\n    w.writerows(offset_addresses)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}