{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":6799,"databundleVersionId":4225553},{"sourceType":"competition","sourceId":13451,"datasetId":654585,"databundleVersionId":1188070}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"We use ImageNet's recommended dataset to create images of objects in chosen categories. Each category has its own folder containing cropped images focused on the object's bounding box.","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nfrom shutil import copy, make_archive\nimport pandas as pd\nfrom PIL import Image\nimport xml.etree.ElementTree as xml\n\ndata_root = '/kaggle/input/imagenet-object-localization-challenge'\nk = 100 # randomly select 100 images from both train and test data folder\nos.makedirs('./dataset', exist_ok=True) # create a dataset folder to hold all the files that I wanted to download\n\nobjs = ['banana', 'pineapple', 'electric guitar', 'envelope', 'football helmet', 'frypan', 'hammer', 'handkerchief', 'joystick', 'lipstick', 'loudspeaker', 'microwave', 'paper towel', 'pillow', 'ping-pong ball', 'pot', 'refrigerator', 'saxophone', 'soccer ball', 'tripod', 'plate', 'pretzel', 'hotdog', 'pizza']\n\nwith open(os.path.join(data_root, 'LOC_synset_mapping.txt')) as f: mapping_file = f.read()\nmappings = [\n    mapping\n    for mapping in (\n        (\n            # 2. first word is the ID\n            words[0],\n            # 3. every other word can be joined back to get our label, which is a comma-separated list of different names of the category\n            [version.strip() for version in \" \".join(words[1:]).split(\",\")],\n        )\n        for words in\n        # 1. split file into lines, then the each line into words\n        (line.lower().split() for line in mapping_file.split(\"\\n\") if line)\n    )\n    # 4. filter out the mappings that are not in our list of objects\n    if any((category in objs) for category in mapping[1])\n]\n\nfor mapping in mappings:\n    id, names = mapping\n    files = os.listdir(os.path.join(data_root, 'ILSVRC/Annotations/CLS-LOC/train', id))\n    target_dir = os.path.join('./dataset', names[0])\n    os.makedirs(target_dir, exist_ok=True)\n    print(names[0])\n    \n    for file in files:\n        image_filename = file.replace('.xml', '.JPEG')\n        src_image_path = os.path.join(data_root, 'ILSVRC/Data/CLS-LOC/train', id, image_filename)\n        target_image_path = os.path.join(target_dir, file.replace('.xml', '.JPEG'))\n\n        if not os.path.exists(src_image_path):\n            print(f\"Image not found: {src_image_path}\")\n            continue\n        if os.path.exists(target_image_path):\n            print(f\"Target image already exists: {target_image_path}\")\n            continue\n        \n        # get bounding box\n        annotation_path = os.path.join(data_root, 'ILSVRC/Annotations/CLS-LOC/train', id, file)\n        with open(annotation_path) as f:\n            annotation = f.read()\n        tree = xml.parse(annotation_path)\n        root = tree.getroot()\n        bndbox = root.find('object/bndbox')\n        xmin = int(bndbox.find('xmin').text)\n        ymin = int(bndbox.find('ymin').text)\n        xmax = int(bndbox.find('xmax').text)\n        ymax = int(bndbox.find('ymax').text)\n\n        # crop and save the image\n        image = Image.open(src_image_path)\n        image = image.crop((xmin, ymin, xmax, ymax))\n        \n        image.save(target_image_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T14:32:54.62697Z","iopub.execute_input":"2025-04-29T14:32:54.627299Z","iopub.status.idle":"2025-04-29T14:39:31.551536Z","shell.execute_reply.started":"2025-04-29T14:32:54.627268Z","shell.execute_reply":"2025-04-29T14:39:31.550466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from shutil import copy, make_archive\nmake_archive(base_name='download_dataset', format='zip', root_dir='dataset')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-29T13:11:36.557441Z","iopub.execute_input":"2025-04-29T13:11:36.558379Z","iopub.status.idle":"2025-04-29T13:11:36.821137Z","shell.execute_reply.started":"2025-04-29T13:11:36.558338Z","shell.execute_reply":"2025-04-29T13:11:36.82008Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now you can download the `download_dataset.zip` from the kernel interface.\n\n![](https://user-images.githubusercontent.com/1262709/69744216-c2e11780-110d-11ea-82b4-88006cc6d0aa.png)","metadata":{}}]}