{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nimport shutil\nimport pandas as pd\n\ntraining_data = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n\ncancerous_data = training_data[training_data['cancer'] == 1]\n\ncancer_vector = cancerous_data['patient_id'].drop_duplicates().astype('str').tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T23:10:31.026806Z","iopub.execute_input":"2024-10-17T23:10:31.027264Z","iopub.status.idle":"2024-10-17T23:10:31.152236Z","shell.execute_reply.started":"2024-10-17T23:10:31.027221Z","shell.execute_reply":"2024-10-17T23:10:31.150978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport shutil\n\ndata_root = '/kaggle/input/rsna-breast-cancer-detection'\nk = 400 # randomly select images from train folder\nos.makedirs('./dataset', exist_ok=True) # create a dataset folder to hold all the files that I wanted to download\nshutil.copy(os.path.join(data_root, 'train.csv'), 'dataset/train.csv')\n\ndir_path = os.path.join(data_root, 'train_images')\nfiles = os.listdir(dir_path)\n\nfiles = [file for file in files if file not in cancer_vector]\n\nfiles = list(set(files))\n\ntarget_dir = os.path.join('dataset', 'train_images')\nos.makedirs(target_dir, exist_ok=True) \n\nfor f in random.choices(cancer_vector, k=k):\n    src_file = os.path.join(dir_path, f)\n    destination = os.path.join(target_dir, f)\n    shutil.copytree(src_file, destination, dirs_exist_ok=True)\n\nfor f in random.choices(files, k=k): # randomly select k images and copy them to the target folder\n    src_file = os.path.join(dir_path, f)\n    destination = os.path.join(target_dir, f)\n    shutil.copytree(src_file, destination, dirs_exist_ok=True)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T23:10:31.155081Z","iopub.execute_input":"2024-10-17T23:10:31.155647Z","iopub.status.idle":"2024-10-17T23:15:29.825211Z","shell.execute_reply.started":"2024-10-17T23:10:31.155566Z","shell.execute_reply":"2024-10-17T23:15:29.823378Z"},"trusted":true},"execution_count":null,"outputs":[]}]}