{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":24800,"databundleVersionId":1831594,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-04T06:28:20.507683Z","iopub.execute_input":"2024-10-04T06:28:20.508151Z","iopub.status.idle":"2024-10-04T06:28:48.443868Z","shell.execute_reply.started":"2024-10-04T06:28:20.508108Z","shell.execute_reply":"2024-10-04T06:28:48.441591Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Conv2D, MaxPooling2D, Flatten, Dropout\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2024-10-04T06:29:15.247755Z","iopub.execute_input":"2024-10-04T06:29:15.248316Z","iopub.status.idle":"2024-10-04T06:29:30.022434Z","shell.execute_reply.started":"2024-10-04T06:29:15.248268Z","shell.execute_reply":"2024-10-04T06:29:30.021262Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport shutil\n\n# Load the labels file\nlabels_df = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\n\n# Create directories for each class\nbase_train_dir = '/kaggle/working/train/'\nos.makedirs(base_train_dir, exist_ok=True)\n\n# Loop through the dataframe and move images into class-based folders\nfor index, row in labels_df.iterrows():\n    class_label = str(row['class'])  # Assuming the class column is 'class'\n    image_id = row['image_id'] + '.dicom'  # Assuming images are .jpg, adjust if needed\n    \n    # Create class-specific folder if it doesn't exist\n    class_dir = os.path.join(base_train_dir, class_label)\n    os.makedirs(class_dir, exist_ok=True)\n    \n    # Source path (where the images currently are)\n    src_path = f'/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/{image_id}'\n    \n    # Destination path (class-based folder)\n    dest_path = os.path.join(class_dir, image_id)\n    \n    # Move the image to the class-specific folder\n    shutil.copy(src_path, dest_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-04T07:04:40.793755Z","iopub.execute_input":"2024-10-04T07:04:40.794229Z","iopub.status.idle":"2024-10-04T07:04:41.883012Z","shell.execute_reply.started":"2024-10-04T07:04:40.794188Z","shell.execute_reply":"2024-10-04T07:04:41.881503Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir","metadata":{"execution":{"iopub.status.busy":"2024-10-04T06:37:56.224795Z","iopub.execute_input":"2024-10-04T06:37:56.226558Z","iopub.status.idle":"2024-10-04T06:37:56.235222Z","shell.execute_reply.started":"2024-10-04T06:37:56.226491Z","shell.execute_reply":"2024-10-04T06:37:56.233896Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Create an ImageDataGenerator for augmenting training images\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,            # Normalize pixel values\n    rotation_range=20,         # Randomly rotate images\n    width_shift_range=0.2,     # Randomly shift images horizontally\n    height_shift_range=0.2,    # Randomly shift images vertically\n    shear_range=0.2,           # Randomly shear images\n    zoom_range=0.2,            # Randomly zoom images\n    horizontal_flip=True,      # Randomly flip images horizontally\n    validation_split=0.2       # Use 20% of training data for validation\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-04T06:35:02.298811Z","iopub.execute_input":"2024-10-04T06:35:02.299238Z","iopub.status.idle":"2024-10-04T06:35:02.306735Z","shell.execute_reply.started":"2024-10-04T06:35:02.299197Z","shell.execute_reply":"2024-10-04T06:35:02.305037Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data generator for training set\ntrain_generator = train_datagen.flow_from_directory(\n    train_dir,                 # The directory of the training images\n    target_size=(15000, 15000),     # Resize all images to 150x150\n    batch_size=32,\n    class_mode='categorical',        # Binary classification (use 'categorical' for multi-class)\n    subset='training'           # Use the training subset of the data\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T06:55:43.15982Z","iopub.execute_input":"2024-10-04T06:55:43.160416Z","iopub.status.idle":"2024-10-04T06:55:43.298272Z","shell.execute_reply.started":"2024-10-04T06:55:43.160337Z","shell.execute_reply":"2024-10-04T06:55:43.296946Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data generator for validation set\nvalidation_generator = train_datagen.flow_from_directory(\n    train_dir,\n    target_size=(150, 150),\n    batch_size=32,\n    class_mode='binary',\n    subset='validation'         # Use the validation subset of the data\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T06:40:22.917276Z","iopub.execute_input":"2024-10-04T06:40:22.917761Z","iopub.status.idle":"2024-10-04T06:40:30.637536Z","shell.execute_reply.started":"2024-10-04T06:40:22.917717Z","shell.execute_reply":"2024-10-04T06:40:30.636331Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train directory path: \", train_dir)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T06:50:31.110711Z","iopub.execute_input":"2024-10-04T06:50:31.111199Z","iopub.status.idle":"2024-10-04T06:50:31.122806Z","shell.execute_reply.started":"2024-10-04T06:50:31.11115Z","shell.execute_reply":"2024-10-04T06:50:31.121474Z"},"trusted":true},"outputs":[],"execution_count":null}]}