{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":112899,"databundleVersionId":13449579,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nbase = \"/kaggle/input\"\nfile_list = []\nfor dirname, _, filenames in os.walk(base):\n    for filename in filenames:\n        file_list.append(os.path.join(dirname, filename))\n\nprint(\"Total files found:\", len(file_list))\nprint(\"First 10 files:\", file_list[:10])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T11:58:38.204142Z","iopub.execute_input":"2025-09-25T11:58:38.204351Z","iopub.status.idle":"2025-09-25T12:01:34.565366Z","shell.execute_reply.started":"2025-09-25T11:58:38.204318Z","shell.execute_reply":"2025-09-25T12:01:34.564565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:34.566866Z","iopub.execute_input":"2025-09-25T12:01:34.567129Z","iopub.status.idle":"2025-09-25T12:01:34.833367Z","shell.execute_reply.started":"2025-09-25T12:01:34.567109Z","shell.execute_reply":"2025-09-25T12:01:34.832533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/grand-xray-slam-division-a/train1.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:34.834115Z","iopub.execute_input":"2025-09-25T12:01:34.834439Z","iopub.status.idle":"2025-09-25T12:01:35.15307Z","shell.execute_reply.started":"2025-09-25T12:01:34.834409Z","shell.execute_reply":"2025-09-25T12:01:35.152058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:35.154106Z","iopub.execute_input":"2025-09-25T12:01:35.154389Z","iopub.status.idle":"2025-09-25T12:01:35.207291Z","shell.execute_reply.started":"2025-09-25T12:01:35.154368Z","shell.execute_reply":"2025-09-25T12:01:35.206622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:35.207951Z","iopub.execute_input":"2025-09-25T12:01:35.208136Z","iopub.status.idle":"2025-09-25T12:01:35.237515Z","shell.execute_reply.started":"2025-09-25T12:01:35.20812Z","shell.execute_reply":"2025-09-25T12:01:35.236804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\n\nIMG_DIR = \"/kaggle/input/grand-xray-slam-division-a/train1\"\nfor i in range(5):\n    row = train_df.iloc[i]\n    path = os.path.join(IMG_DIR, row[\"Image_name\"])\n    img = mpimg.imread(path)\n    plt.imshow(img, cmap=\"gray\")\n    plt.title(f\"Label: {row['Patient_ID']}\")\n    plt.axis(\"off\")\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:35.300913Z","iopub.execute_input":"2025-09-25T12:01:35.301182Z","iopub.status.idle":"2025-09-25T12:01:37.755342Z","shell.execute_reply.started":"2025-09-25T12:01:35.301153Z","shell.execute_reply":"2025-09-25T12:01:37.754619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimg_path = os.path.join(IMG_DIR, os.listdir(IMG_DIR)[0])\nprint(\"Sample path:\", img_path)\n\n# open and check size\nimg = Image.open(img_path)\nprint(\"Image size (W,H):\", img.size)\nprint(\"Mode:\", img.mode)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:37.756042Z","iopub.execute_input":"2025-09-25T12:01:37.756254Z","iopub.status.idle":"2025-09-25T12:01:38.549097Z","shell.execute_reply.started":"2025-09-25T12:01:37.756238Z","shell.execute_reply":"2025-09-25T12:01:38.54849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# IMG_DIR = \"/kaggle/input/grand-xray-slam-division-a/train1\"\n\n# # Choose the series ID you want to see\n# series_id = \"00000011\"\n\n# # Get all files with that series id\n# series_files = [f for f in os.listdir(IMG_DIR) if f.split(\"_\")[0] == series_id]\n\n# print(f\"Found {len(series_files)} images for series {series_id}\")\n\n# # Plot them\n# plt.figure(figsize=(20, 10))\n# for i, fname in enumerate(series_files):\n#     path = os.path.join(IMG_DIR, fname)\n#     img = Image.open(path)\n    \n#     plt.subplot(1, len(series_files), i+1)\n#     plt.imshow(img, cmap=\"gray\")\n#     plt.axis(\"off\")\n#     plt.title(fname)\n\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:38.567776Z","iopub.execute_input":"2025-09-25T12:01:38.568008Z","iopub.status.idle":"2025-09-25T12:01:38.581014Z","shell.execute_reply.started":"2025-09-25T12:01:38.567993Z","shell.execute_reply":"2025-09-25T12:01:38.580197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# from PIL import Image\n# import numpy as np\n\n# # Path to a sample grayscale X-ray\n# img_path = \"/kaggle/input/grand-xray-slam-division-a/train1/00000011_013_001.jpg\"\n\n# # Open grayscale image\n# img_gray = Image.open(img_path).convert(\"L\")  # ensure grayscale\n\n# # Convert grayscale to RGB by repeating the channel\n# img_rgb = img_gray.convert(\"RGB\")  # or np.stack([np.array(img_gray)]*3, axis=-1)\n\n# # Plot both images side by side\n# plt.figure(figsize=(10,5))\n\n# plt.subplot(1,2,1)\n# plt.imshow(img_gray, cmap=\"gray\")\n# plt.title(\"Original Grayscale\")\n# plt.axis(\"off\")\n\n# plt.subplot(1,2,2)\n# plt.imshow(img_rgb)\n# plt.title(\"Converted to RGB\")\n# plt.axis(\"off\")\n\n# plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:38.581906Z","iopub.execute_input":"2025-09-25T12:01:38.582637Z","iopub.status.idle":"2025-09-25T12:01:39.486061Z","shell.execute_reply.started":"2025-09-25T12:01:38.582613Z","shell.execute_reply":"2025-09-25T12:01:39.485222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications import ResNet50","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T14:21:34.12117Z","iopub.execute_input":"2025-09-26T14:21:34.121378Z","iopub.status.idle":"2025-09-26T14:21:53.175774Z","shell.execute_reply.started":"2025-09-26T14:21:34.121352Z","shell.execute_reply":"2025-09-26T14:21:53.17508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model = ResNet50(\n    input_shape=(224, 224, 3),\n    include_top = False,\n    weights = None\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T14:21:57.236969Z","iopub.execute_input":"2025-09-26T14:21:57.237578Z","iopub.status.idle":"2025-09-26T14:22:01.244574Z","shell.execute_reply.started":"2025-09-26T14:21:57.237552Z","shell.execute_reply":"2025-09-26T14:22:01.243784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers, models\nbase_model.trainable = False\nx = base_model.get_layer(\"conv4_block4_out\").output\nx = layers.GlobalAveragePooling2D()(x)\nx = layers.Dense(500,activation ='relu')(x)\nx = layers.Dense(250,activation ='relu')(x)\nx = layers.Dropout(0.2)(x)\noutput = layers.Dense(14,activation ='sigmoid')(x)\n\nmodel  = tf.keras.Model(inputs =base_model.input,outputs=output)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T14:23:03.486839Z","iopub.execute_input":"2025-09-26T14:23:03.487133Z","iopub.status.idle":"2025-09-26T14:23:03.528605Z","shell.execute_reply.started":"2025-09-26T14:23:03.487111Z","shell.execute_reply":"2025-09-26T14:23:03.528055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\nearly_stop = EarlyStopping(\n    monitor='val_loss',    \n    patience=3,            \n    restore_best_weights=True, \n    min_delta=1e-4,        \n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T14:23:25.17092Z","iopub.execute_input":"2025-09-26T14:23:25.171495Z","iopub.status.idle":"2025-09-26T14:23:25.178579Z","shell.execute_reply.started":"2025-09-26T14:23:25.171468Z","shell.execute_reply":"2025-09-26T14:23:25.178044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.metrics import AUC\nmodel.compile(optimizer= tf.keras.optimizers.Adam(0.001),\n              loss='binary_crossentropy',\n              metrics=[AUC(name='auc', multi_label=True, num_labels= 14)]\n             )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T14:28:01.732171Z","iopub.execute_input":"2025-09-26T14:28:01.732503Z","iopub.status.idle":"2025-09-26T14:28:01.744819Z","shell.execute_reply.started":"2025-09-26T14:28:01.732481Z","shell.execute_reply":"2025-09-26T14:28:01.744267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_list = [img for img in train_df['Image_name']]\nimg_path  = [os.path.join(IMG_DIR,img) for img in train_df['Image_name']]\ntarget = train_df[train_df.columns[7:]].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:56.959791Z","iopub.execute_input":"2025-09-25T12:01:56.960036Z","iopub.status.idle":"2025-09-25T12:01:57.066049Z","shell.execute_reply.started":"2025-09-25T12:01:56.960006Z","shell.execute_reply":"2025-09-25T12:01:57.065478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = (224,224)\ndef pre_process(path, label):\n    img = tf.io.read_file(path)\n    \n    img = tf.image.decode_jpeg(img, channels=3)\n    \n    img = tf.image.resize(img, IMG_SIZE)\n    \n    img = tf.keras.applications.resnet50.preprocess_input(img)\n    \n    label = tf.cast(label, tf.float32)\n    \n    return img, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T12:01:57.066736Z","iopub.execute_input":"2025-09-25T12:01:57.066975Z","iopub.status.idle":"2025-09-25T12:01:57.076006Z","shell.execute_reply.started":"2025-09-25T12:01:57.066957Z","shell.execute_reply":"2025-09-25T12:01:57.075392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_paths, val_paths, train_labels, val_labels = train_test_split(\n    img_path, target, test_size=0.1, random_state=42\n)\ndataset = tf.data.Dataset.from_tensor_slices((train_paths, train_labels))\ndataset = dataset.map(pre_process, num_parallel_calls=tf.data.AUTOTUNE)\ndataset = dataset.shuffle(10000).batch(1000).prefetch(tf.data.AUTOTUNE)\n\n\n\nval_dataset = tf.data.Dataset.from_tensor_slices((val_paths, val_labels))\nval_dataset = val_dataset.map(pre_process, num_parallel_calls=tf.data.AUTOTUNE)\nval_dataset = val_dataset.batch(1000).prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T14:24:11.366771Z","iopub.execute_input":"2025-09-26T14:24:11.367112Z","iopub.status.idle":"2025-09-26T14:24:11.589062Z","shell.execute_reply.started":"2025-09-26T14:24:11.367088Z","shell.execute_reply":"2025-09-26T14:24:11.588018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#checkpoint_path = \"/kaggle/working/resnet50_multi_label.weights.h5\"\n# checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n#     filepath=checkpoint_path,\n#     save_weights_only=True,\n#     monitor='val_loss',\n#     mode='min',\n#     save_best_only=True,\n#     verbose=1\n# )\nhistory = model.fit(\n    dataset,\n    validation_data = val_dataset,\n    epochs = 10,\n    callbacks=[early_stop]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T14:24:05.796357Z","iopub.execute_input":"2025-09-26T14:24:05.797083Z","iopub.status.idle":"2025-09-26T14:24:05.811644Z","shell.execute_reply.started":"2025-09-26T14:24:05.797059Z","shell.execute_reply":"2025-09-26T14:24:05.810781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(history.history.keys())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T17:20:02.640537Z","iopub.execute_input":"2025-09-25T17:20:02.641284Z","iopub.status.idle":"2025-09-25T17:20:02.645107Z","shell.execute_reply.started":"2025-09-25T17:20:02.641256Z","shell.execute_reply":"2025-09-25T17:20:02.644403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot Loss\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\nplt.show()\n\n# Plot AUC\nplt.plot(history.history['auc'], label='Train AUC')\nplt.plot(history.history['val_auc'], label='Validation AUC')\nplt.xlabel('Epochs')\nplt.ylabel('AUC')\nplt.title('Training and Validation AUC')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-25T17:20:27.006283Z","iopub.execute_input":"2025-09-25T17:20:27.006802Z","iopub.status.idle":"2025-09-25T17:20:27.577137Z","shell.execute_reply.started":"2025-09-25T17:20:27.006776Z","shell.execute_reply":"2025-09-25T17:20:27.576471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/grand-xray-slam-division-a/sample_submission_1.csv')\n\n\ntest_dir = \"/kaggle/input/grand-xray-slam-division-a/test1\"\ntest_paths = [os.path.join(test_dir, fname) for fname in submission['Image_name']]\n\n\nIMG_SIZE = (224, 224)\ndef preprocess_test(path):\n    img = tf.io.read_file(path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, IMG_SIZE)\n    img = tf.keras.applications.resnet50.preprocess_input(img)\n    return img\n\n\ntest_ds = tf.data.Dataset.from_tensor_slices(test_paths)\ntest_ds = test_ds.map(preprocess_test, num_parallel_calls=tf.data.AUTOTUNE)\ntest_ds = test_ds.batch(32).prefetch(tf.data.AUTOTUNE)\n\n\npreds = model.predict(test_ds, verbose=1)\n\n\nthreshold = 0.5\npreds_binary = (preds >= threshold).astype(int)\n\nsubmission.iloc[:, 1:] = preds_binary\n\n\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\nprint(\"✅ Submission saved:\", submission.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-26T14:28:08.627131Z","iopub.execute_input":"2025-09-26T14:28:08.627853Z","iopub.status.idle":"2025-09-26T14:28:08.644689Z","shell.execute_reply.started":"2025-09-26T14:28:08.627828Z","shell.execute_reply":"2025-09-26T14:28:08.643624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}