{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.18"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":113002,"databundleVersionId":13471427,"sourceType":"competition"}],"dockerImageVersionId":31091,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":29864.681816,"end_time":"2025-10-13T04:10:55.383866","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-10-12T19:53:10.70205","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"4126cc93","cell_type":"code","source":"# Grand X-Ray Slam: Division B - EfficientNetB0 Model + TPU\n# An advanced transfer learning approach for image-only classification.\n# VERSION: Reverted to image-only model, paths consolidated.\n\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport os\nimport sys\nfrom io import StringIO\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, GlobalAveragePooling2D, Lambda, Input\nfrom tensorflow.keras.optimizers import Adam\nimport tensorflow as tf\nfrom tensorflow.keras.applications import efficientnet\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm\nimport numpy as np\n\n\n# --- Global Settings ---\nDEBUG = False\nEPOCHS = 8 if not DEBUG else 3 \n\n# --- Path Definitions ---\nBASE_PATH = '/kaggle/input/grand-xray-slam-division-b/'\nTRAIN_CSV_PATH = os.path.join(BASE_PATH, 'train2.csv')\nTRAIN_IMAGE_DIR = os.path.join(BASE_PATH, 'train2/')\nTEST_IMAGE_DIR = os.path.join(BASE_PATH, 'test2/')\nSAMPLE_SUBMISSION_PATH = os.path.join(BASE_PATH, 'sample_submission_2.csv')\nSUBMISSION_PATH = '/kaggle/working/submission.csv'\n\nprint(f\"TensorFlow version: {tf.__version__}\")\nif DEBUG:\n    print(\"🔥🔥🔥 RUNNING IN DEBUG MODE ON A SMALL SUBSET OF DATA 🔥🔥🔥\")\n\n# --- Step 1: Detect TPU, Define Strategy, and Set Mixed Precision ---\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect(tpu='local')\n    print('✅ Running on TPU ', tpu.master())\n    strategy = tf.distribute.TPUStrategy(tpu)\n    tf.keras.mixed_precision.set_global_policy('mixed_bfloat16')\n    print('✅ Mixed precision (mixed_bfloat16) enabled for TPU.')\nexcept ValueError:\n    print('❌ No TPU found, using CPU/GPU strategy')\n    tpu = None\n    strategy = tf.distribute.get_strategy()\n\nprint(f\"REPLICAS: {strategy.num_replicas_in_sync}\")\n\n# --- Step 2: Load and Prepare Data ---\nprint(\"\\n--- Loading and Preparing Data ---\")\ntry:\n    df_train = pd.read_csv(TRAIN_CSV_PATH)\n    print(f\"Loaded train2.csv with {len(df_train)} rows\")\nexcept FileNotFoundError:\n    print(f\"Error: {TRAIN_CSV_PATH} not found. Please ensure the Kaggle dataset is attached.\")\n    sys.exit()\n\nif DEBUG:\n    df_train = df_train.sample(n=100, random_state=42)\n    print(f\"DEBUG MODE: Sliced training data to {len(df_train)} rows.\")\n\nconditions = [\n    'Atelectasis', 'Cardiomegaly', 'Consolidation', 'Edema', 'Enlarged Cardiomediastinum',\n    'Fracture', 'Lung Lesion', 'Lung Opacity', 'No Finding', 'Pleural Effusion',\n    'Pleural Other', 'Pneumonia', 'Pneumothorax', 'Support Devices'\n]\n\ntrain_set, val_set = train_test_split(\n    df_train, test_size=0.15, random_state=42, stratify=None if DEBUG else df_train['No Finding']\n)\nprint(f\"Training samples: {len(train_set)}, Validation samples: {len(val_set)}\")\n\n\n# --- Step 3: Calculate Class Weights for Weighted Loss ---\nprint(\"\\n--- Calculating Class Weights for Imbalance ---\")\ndef calculate_class_weights(df, max_weight_cap=20.0):\n    pos_weights = {}\n    neg_weights = {}\n    total_samples = len(df)\n    for i, condition in enumerate(conditions):\n        pos_counts = df[condition].sum()\n        neg_counts = total_samples - pos_counts\n        pos_w = total_samples / (2 * pos_counts) if pos_counts > 0 else 1.0\n        neg_w = total_samples / (2 * neg_counts) if neg_counts > 0 else 1.0\n        pos_weights[i] = min(pos_w, max_weight_cap)\n        neg_weights[i] = neg_w\n    return pos_weights, neg_weights\n\npos_weights, neg_weights = calculate_class_weights(train_set)\npos_weights_tensor = tf.constant([pos_weights[i] for i in range(len(conditions))], dtype=tf.float32)\nneg_weights_tensor = tf.constant([neg_weights[i] for i in range(len(conditions))], dtype=tf.float32)\n\ndef get_weighted_loss(pos_weights, neg_weights):\n    def weighted_loss(y_true, y_pred):\n        y_pred = tf.clip_by_value(y_pred, tf.keras.backend.epsilon(), 1 - tf.keras.backend.epsilon())\n        loss = pos_weights * y_true * tf.math.log(y_pred) + neg_weights * (1 - y_true) * tf.math.log(1 - y_pred)\n        return -tf.reduce_mean(loss)\n    return weighted_loss\n\n\n# --- Step 4: Create High-Performance tf.data Pipeline ---\nprint(\"\\n--- Creating tf.data Pipelines with Augmentation ---\")\n\nBATCH_SIZE_PER_REPLICA = 16\nGLOBAL_BATCH_SIZE = BATCH_SIZE_PER_REPLICA * strategy.num_replicas_in_sync\nIMG_SIZE = (512, 512)\nAUTOTUNE = tf.data.AUTOTUNE\n\ndef augment(image, label):\n    # Random horizontal flip\n    image = tf.image.random_flip_left_right(image)\n    \n    # Random small rotation using rot90 (approximate)\n    k = tf.random.uniform([], minval=0, maxval=4, dtype=tf.int32)\n    image = tf.image.rot90(image, k)\n    \n    # Random brightness & contrast\n    image = tf.image.random_brightness(image, max_delta=0.05)\n    image = tf.image.random_contrast(image, lower=0.9, upper=1.1)\n    \n    # Add Gaussian noise\n    # noise = tf.random.normal(tf.shape(image), mean=0.0, stddev=0.01)\n    # image = tf.clip_by_value(image + noise, 0.0, 1.0)\n    \n    return image, label\n\ndef parse_function(filename, labels):\n    \"\"\"Loads and preprocesses an image file.\"\"\"\n    image_string = tf.io.read_file(filename)\n    is_not_empty = tf.strings.length(image_string) > 0\n    \n    def process_normal():\n        image = tf.io.decode_image(image_string, channels=3, expand_animations=False)\n        is_valid_image = tf.greater(tf.size(image), 0)\n        def resize_and_prep():\n            return efficientnet.preprocess_input(tf.image.resize(image, IMG_SIZE))\n        def empty_img():\n             return tf.zeros((*IMG_SIZE, 3), dtype=tf.float32)\n        return tf.cond(is_valid_image, resize_and_prep, empty_img)\n\n    def process_empty():\n        return tf.zeros((*IMG_SIZE, 3), dtype=tf.float32)\n\n    processed_image = tf.cond(is_not_empty, process_normal, process_empty)\n    processed_image.set_shape((*IMG_SIZE, 3))\n    \n    return processed_image, labels\n\ndef create_dataset(dataframe, image_dir, is_training=False, is_test=False):\n    \"\"\"Creates a high-performance tf.data.Dataset.\"\"\"\n    filepaths = image_dir + dataframe['Image_name']\n    \n    if is_test:\n        labels = np.zeros((len(dataframe), len(conditions)), dtype=np.float32)\n    else:\n        labels = dataframe[conditions].values.astype(np.float32)\n\n    ds = tf.data.Dataset.from_tensor_slices((filepaths, labels))\n    ds = ds.map(parse_function, num_parallel_calls=AUTOTUNE)\n    \n    if is_training:\n        ds = ds.shuffle(buffer_size=1000)\n        ds = ds.map(augment, num_parallel_calls=AUTOTUNE)\n        ds = ds.repeat()\n\n    ds = ds.batch(GLOBAL_BATCH_SIZE)\n    ds = ds.prefetch(buffer_size=AUTOTUNE)\n    return ds\n\ntrain_ds = create_dataset(train_set, TRAIN_IMAGE_DIR, is_training=True)\nval_ds = create_dataset(val_set, TRAIN_IMAGE_DIR, is_training=False)\nprint(f\"Global batch size: {GLOBAL_BATCH_SIZE}\")\n\n\n# --- Step 5: Build EfficientNetB0 Model ---\nprint(\"\\n--- Building and Compiling EfficientNetB0 Model with Fine-Tuning ---\")\n\ndef build_efficientnet_model(num_classes=14):\n    \"\"\"Builds an EfficientNetB0 model and fine-tunes the top layers.\"\"\"\n    image_input = Input(shape=(*IMG_SIZE, 3), name='image_input')\n    base_model = efficientnet.EfficientNetB0(weights='imagenet', include_top=False, input_tensor=image_input)\n    base_model.trainable = True\n    \n    fine_tune_at = 100 \n    for layer in base_model.layers[:fine_tune_at]:\n        layer.trainable = False\n        \n    x = GlobalAveragePooling2D()(base_model.output)\n    x = Dropout(0.5)(x)\n    outputs = Dense(num_classes, activation='sigmoid', dtype='float32')(x)\n    \n    model = Model(inputs=image_input, outputs=outputs, name=\"EfficientNetB0_Image_Only\")\n    return model\n\nwith strategy.scope():\n    model = build_efficientnet_model(num_classes=len(conditions))\n    scaled_lr = 1e-4 * strategy.num_replicas_in_sync \n    weighted_binary_crossentropy = get_weighted_loss(pos_weights_tensor, neg_weights_tensor)\n    model.compile(optimizer=Adam(learning_rate=scaled_lr), loss=weighted_binary_crossentropy, metrics=['AUC'])\n\nprint(\"Model Architecture: EfficientNetB0\")\nmodel.summary()\n\n\n# --- Step 6: Train the Model ---\nprint(\"\\n--- Starting Model Training ---\")\nearly_stopper = EarlyStopping(monitor='val_loss', patience=30, verbose=1, restore_best_weights=True)\nlr_reducer = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=1, verbose=1, min_lr=1e-6)\nsteps_per_epoch = max(1, len(train_set) // GLOBAL_BATCH_SIZE)\nvalidation_steps = max(1, len(val_set) // GLOBAL_BATCH_SIZE)\n\nhistory = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS,\n    steps_per_epoch=steps_per_epoch,\n    validation_steps=validation_steps,\n    callbacks=[early_stopper, lr_reducer],\n    verbose=1\n)\n\nval_auc = max(history.history.get('val_AUC', [0.0]))\nprint(f\"Best Validation AUC-ROC achieved: {val_auc:.4f}\")\n\n\n# --- Step 7: Analyze Validation Predictions ---\nprint(\"\\n--- Analyzing Validation Predictions and Displaying Examples ---\")\n\ndef plot_per_label_auc(true_labels, predictions):\n    plt.figure(figsize=(10, 8))\n    auc_scores = [roc_auc_score(true_labels[:, i], predictions[:, i]) for i in range(len(conditions)) if len(np.unique(true_labels[:, i])) > 1]\n    valid_conditions = [c for i, c in enumerate(conditions) if len(np.unique(true_labels[:, i])) > 1]\n    auc_df = pd.DataFrame({'Condition': valid_conditions, 'AUC': auc_scores})\n    auc_df = auc_df.sort_values(by='AUC', ascending=True)\n    bars = plt.barh(auc_df['Condition'], auc_df['AUC'], color='skyblue')\n    plt.xlabel('AUC Score'); plt.title('AUC per Condition'); plt.xlim(0, 1)\n    for bar in bars: plt.text(bar.get_width() + 0.01, bar.get_y() + bar.get_height()/2, f'{bar.get_width():.3f}', va='center', ha='left')\n    plt.tight_layout(); plt.show()\n\ndef plot_prediction_examples(indices, predictions, ground_truth_df, image_dir, status):\n    print(f\"\\n--- Displaying 5 {status.capitalize()} Prediction Examples ---\")\n    for i, idx in enumerate(indices):\n        fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 5), gridspec_kw={'width_ratios': [1, 1.2]})\n        img_name = ground_truth_df.iloc[idx]['Image_name']\n        img_path = os.path.join(image_dir, img_name)\n        image = cv2.cvtColor(cv2.imread(img_path), cv2.COLOR_BGR2RGB)\n        ax1.imshow(image); ax1.set_title(f\"Example #{i+1}\\n{img_name}\"); ax1.axis('off')\n        \n        ax2.axis('off')\n        pred_labels = predictions[idx]\n        true_labels = ground_truth_df.iloc[idx][conditions].values\n        table_data = [[\"Condition\", \"Actual\", \"Predicted\"]]\n        table_data.extend([[c, int(t), f\"{p:.3f}\"] for c, t, p in zip(conditions, true_labels, pred_labels)])\n        table = ax2.table(cellText=table_data, colWidths=[0.5, 0.2, 0.2], cellLoc='left', loc='center')\n        table.auto_set_font_size(False); table.set_fontsize(9); table.scale(1.0, 1.2)\n        \n        for (row, col), cell in table.get_celld().items():\n            if row == 0: cell.set_text_props(weight='bold')\n            if col == 1 and row > 0: cell.set_facecolor(\"#90EE90\" if table_data[row][1] == 1 else \"#FFB6C1\")\n            if col == 2 and row > 0: cell.set_facecolor(\"#90EE90\" if float(table_data[row][2]) >= 0.5 else \"#FFB6C1\")\n\n        plt.tight_layout(); plt.show()\n\nval_predictions = model.predict(val_ds, verbose=1)\nval_predictions = val_predictions[:len(val_set)]\ntrue_labels_val = val_set[conditions].values\n\nplot_per_label_auc(true_labels_val, val_predictions)\nmae = np.mean(np.abs(true_labels_val - val_predictions), axis=1)\nsorted_indices = np.argsort(mae)\nplot_prediction_examples(sorted_indices[:5], val_predictions, val_set, TRAIN_IMAGE_DIR, \"good\")\nplot_prediction_examples(sorted_indices[-5:], val_predictions, val_set, TRAIN_IMAGE_DIR, \"wrong\")\n\n\n# --- Step 8: Generate Submission with Test-Time Augmentation (TTA) ---\nprint(\"\\n--- Generating Submission File with Test-Time Augmentation (TTA) ---\")\ntry:\n    submission_df = pd.read_csv(SAMPLE_SUBMISSION_PATH)\n    print(f\"Loaded sample_submission_2.csv with {len(submission_df)} rows\")\nexcept FileNotFoundError:\n    print(f\"Error: {SAMPLE_SUBMISSION_PATH} not found.\"); sys.exit()\n\nif DEBUG:\n    submission_df = submission_df.head(100)\n\n# Create a dataset for the original test images\nprint(\"🚀 Predicting on original test images...\")\ntest_ds_orig = create_dataset(submission_df, TEST_IMAGE_DIR, is_test=True)\npredictions_orig = model.predict(test_ds_orig, verbose=1)\npredictions_orig = predictions_orig[:len(submission_df)]\n\n# Create a dataset for the flipped test images\nprint(\"🔄 Predicting on horizontally flipped test images...\")\ndef flip_parse_function(filename, labels):\n    \"\"\"Loads and preprocesses an image file and flips it.\"\"\"\n    image, labels = parse_function(filename, labels)\n    image = tf.image.flip_left_right(image)\n    return image, labels\n\ndef create_flipped_dataset(dataframe, image_dir, is_test=True):\n    filepaths = image_dir + dataframe['Image_name']\n    labels = np.zeros((len(dataframe), len(conditions)), dtype=np.float32)\n    ds = tf.data.Dataset.from_tensor_slices((filepaths, labels))\n    ds = ds.map(flip_parse_function, num_parallel_calls=AUTOTUNE)\n    ds = ds.batch(GLOBAL_BATCH_SIZE)\n    ds = ds.prefetch(buffer_size=AUTOTUNE)\n    return ds\n\ntest_ds_flipped = create_flipped_dataset(submission_df, TEST_IMAGE_DIR, is_test=True)\npredictions_flipped = model.predict(test_ds_flipped, verbose=1)\npredictions_flipped = predictions_flipped[:len(submission_df)]\n\n# Average the predictions from both original and flipped images\nprint(\"📊 Averaging predictions...\")\nfinal_predictions = (predictions_orig + predictions_flipped) / 2.0\n\n# Generate the final submission file\nsubmission_df[conditions] = final_predictions\nsubmission_df.to_csv(SUBMISSION_PATH, index=False)\n\nprint(f\"\\nSubmission file created with TTA: {SUBMISSION_PATH}\")\nprint(\"Script finished successfully!\")\n\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2025-10-12T19:53:13.593721Z","iopub.status.busy":"2025-10-12T19:53:13.59355Z","iopub.status.idle":"2025-10-13T04:10:45.701519Z","shell.execute_reply":"2025-10-13T04:10:45.695624Z"},"papermill":{"duration":29852.115436,"end_time":"2025-10-13T04:10:45.704455","exception":false,"start_time":"2025-10-12T19:53:13.589019","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}