{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Gaussian Log-Likelihood Function: Custom loss function is implemented as per your project requirements for spectra and uncertainty prediction.\n\nModel Architecture: A CNN model was built using Conv1D layers to handle spectral data. It predicts both the mean (mu) and uncertainty (sigma).\n\nMetrics Plotting: Functions to plot loss, accuracy, confusion matrix, and ROC curve were added to visually evaluate the model's performance.\n\nPrediction Submission: The predictions are generated in the format required for submission, and the result is saved as a CSV file.\n\nHere’s a rewritten version of the provided code outline for improving probabilities through Data Augmentation (using spectral variations and synthetic data):\n\nThis code implements the suggested improvements, such as data augmentation via noise and synthetic sample generation, to boost model performance. Additionally, it includes the Gaussian Log-likelihood (GLL) loss function and evaluates model performance using the appropriate metrics for the challenge task.\n\nThis section of the algorithms in the esenmble solutions returns file not found!. Does any one have a way around this?\n\n     \" signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/{dataset}/{planet_id}/{sensor}_signal.parquet').to_numpy()\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, roc_curve, auc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# --------------------- Custom Loss Function: Gaussian Log-Likelihood (GLL) ---------------------\ndef gaussian_log_likelihood(y_true, y_pred):\n    \"\"\"Calculate Gaussian Log-Likelihood for spectral predictions and uncertainties.\"\"\"\n    y_true_mu = y_true[:, :283]  # Extract true spectral values (mean)\n    y_pred_mu = y_pred[:, :283]  # Predicted spectral values (mean)\n    y_pred_sigma = y_pred[:, 283:]  # Predicted uncertainties (sigma)\n\n    # GLL formula: -(1/2) * (log(2π) + log(σ^2) + ((y - μ)^2) / σ^2)\n    log_term = tf.math.log(2.0 * np.pi) + tf.math.log(tf.square(y_pred_sigma))\n    squared_diff = tf.square(y_true_mu - y_pred_mu)\n    loss = 0.5 * (log_term + squared_diff / tf.square(y_pred_sigma))\n    return tf.reduce_mean(loss)\n\n# --------------------- Data Augmentation ---------------------\ndef augment_spectral_data(X):\n    \"\"\"Apply augmentations such as noise, shifts, or scaling to spectral data.\"\"\"\n    noise_factor = 0.01  # Noise to be added to the spectral data\n    scale_factor = np.random.uniform(0.95, 1.05, X.shape)  # Scaling factor\n    \n    # Add random noise\n    noisy_data = X + noise_factor * np.random.normal(loc=0.0, scale=1.0, size=X.shape)\n    \n    # Apply scaling transformations\n    scaled_data = noisy_data * scale_factor\n\n    return scaled_data\n\n# --------------------- Synthetic Data Generation ---------------------\ndef generate_synthetic_data(X, num_samples):\n    \"\"\"Generate synthetic spectral data by sampling from the real data distribution.\"\"\"\n    synthetic_data = np.tile(X, (num_samples // X.shape[0] + 1, 1, 1))  # Replicate the original data\n    synthetic_data = augment_spectral_data(synthetic_data[:num_samples])  # Apply augmentations\n    return synthetic_data\n\n# --------------------- Data Preprocessing ---------------------\ndef preprocess_data(train_adc_info, test_adc_info, wavelengths, train_labels):\n    \"\"\"Preprocess the data and generate synthetic samples for training.\"\"\"\n    # Read CSV files\n    train_adc_info_df = pd.read_csv(train_adc_info)\n    test_adc_info_df = pd.read_csv(test_adc_info)\n    train_labels_df = pd.read_csv(train_labels)\n    wavelengths_df = pd.read_csv(wavelengths)\n\n    # Merge training data with labels\n    X_train = train_adc_info_df.merge(train_labels_df, on='planet_id')\n    y_train = X_train.iloc[:, 1:284].values  # Extract spectral data (mu)\n\n    # Drop planet_id and reshape the input data for the model\n    X_train_no_id = X_train.drop(columns=['planet_id'])\n    num_features = X_train_no_id.shape[1]\n    X_train = X_train_no_id.values.reshape(X_train.shape[0], num_features, 1)\n\n    # Apply data augmentation to the training data\n    X_train = augment_spectral_data(X_train)\n\n    # Generate synthetic data for training\n    num_synthetic_samples = 5000\n    synthetic_data = generate_synthetic_data(X_train, num_samples=num_synthetic_samples)\n    X_train = np.concatenate([X_train, synthetic_data], axis=0)\n\n    # Adjust y_train to match the size of augmented X_train\n    y_train = np.tile(y_train, ((X_train.shape[0] + y_train.shape[0] - 1) // y_train.shape[0], 1))\n    y_train = y_train[:X_train.shape[0]]\n\n    # Prepare the test data\n    test_data = test_adc_info_df.merge(wavelengths_df, how='cross')\n    test_data_no_id = test_data.drop(columns=['planet_id'])\n    num_features_test = test_data_no_id.shape[1]\n    test_data = test_data_no_id.values.reshape(test_data.shape[0], num_features_test, 1)\n\n    return X_train, y_train, test_data\n\n# --------------------- CNN Model Architecture ---------------------\ndef build_cnn_model(input_shape):\n    \"\"\"Build a CNN model with multimodal input to predict spectral data and uncertainties.\"\"\"\n    input_layer = layers.Input(shape=input_shape)\n\n    # Branch 1: For AIRS-CH0 sensor data\n    branch1 = layers.Conv1D(32, kernel_size=3, activation='relu')(input_layer)\n    branch1 = layers.Conv1D(64, kernel_size=3, activation='relu')(branch1)\n    branch1 = layers.GlobalAveragePooling1D()(branch1)\n\n    # Branch 2: For FGS1 sensor data\n    branch2 = layers.Conv1D(32, kernel_size=3, activation='relu')(input_layer)\n    branch2 = layers.Conv1D(64, kernel_size=3, activation='relu')(branch2)\n    branch2 = layers.GlobalAveragePooling1D()(branch2)\n\n    # Concatenate both branches\n    concatenated = layers.concatenate([branch1, branch2])\n\n    # Dense layers after concatenation\n    dense = layers.Dense(128, activation='relu')(concatenated)\n    dense = layers.Dense(64, activation='relu')(dense)\n\n    # Output layers: 283 values for predicted spectra (mean) + 283 for uncertainty (sigma)\n    output_mu = layers.Dense(283, activation='linear')(dense)  # Spectral mean prediction\n    output_sigma = layers.Dense(283, activation='softplus')(dense)  # Uncertainty prediction\n\n    # Combine mean and uncertainty into a single output\n    output = layers.concatenate([output_mu, output_sigma])\n\n    # Compile the model with GLL loss function\n    model = models.Model(inputs=input_layer, outputs=output)\n    model.compile(optimizer='adam', loss=gaussian_log_likelihood, metrics=['mae'])\n\n    return model\n\n# --------------------- Model Training ---------------------\ndef train_model(model, X_train, y_train, X_val, y_val):\n    \"\"\"Train the model using training and validation data.\"\"\"\n    # Prepare target data with dummy sigma values\n    y_train_sigma = np.zeros_like(y_train)\n    y_train_with_sigma = np.concatenate([y_train, y_train_sigma], axis=1)\n\n    # Prepare validation data similarly\n    y_val_sigma = np.zeros_like(y_val)\n    y_val_with_sigma = np.concatenate([y_val, y_val_sigma], axis=1)\n\n    # Train the model\n    history = model.fit(\n        X_train, y_train_with_sigma,\n        validation_data=(X_val, y_val_with_sigma),\n        epochs=50, batch_size=32, verbose=1\n    )\n    return history\n\n# --------------------- Metrics Plotting ---------------------\ndef plot_metrics(history):\n    \"\"\"Plot training and validation loss and MAE over epochs.\"\"\"\n    plt.figure(figsize=(12, 4))\n\n    # Plot loss\n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['loss'], label='Train Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Epoch Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n\n    # Plot Mean Absolute Error (MAE)\n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['mae'], label='Train MAE')\n    plt.plot(history.history['val_mae'], label='Validation MAE')\n    plt.title('Mean Absolute Error')\n    plt.xlabel('Epoch')\n    plt.ylabel('MAE')\n    plt.legend()\n\n    plt.show()\n\n# --------------------- Confusion Matrix Plot ---------------------\ndef plot_confusion_matrix(y_true, y_pred):\n    \"\"\"Plot confusion matrix for true vs predicted labels.\"\"\"\n    cm = confusion_matrix(y_true, np.argmax(y_pred, axis=1))\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\n    plt.title(\"Confusion Matrix\")\n    plt.xlabel(\"Predicted\")\n    plt.ylabel(\"True\")\n    plt.show()\n\n# --------------------- ROC Curve Plot ---------------------\ndef plot_roc_curve(y_true, y_pred):\n    \"\"\"Plot ROC curve for binary classification.\"\"\"\n    fpr, tpr, thresholds = roc_curve(y_true, np.argmax(y_pred, axis=1))\n    roc_auc = auc(fpr, tpr)\n\n    plt.figure(figsize=(8, 6))\n    plt.plot(fpr, tpr, label=f'ROC curve (AUC = {roc_auc:.2f})')\n    plt.plot([0, 1], [0, 1], 'k--', lw=2)\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Receiver Operating Characteristic (ROC)')\n    plt.legend(loc='lower right')\n    plt.show()\n\n# --------------------- Submission Generation ---------------------\ndef generate_submission(model, test_data, submission_template, output_file):\n    \"\"\"Generate predictions and save them in the required submission format.\"\"\"\n    predictions = model.predict(test_data)\n\n    # Split predictions into spectral mean and uncertainty\n    pred_mu = predictions[:, :283]\n    pred_sigma = predictions[:, 283:]\n\n    # Load the sample submission template\n    submission_df = pd.read_csv(submission_template)\n\n    # Replace the submission template columns with predicted values\n    submission_df.iloc[:, 1:284] = pred_mu\n    submission_df.iloc[:, 284:] = pred_sigma\n\n    # Save the predictions to the submission file\n    submission_df.to_csv(output_file, index=False)\n\n# --------------------- Main Pipeline ---------------------\ndef main():\n    # Load and preprocess data\n    train_adc_info = '/kaggle/input/ariel-data-challenge-2024/train_adc_info.csv'\n    test_adc_info = '/kaggle/input/ariel-data-challenge-2024/test_adc_info.csv'\n    wavelengths = '/kaggle/input/ariel-data-challenge-2024/wavelengths.csv'\n    train_labels = '/kaggle/input/ariel-data-challenge-2024/train_labels.csv'\n\n    # Preprocess data, augment and handle synthetic data generation\n    X_train, y_train, test_data = preprocess_data(train_adc_info, test_adc_info, wavelengths, train_labels)\n\n    # Ensure y_train matches X_train size after augmentation\n    num_synthetic_samples = X_train.shape[0] - y_train.shape[0]\n    if num_synthetic_samples > 0:\n        y_train = np.tile(y_train, (num_synthetic_samples // y_train.shape[0] + 1, 1))[:X_train.shape[0]]\n\n    # Split data into train and validation sets\n    X_train, X_val, y_train, y_val = train_test_split(X_train, y_train, test_size=0.2, random_state=42)\n\n    # Build the CNN model\n    model = build_cnn_model(input_shape=(X_train.shape[1], X_train.shape[2]))\n\n    # Train the model\n    history = train_model(model, X_train, y_train, X_val, y_val)\n\n    # Plot training metrics\n    plot_metrics(history)\n\n    # Generate predictions and save submission file\n    submission_template = '/kaggle/input/ariel-data-challenge-2024/sample_submission.csv'\n    output_file = '/kaggle/working/submission.csv'\n    generate_submission(model, test_data, submission_template, output_file)\n\n    pd.read_csv(output_file)  # Load the submission file for review\n\nif __name__ == \"__main__\":\n    main()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-16T11:56:13.458653Z","iopub.execute_input":"2024-10-16T11:56:13.459052Z","iopub.status.idle":"2024-10-16T11:58:41.627068Z","shell.execute_reply.started":"2024-10-16T11:56:13.459014Z","shell.execute_reply":"2024-10-16T11:58:41.625831Z"},"trusted":true},"execution_count":null,"outputs":[]}]}