{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"sourceType":"competition"},{"sourceId":9629432,"sourceType":"datasetVersion","datasetId":5846888}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**References and Acknowledgments**\nI would like to thank all the Kaggle community members whose work has inspired and helped me improve this notebook. I’ve referenced and built upon ideas from various public kernels—special thanks to those whose notebooks provided valuable insights. I’ve made every effort to understand, adapt, and enhance the approaches shared.\n\nCredit to the original authors for their excellent contributions. Links to referenced works are included in the notebook for transparency and further learning.\n\nThank you all for sharing your knowledge and helping others grow.\n\nhttps://www.kaggle.com/code/mpwolke/ariel-2025-exoplanets-chemistry\n\nhttps://www.kaggle.com/competitions/ariel-data-challenge-2024/code\n\nhttps://www.kaggle.com/code/vitalykudelya/neurips-non-ml-transit-curve-fitting/notebook?scriptVersionId=250419499\n","metadata":{}},{"cell_type":"markdown","source":"## Install dependencies","metadata":{}},{"cell_type":"code","source":"!pip install optuna\n!pip install shap\n# install pqdm for parallel processing\n!pip install --no-index --find-links=/kaggle/input/ariel-2024-pqdm pqdm","metadata":{"trusted":true,"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:09.246629Z","iopub.execute_input":"2025-09-12T08:41:09.247349Z","iopub.status.idle":"2025-09-12T08:41:21.617115Z","shell.execute_reply.started":"2025-09-12T08:41:09.247306Z","shell.execute_reply":"2025-09-12T08:41:21.61566Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Packages","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport glob\n\nimport pyarrow.parquet as pq\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore', category=DeprecationWarning)\n\nfrom tqdm import tqdm\nfrom pqdm.threads import pqdm\nimport itertools\nimport pickle\n\nfrom scipy.optimize import minimize\nfrom sklearn.metrics import mean_squared_error\n\nimport plotly.express as px\n\nfrom astropy.stats import sigma_clip\nfrom scipy.signal import savgol_filter   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:21.61959Z","iopub.execute_input":"2025-09-12T08:41:21.619943Z","iopub.status.idle":"2025-09-12T08:41:21.628253Z","shell.execute_reply.started":"2025-09-12T08:41:21.619908Z","shell.execute_reply":"2025-09-12T08:41:21.626982Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Datasets","metadata":{}},{"cell_type":"code","source":"# Load metadata CSV files\nadc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/adc_info.csv')\naxis_info = pq.read_table('/kaggle/input/ariel-data-challenge-2025/axis_info.parquet').to_pandas()\ntrain_star_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train_star_info.csv')\ntest_star_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/test_star_info.csv')\nwavelengths = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/wavelengths.csv')\nsample_submission = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/sample_submission.csv')\n\n# Optional: Ground truth for training\ntrain_df= pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:21.629414Z","iopub.execute_input":"2025-09-12T08:41:21.629702Z","iopub.status.idle":"2025-09-12T08:41:21.808791Z","shell.execute_reply.started":"2025-09-12T08:41:21.62967Z","shell.execute_reply":"2025-09-12T08:41:21.807965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"adc_info.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:21.809686Z","iopub.execute_input":"2025-09-12T08:41:21.810023Z","iopub.status.idle":"2025-09-12T08:41:21.82311Z","shell.execute_reply.started":"2025-09-12T08:41:21.81Z","shell.execute_reply":"2025-09-12T08:41:21.821921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_star_info.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:21.825959Z","iopub.execute_input":"2025-09-12T08:41:21.82636Z","iopub.status.idle":"2025-09-12T08:41:21.850647Z","shell.execute_reply.started":"2025-09-12T08:41:21.826324Z","shell.execute_reply":"2025-09-12T08:41:21.849344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_star_info.head(6)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:21.851471Z","iopub.execute_input":"2025-09-12T08:41:21.851731Z","iopub.status.idle":"2025-09-12T08:41:21.880243Z","shell.execute_reply.started":"2025-09-12T08:41:21.851711Z","shell.execute_reply":"2025-09-12T08:41:21.878955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wavelengths.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:21.881302Z","iopub.execute_input":"2025-09-12T08:41:21.881562Z","iopub.status.idle":"2025-09-12T08:41:21.916071Z","shell.execute_reply.started":"2025-09-12T08:41:21.881544Z","shell.execute_reply":"2025-09-12T08:41:21.914928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:21.91725Z","iopub.execute_input":"2025-09-12T08:41:21.917535Z","iopub.status.idle":"2025-09-12T08:41:21.949323Z","shell.execute_reply.started":"2025-09-12T08:41:21.917513Z","shell.execute_reply":"2025-09-12T08:41:21.948154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:21.950634Z","iopub.execute_input":"2025-09-12T08:41:21.951009Z","iopub.status.idle":"2025-09-12T08:41:21.9861Z","shell.execute_reply.started":"2025-09-12T08:41:21.950972Z","shell.execute_reply":"2025-09-12T08:41:21.985105Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load ADC Calibration Info","metadata":{}},{"cell_type":"code","source":"adc_info = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/adc_info.csv\")\n\n# Extract AIRS-CH0 gain and offset\ninstrument = \"AIRS-CH0\"\ngain = adc_info[f\"{instrument}_adc_gain\"].iloc[0]\noffset = adc_info[f\"{instrument}_adc_offset\"].iloc[0]\n\nprint(f\"{instrument} gain = {gain}, offset = {offset}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:21.986945Z","iopub.execute_input":"2025-09-12T08:41:21.987245Z","iopub.status.idle":"2025-09-12T08:41:22.01238Z","shell.execute_reply.started":"2025-09-12T08:41:21.987223Z","shell.execute_reply":"2025-09-12T08:41:22.01097Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load and Calibrate Raw Detector Readings","metadata":{}},{"cell_type":"code","source":"%%time\n# Path to the raw detector data\nread_path = \"/kaggle/input/ariel-data-challenge-2025/train/1010375142/AIRS-CH0_calibration_0/read.parquet\"\n\n# Load the raw ADC signal\nraw_df = pd.read_parquet(read_path)\nprint(\"Raw shape:\", raw_df.shape)\nprint(raw_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:22.013629Z","iopub.execute_input":"2025-09-12T08:41:22.013994Z","iopub.status.idle":"2025-09-12T08:41:22.063008Z","shell.execute_reply.started":"2025-09-12T08:41:22.013967Z","shell.execute_reply":"2025-09-12T08:41:22.061693Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Sharp","metadata":{}},{"cell_type":"code","source":"%%time\nimport pandas as pd\n\n# Set your dataset path (example: one PartQuest ID folder from train set)\ndataset_path = \"/kaggle/input/ariel-data-challenge-2025/train/1010375142/AIRS-CH0_calibration_0/\"\n\n# Load the read.parquet file\nairs_data = pd.read_parquet(dataset_path + \"read.parquet\")\n\n# Print basic info\nprint(\"Shape of airs_data:\", airs_data.shape)\nairs_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:22.064129Z","iopub.execute_input":"2025-09-12T08:41:22.064417Z","iopub.status.idle":"2025-09-12T08:41:22.109814Z","shell.execute_reply.started":"2025-09-12T08:41:22.064395Z","shell.execute_reply":"2025-09-12T08:41:22.108582Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Apply Calibration (Gain + Offset)","metadata":{}},{"cell_type":"code","source":"# Apply calibration\ncorrected_df = raw_df * gain + offset\n\n# Display summary\nprint(corrected_df.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:22.111058Z","iopub.execute_input":"2025-09-12T08:41:22.111361Z","iopub.status.idle":"2025-09-12T08:41:22.58546Z","shell.execute_reply.started":"2025-09-12T08:41:22.11134Z","shell.execute_reply":"2025-09-12T08:41:22.58424Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3D orbit animation","metadata":{}},{"cell_type":"markdown","source":"###  with Plotly","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport plotly.graph_objects as go\n\n# Load the star-planet system info\nstar_info = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/train_star_info.csv\")\n\n# Select a system\nrow = star_info.iloc[0]\nplanet_id = row[\"planet_id\"]\n\n# Extract parameters\nRs = row[\"Rs\"]                    # Star radius (solar radii)\nsma = row[\"sma\"]                  # Semi-major axis (AU)\ne = row[\"e\"]                      # Eccentricity\ni_deg = row[\"i\"]                  # Inclination (degrees)\ni_rad = np.radians(i_deg)\n\n# Convert sma from AU to solar radii\nAU_TO_SOLAR_RADII = 215.032\nsma_rsun = sma * AU_TO_SOLAR_RADII\n\n# Orbital ellipse in polar coordinates\ntheta = np.linspace(0, 2 * np.pi, 500)\nr = sma_rsun * (1 - e**2) / (1 + e * np.cos(theta))\n\n# Coordinates in orbital plane\nx_orb = r * np.cos(theta)\ny_orb = r * np.sin(theta)\nz_orb = np.zeros_like(theta)\n\n# Rotate by inclination (around x-axis)\ny_inc = y_orb * np.cos(i_rad)\nz_inc = y_orb * np.sin(i_rad)\n\n# 3D plot\nfig = go.Figure()\n\n# Star at origin\nfig.add_trace(go.Scatter3d(\n    x=[0], y=[0], z=[0],\n    mode='markers',\n    marker=dict(size=8, color='yellow'),\n    name='Star'\n))\n\n# Planet orbit\nfig.add_trace(go.Scatter3d(\n    x=x_orb, y=y_inc, z=z_inc,\n    mode='lines',\n    line=dict(color='deepskyblue'),\n    name='Orbit'\n))\n\n# Planet position (e.g., at periapsis)\nfig.add_trace(go.Scatter3d(\n    x=[x_orb[0]], y=[y_inc[0]], z=[z_inc[0]],\n    mode='markers',\n    marker=dict(size=4, color='blue'),\n    name='Planet (start)'\n))\n\n# Layout\nfig.update_layout(\n    title=f\"3D Orbit of Planet {planet_id} (Inclination: {i_deg:.1f}°)\",\n    scene=dict(\n        xaxis_title='X (solar radii)',\n        yaxis_title='Y (inclined)',\n        zaxis_title='Z (inclined)',\n        aspectmode='data',\n    ),\n    margin=dict(l=0, r=0, b=0, t=50)\n)\n\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:22.58865Z","iopub.execute_input":"2025-09-12T08:41:22.588981Z","iopub.status.idle":"2025-09-12T08:41:22.62341Z","shell.execute_reply.started":"2025-09-12T08:41:22.588956Z","shell.execute_reply":"2025-09-12T08:41:22.622172Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### with Matplotlib","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib.animation import FuncAnimation\nfrom mpl_toolkits.mplot3d import Axes3D\n\n# Select a planet system (first row for demo)\nplanet = star_info.iloc[0]\n\n# Parameters\nsma = planet[\"sma\"]        # semi-major axis in stellar radii\ne = planet[\"e\"]            # eccentricity\ni = np.radians(planet[\"i\"])  # inclination in radians\nP = planet[\"P\"]            # orbital period (not needed here unless time scaling)\nRs = planet[\"Rs\"]          # star radius\n\n# Compute orbit (Keplerian ellipse in x-y plane, rotate for inclination)\ntheta = np.linspace(0, 2 * np.pi, 200)\nr = (sma * (1 - e**2)) / (1 + e * np.cos(theta))  # polar equation of ellipse\n\n# Convert to 3D Cartesian\nx = r * np.cos(theta)\ny = r * np.sin(theta)\nz = y * np.sin(i)\ny = y * np.cos(i)  # adjust y after inclination\n\n# Set up 3D plot\nfig = plt.figure(figsize=(8, 6))\nax = fig.add_subplot(111, projection='3d')\nax.set_box_aspect([1,1,1])\nax.set_xlim(-sma*1.2, sma*1.2)\nax.set_ylim(-sma*1.2, sma*1.2)\nax.set_zlim(-sma*1.2, sma*1.2)\nax.set_title(\"Animated Star-Planet System\")\n\n# Star\nax.scatter(0, 0, 0, color='yellow', s=300, label='Star')\n\n# Static orbit\nax.plot(x, y, z, color='gray', linestyle='--', alpha=0.5)\n\n# Planet dot (animated)\nplanet_dot, = ax.plot([], [], [], 'o', color='blue', markersize=10, label='Planet')\n\ndef update(frame):\n    planet_dot.set_data(x[frame], y[frame])\n    planet_dot.set_3d_properties(z[frame])\n    return planet_dot,\n\nani = FuncAnimation(fig, update, frames=len(theta), interval=50, blit=True)\n\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:22.624713Z","iopub.execute_input":"2025-09-12T08:41:22.624993Z","iopub.status.idle":"2025-09-12T08:41:22.929172Z","shell.execute_reply.started":"2025-09-12T08:41:22.624973Z","shell.execute_reply":"2025-09-12T08:41:22.927911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualization & Insight","metadata":{}},{"cell_type":"code","source":"trail_length = 30  # frames\n\ndef update(frame):\n    start = max(0, frame - trail_length)\n    planet_dot.set_data(x[frame], y[frame])\n    planet_dot.set_3d_properties(z[frame])\n    ax.plot(x[start:frame], y[start:frame], z[start:frame], color='blue', alpha=0.4)\n    return planet_dot,","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:22.930224Z","iopub.execute_input":"2025-09-12T08:41:22.930517Z","iopub.status.idle":"2025-09-12T08:41:22.936657Z","shell.execute_reply.started":"2025-09-12T08:41:22.930494Z","shell.execute_reply":"2025-09-12T08:41:22.935571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from matplotlib.cm import plasma\ntemp_norm = (planet[\"Ts\"] - 3000) / (7000 - 3000)  # normalize temp to [0,1]\nstar_color = plasma(temp_norm)\nax.scatter(0, 0, 0, color=star_color, s=300)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:22.93761Z","iopub.execute_input":"2025-09-12T08:41:22.93792Z","iopub.status.idle":"2025-09-12T08:41:22.96344Z","shell.execute_reply.started":"2025-09-12T08:41:22.937899Z","shell.execute_reply":"2025-09-12T08:41:22.962348Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualize a Few Detector Pixels Over Time (Optional)","metadata":{}},{"cell_type":"code","source":"# Plot a few pixel time series\nnum_pixels_to_plot = 5\nplt.figure(figsize=(12, 5))\n\nfor i in range(num_pixels_to_plot):\n    plt.plot(corrected_df.iloc[i], label=f'Pixel {i}')\n\nplt.title(f'{instrument} Calibrated Signal (First {num_pixels_to_plot} Pixels)')\nplt.xlabel('Time')\nplt.ylabel('Calibrated Signal')\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:22.964363Z","iopub.execute_input":"2025-09-12T08:41:22.964682Z","iopub.status.idle":"2025-09-12T08:41:25.238948Z","shell.execute_reply.started":"2025-09-12T08:41:22.964659Z","shell.execute_reply":"2025-09-12T08:41:25.23795Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualization Code (for AIRS-CH0)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# 1. Load gain and offset for AIRS-CH0\nadc_info = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/adc_info.csv\")\ninstrument = \"AIRS-CH0\"\ngain = adc_info[f\"{instrument}_adc_gain\"].iloc[0]\noffset = adc_info[f\"{instrument}_adc_offset\"].iloc[0]\nprint(f\"{instrument} gain = {gain}, offset = {offset}\")\n\n# 2. Load AIRS-CH0 signal file\nsignal_path = \"/kaggle/input/ariel-data-challenge-2025/train/1253730513/AIRS-CH0_signal_0.parquet\"\nsignal_df = pd.read_parquet(signal_path)\n\n# 3. Apply gain and offset correction\nsignal_corrected = signal_df.values.astype(np.float32) * gain + offset\n\n# 4. Reshape to (time, 32, 356)\nsignal_reshaped = signal_corrected.reshape(-1, 32, 356)\n\n# 5. Plot a few frames\nn_frames_to_plot = 3\nfig, axes = plt.subplots(1, n_frames_to_plot, figsize=(15, 5))\n\nfor i in range(n_frames_to_plot):\n    axes[i].imshow(signal_reshaped[i], cmap='viridis', aspect='auto')\n    axes[i].set_title(f\"Frame {i}\")\n    axes[i].axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:25.240208Z","iopub.execute_input":"2025-09-12T08:41:25.240535Z","iopub.status.idle":"2025-09-12T08:41:27.156478Z","shell.execute_reply.started":"2025-09-12T08:41:25.240512Z","shell.execute_reply":"2025-09-12T08:41:27.155432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"Feature Enginner inherit to test","metadata":{}},{"cell_type":"code","source":"%%time\ndef feature_engineering(f_raw, a_raw):\n    \"\"\"Create a dataframe with two features from the raw data.\n    \n    Parameters:\n    f_raw: ndarray of shape (n_planets, 67500)\n    a_raw: ndarray of shape (n_planets, 5625)\n    \n    Return value:\n    df: DataFrame of shape (n_planets, 2)\n    \"\"\"\n    # Feature from f_raw: broadband flux\n    f_obscured = f_raw[:, 23500:44000].mean(axis=1)  # Transit/eclipse in flux\n    f_unobscured = (f_raw[:, :20500].mean(axis=1) + f_raw[:, 47000:].mean(axis=1)) / 2\n    f_relative_reduction = (f_unobscured - f_obscured) / f_unobscured\n\n    # Feature from a_raw: spectroscopic data\n    a_obscured = a_raw[:, 1958:3666].mean(axis=1)  # Corresponding occultation window\n    a_unobscured = (a_raw[:, :1708].mean(axis=1) + a_raw[:, 3916:].mean(axis=1)) / 2\n    a_relative_reduction = (a_unobscured - a_obscured) / a_unobscured\n\n    # Create DataFrame with both features\n    df = pd.DataFrame({\n        'a_relative_reduction': a_relative_reduction,\n        'f_relative_reduction': f_relative_reduction\n    })\n    \n    return df   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.157641Z","iopub.execute_input":"2025-09-12T08:41:27.158113Z","iopub.status.idle":"2025-09-12T08:41:27.168214Z","shell.execute_reply.started":"2025-09-12T08:41:27.158068Z","shell.execute_reply":"2025-09-12T08:41:27.166834Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Extracting features for one planet","metadata":{}},{"cell_type":"code","source":"%%time\ndef build_feature_vector(planet_id):\n    # Get spectrum from train.csv\n    spec_row = train_df[train_df[\"planet_id\"] == planet_id]\n    if spec_row.empty:\n        print(f\"[!] planet_id {planet_id} not found in train.csv\")\n        return None\n\n    # Get star/planet metadata\n    phys_row = star_info[star_info[\"planet_id\"] == planet_id]\n    if phys_row.empty:\n        print(f\"[!] planet_id {planet_id} not found in star_info.csv\")\n        return None\n\n    # Drop planet_id to avoid duplication\n    spectrum = spec_row.drop(columns=[\"planet_id\"]).reset_index(drop=True)\n    phys = phys_row.drop(columns=[\"planet_id\"]).reset_index(drop=True)\n\n    # Combine\n    combined = pd.concat([spectrum, phys], axis=1)\n    combined[\"planet_id\"] = planet_id  # Optional: keep ID\n\n    return combined","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.169709Z","iopub.execute_input":"2025-09-12T08:41:27.170132Z","iopub.status.idle":"2025-09-12T08:41:27.195957Z","shell.execute_reply.started":"2025-09-12T08:41:27.170096Z","shell.execute_reply":"2025-09-12T08:41:27.19455Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Training Preparation","metadata":{}},{"cell_type":"markdown","source":"### Full Feature Matrix With Labels","metadata":{}},{"cell_type":"code","source":"%%time\n\ndef build_full_training_data():\n    feature_rows = []\n    for idx, row in train_df.iterrows():\n        planet_id = row[\"planet_id\"]\n        features = build_feature_vector(planet_id)\n        if features is not None:\n            # Add label\n            features[\"wl_276\"] = row[\"wl_276\"]\n            feature_rows.append(features)\n\n    return pd.concat(feature_rows, axis=0).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.197012Z","iopub.execute_input":"2025-09-12T08:41:27.197315Z","iopub.status.idle":"2025-09-12T08:41:27.219603Z","shell.execute_reply.started":"2025-09-12T08:41:27.197294Z","shell.execute_reply":"2025-09-12T08:41:27.218182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ndef build_full_training_data():\n    feature_rows = []\n    for idx, row in train_df.iterrows():\n        planet_id = row[\"planet_id\"]\n        \n        # Load f_raw and a_raw for this planet\n        f_raw = np.load(f'/kaggle/input/ariel-data-challenge-2025/train/f/{planet_id}.npy')\n        a_raw = np.load(f'/kaggle/input/ariel-data-challenge-2025/train/a/{planet_id}.npy')\n        \n        # Expand dimensions if needed (if single sample)\n        f_raw = f_raw.reshape(1, -1)  # Shape: (1, 67500)\n        a_raw = a_raw.reshape(1, -1)  # Shape: (1, 5625)\n        \n        # Use your feature_engineering function\n        features_df = feature_engineering(f_raw, a_raw)\n        \n        # Add label\n        features_df[\"wl_276\"] = row[\"wl_276\"]\n        \n        feature_rows.append(features_df)\n    \n    # Concatenate all feature rows\n    return pd.concat(feature_rows, axis=0).reset_index(drop=True)   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.22068Z","iopub.execute_input":"2025-09-12T08:41:27.220978Z","iopub.status.idle":"2025-09-12T08:41:27.249111Z","shell.execute_reply.started":"2025-09-12T08:41:27.220956Z","shell.execute_reply":"2025-09-12T08:41:27.247855Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train/Test Split","metadata":{}},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"%%time\nclass Config:\n    # === Paths and Dataset Mode ===\n    DATA_PATH = '/kaggle/input/ariel-data-challenge-2025'\n    DATASET = 'test'  # Change to 'train' for training runs\n\n    # === Signal Processing Parameters ===\n    CUT_INF = 39\n    CUT_SUP = 250\n    N_JOBS = 4  # Parallel processing workers\n\n    # === Model Scaling and Uncertainty ===\n    SCALE = 0.95\n    SIGMA = 0.0009  # Assumed constant uncertainty\n\n    # === Transit Detection Settings ===\n    MODEL_PHASE_DETECTION_SLICE = slice(30, 140)\n    MODEL_OPTIMIZATION_DELTA = 7   # Guard band around transit\n    MODEL_POLYNOMIAL_DEGREE = 3    # Baseline fit degree\n\n    # === Sensor Configuration (Computed Safely) ===\n    @property\n    def SENSOR_CONFIG(self):\n        CUT_SUP = self.CUT_SUP\n        CUT_INF = self.CUT_INF\n\n        return {\n            \"AIRS-CH0\": {\n                \"raw_shape\": [11250, 32, 356],\n                \"calibrated_shape\": [1, 32, CUT_SUP - CUT_INF],\n                \"linear_corr_shape\": (6, 32, 356),\n                \"dt_pattern\": (0.1, 4.5),\n                \"binning\": 30\n            },\n            \"FGS1\": {\n                \"raw_shape\": [135000, 32, 32],\n                \"calibrated_shape\": [1, 32, 32],\n                \"linear_corr_shape\": (6, 32, 32),\n                \"dt_pattern\": (0.1, 0.1),\n                \"binning\": 30 * 12  # 360 frames → 1 binned point\n            }\n        }\n\n    # === Convenience Properties ===\n    @property\n    def is_training(self):\n        \"\"\"Check if current dataset is 'train'.\"\"\"\n        return self.DATASET.lower() == 'train'\n\n    @property\n    def sample_submission_path(self):\n        return f\"{self.DATA_PATH}/sample_submission.csv\"\n\n    @property\n    def star_info_path(self):\n        return f\"{self.DATA_PATH}/{self.DATASET}_star_info.csv\"\n\n    # Optional: Add debug mode\n    @property\n    def DEBUG(self):\n        return False  # Set to True for fast testing on small subset   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.250118Z","iopub.execute_input":"2025-09-12T08:41:27.25041Z","iopub.status.idle":"2025-09-12T08:41:27.273185Z","shell.execute_reply.started":"2025-09-12T08:41:27.250385Z","shell.execute_reply":"2025-09-12T08:41:27.271918Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Signal Processing","metadata":{}},{"cell_type":"code","source":"%%time\n\nclass SignalProcessor:\n    def __init__(self, config):\n        self.cfg = config\n        self.adc_info = pd.read_csv(f\"{self.cfg.DATA_PATH}/adc_info.csv\")\n        self.planet_ids = pd.read_csv(f'{self.cfg.DATA_PATH}/{self.cfg.DATASET}_star_info.csv', index_col='planet_id').index.astype(int)\n\n    def _apply_linear_corr(self, linear_corr, signal):\n        linear_corr_flipped = np.flip(linear_corr, axis=0)\n        corrected_signal = signal.copy()\n        \n        for x, y in itertools.product(range(signal.shape[1]), range(signal.shape[2])):\n            poly = np.poly1d(linear_corr_flipped[:, x, y])\n            corrected_signal[:, x, y] = poly(corrected_signal[:, x, y])\n            \n        return corrected_signal\n\n    def _calibrate_single_signal(self, planet_id, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n\n        signal = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_signal_0.parquet\").to_numpy()\n        dark = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dark.parquet\").to_numpy()\n        dead = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dead.parquet\").to_numpy()\n        flat = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/flat.parquet\").to_numpy()\n        linear_corr = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/linear_corr.parquet\").values.astype(np.float64).reshape(sensor_cfg[\"linear_corr_shape\"])\n\n        signal = signal.reshape(sensor_cfg[\"raw_shape\"])\n        gain = self.adc_info[f\"{sensor}_adc_gain\"].iloc[0]\n        offset = self.adc_info[f\"{sensor}_adc_offset\"].iloc[0]\n        signal = signal / gain + offset\n\n        hot = sigma_clip(dark, sigma=5, maxiters=5).mask\n\n        if sensor == \"AIRS-CH0\":\n            signal = signal[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            linear_corr = linear_corr[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dark = dark[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dead = dead[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            flat = flat[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            hot = hot[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n        \n        base_dt, increment = sensor_cfg[\"dt_pattern\"]\n        dt = np.ones(len(signal)) * base_dt\n        dt[1::2] += increment\n        \n        signal = signal.clip(0)\n        signal = self._apply_linear_corr(linear_corr, signal)\n        signal -= dark * dt[:, np.newaxis, np.newaxis]\n        \n        flat = flat.reshape(sensor_cfg[\"calibrated_shape\"])\n        flat[dead.reshape(sensor_cfg[\"calibrated_shape\"])] = np.nan\n        flat[hot.reshape(sensor_cfg[\"calibrated_shape\"])] = np.nan\n        \n        signal = signal / flat\n        \n        return signal\n\n    def _preprocess_calibrated_signal(self, calibrated_signal, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n        binning = sensor_cfg[\"binning\"]\n\n        if sensor == \"AIRS-CH0\":\n            signal_roi = calibrated_signal[:, 10:22, :]\n        elif sensor == \"FGS1\":\n            signal_roi = calibrated_signal[:, 10:22, 10:22]\n            signal_roi = signal_roi.reshape(signal_roi.shape[0], -1)\n        \n        mean_signal = np.nanmean(signal_roi, axis=1)\n\n        cds_signal = mean_signal[1::2] - mean_signal[0::2]\n\n        n_bins = cds_signal.shape[0] // binning\n        binned = np.array([\n            cds_signal[j*binning : (j+1)*binning].mean(axis=0) \n            for j in range(n_bins)\n        ])\n\n        if sensor == \"FGS1\":\n            binned = binned.reshape((binned.shape[0], 1))\n            \n        return binned\n\n    def _process_planet_sensor(self, args):\n        planet_id, sensor = args['planet_id'], args['sensor']\n        calibrated = self._calibrate_single_signal(planet_id, sensor)\n        preprocessed = self._preprocess_calibrated_signal(calibrated, sensor)\n        return preprocessed\n\n    def process_all_data(self):\n        args_fgs1 = [dict(planet_id=planet_id, sensor=\"FGS1\") for planet_id in self.planet_ids]\n        preprocessed_fgs1 = pqdm(args_fgs1, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        args_airs_ch0 = [dict(planet_id=planet_id, sensor=\"AIRS-CH0\") for planet_id in self.planet_ids]\n        preprocessed_airs_ch0 = pqdm(args_airs_ch0, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        preprocessed_signal = np.concatenate(\n            [np.stack(preprocessed_fgs1), np.stack(preprocessed_airs_ch0)], axis=2\n        )\n        return preprocessed_signal \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.274271Z","iopub.execute_input":"2025-09-12T08:41:27.274542Z","iopub.status.idle":"2025-09-12T08:41:27.306455Z","shell.execute_reply.started":"2025-09-12T08:41:27.274521Z","shell.execute_reply":"2025-09-12T08:41:27.305391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nclass SignalProcessor:\n    def __init__(self, config):\n        self.cfg = config\n        self.adc_info = pd.read_csv(f\"{self.cfg.DATA_PATH}/adc_info.csv\")\n        self.planet_ids = pd.read_csv(f'{self.cfg.DATA_PATH}/{self.cfg.DATASET}_star_info.csv', index_col='planet_id').index.astype(int)\n\n    def _apply_linear_corr(self, linear_corr, signal):\n        linear_corr_flipped = np.flip(linear_corr, axis=0)\n        corrected_signal = signal.copy()\n        \n        for x, y in itertools.product(range(signal.shape[1]), range(signal.shape[2])):\n            poly = np.poly1d(linear_corr_flipped[:, x, y])\n            corrected_signal[:, x, y] = poly(corrected_signal[:, x, y])\n            \n        return corrected_signal\n\n    def _calibrate_single_signal(self, planet_id, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n\n        signal = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_signal_0.parquet\").to_numpy()\n        dark = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dark.parquet\").to_numpy()\n        dead = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/dead.parquet\").to_numpy()\n        flat = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/flat.parquet\").to_numpy()\n        linear_corr = pd.read_parquet(f\"{self.cfg.DATA_PATH}/{self.cfg.DATASET}/{planet_id}/{sensor}_calibration_0/linear_corr.parquet\").values.astype(np.float64).reshape(sensor_cfg[\"linear_corr_shape\"])\n\n        signal = signal.reshape(sensor_cfg[\"raw_shape\"])\n        gain = self.adc_info[f\"{sensor}_adc_gain\"].iloc[0]\n        offset = self.adc_info[f\"{sensor}_adc_offset\"].iloc[0]\n        signal = signal / gain + offset\n\n        hot = sigma_clip(dark, sigma=5, maxiters=5).mask\n\n        if sensor == \"AIRS-CH0\":\n            signal = signal[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            linear_corr = linear_corr[:, :, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dark = dark[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            dead = dead[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            flat = flat[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n            hot = hot[:, self.cfg.CUT_INF : self.cfg.CUT_SUP]\n        \n        base_dt, increment = sensor_cfg[\"dt_pattern\"]\n        dt = np.ones(len(signal)) * base_dt\n        dt[1::2] += increment\n        \n        signal = signal.clip(0)\n        signal = self._apply_linear_corr(linear_corr, signal)\n        signal -= dark * dt[:, np.newaxis, np.newaxis]\n        \n        flat = flat.reshape(sensor_cfg[\"calibrated_shape\"])\n        flat[dead.reshape(sensor_cfg[\"calibrated_shape\"])] = np.nan\n        flat[hot.reshape(sensor_cfg[\"calibrated_shape\"])] = np.nan\n        \n        signal = signal / flat\n        \n        return signal\n\n    def _preprocess_calibrated_signal(self, calibrated_signal, sensor):\n        sensor_cfg = self.cfg.SENSOR_CONFIG[sensor]\n        binning = sensor_cfg[\"binning\"]\n\n        if sensor == \"AIRS-CH0\":\n            signal_roi = calibrated_signal[:, 10:22, :]\n        elif sensor == \"FGS1\":\n            signal_roi = calibrated_signal[:, 10:22, 10:22]\n            signal_roi = signal_roi.reshape(signal_roi.shape[0], -1)\n        \n        mean_signal = np.nanmean(signal_roi, axis=1)\n\n        cds_signal = mean_signal[1::2] - mean_signal[0::2]\n\n        n_bins = cds_signal.shape[0] // binning\n        binned = np.array([\n            cds_signal[j*binning : (j+1)*binning].mean(axis=0) \n            for j in range(n_bins)\n        ])\n\n        if sensor == \"FGS1\":\n            binned = binned.reshape((binned.shape[0], 1))\n            \n        return binned\n\n    def _process_planet_sensor(self, args):\n        planet_id, sensor = args['planet_id'], args['sensor']\n        calibrated = self._calibrate_single_signal(planet_id, sensor)\n        preprocessed = self._preprocess_calibrated_signal(calibrated, sensor)\n        return preprocessed\n\n    def process_all_data(self):\n        args_fgs1 = [dict(planet_id=planet_id, sensor=\"FGS1\") for planet_id in self.planet_ids]\n        preprocessed_fgs1 = pqdm(args_fgs1, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        args_airs_ch0 = [dict(planet_id=planet_id, sensor=\"AIRS-CH0\") for planet_id in self.planet_ids]\n        preprocessed_airs_ch0 = pqdm(args_airs_ch0, self._process_planet_sensor, n_jobs=self.cfg.N_JOBS)\n\n        preprocessed_signal = np.concatenate(\n            [np.stack(preprocessed_fgs1), np.stack(preprocessed_airs_ch0)], axis=2\n        )\n        return preprocessed_signal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.30745Z","iopub.execute_input":"2025-09-12T08:41:27.307719Z","iopub.status.idle":"2025-09-12T08:41:27.340716Z","shell.execute_reply.started":"2025-09-12T08:41:27.3077Z","shell.execute_reply":"2025-09-12T08:41:27.33959Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Model","metadata":{}},{"cell_type":"code","source":"%%time\nclass TransitModel:\n    def __init__(self, config):\n        self.cfg = config\n\n    def _phase_detector(self, signal):\n        search_slice = self.cfg.MODEL_PHASE_DETECTION_SLICE\n        min_index = np.argmin(signal[search_slice]) + search_slice.start\n        \n        signal1 = signal[:min_index]\n        signal2 = signal[min_index:]\n\n        grad1 = np.gradient(signal1)\n        grad1 /= grad1.max()\n        \n        grad2 = np.gradient(signal2)\n        grad2 /= grad2.max()\n\n        phase1 = np.argmin(grad1)\n        phase2 = np.argmax(grad2) + min_index\n\n        return phase1, phase2\n    \n    def _objective_function(self, s, signal, phase1, phase2):\n        delta = self.cfg.MODEL_OPTIMIZATION_DELTA\n        power = self.cfg.MODEL_POLYNOMIAL_DEGREE\n\n        if phase1 - delta <= 0 or phase2 + delta >= len(signal) or phase2 - delta - (phase1 + delta) < 5:\n            delta = 2\n\n        y = np.concatenate([\n            signal[: phase1 - delta],\n            signal[phase1 + delta : phase2 - delta] * (1 + s),\n            signal[phase2 + delta :]\n        ])\n        x = np.arange(len(y))\n\n        coeffs = np.polyfit(x, y, deg=power)\n        poly = np.poly1d(coeffs)\n        error = np.abs(poly(x) - y).mean()\n        \n        return error\n\n    def predict(self, single_preprocessed_signal):\n        signal_1d = single_preprocessed_signal[:, 1:].mean(axis=1)\n        signal_1d = savgol_filter(signal_1d, 20, 2)\n        \n        phase1, phase2 = self._phase_detector(signal_1d)\n\n        phase1 = max(self.cfg.MODEL_OPTIMIZATION_DELTA, phase1)\n        phase2 = min(len(signal_1d) - self.cfg.MODEL_OPTIMIZATION_DELTA - 1, phase2)    \n\n        result = minimize(\n            fun=self._objective_function,\n            x0=[0.0001],\n            args=(signal_1d, phase1, phase2),\n            method=\"Nelder-Mead\"\n        )\n        \n        return result.x[0]\n\n    def predict_all(self, preprocessed_signals):\n        predictions = [\n            self.predict(preprocessed_signal)\n            for preprocessed_signal in tqdm(preprocessed_signals)\n        ]\n        return np.array(predictions) * self.cfg.SCALE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.341949Z","iopub.execute_input":"2025-09-12T08:41:27.342551Z","iopub.status.idle":"2025-09-12T08:41:27.374223Z","shell.execute_reply.started":"2025-09-12T08:41:27.342523Z","shell.execute_reply":"2025-09-12T08:41:27.37301Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission Gen.","metadata":{}},{"cell_type":"code","source":"%%time\n\nclass SubmissionGenerator:\n    def __init__(self, config):\n        self.cfg = config\n        self.sample_submission = pd.read_csv(f\"{self.cfg.DATA_PATH}/sample_submission.csv\", index_col=\"planet_id\")\n        \n\n    def create(self, predictions):\n        planet_ids = self.sample_submission.index\n        repeated_predictions = np.repeat(predictions, self.sample_submission.shape[1] // 2).reshape(len(predictions), -1)\n        repeated_predictions = repeated_predictions.clip(0)\n        \n        sigmas = np.ones_like(repeated_predictions) * self.cfg.SIGMA\n\n        submission_df = pd.DataFrame(\n            np.concatenate([repeated_predictions, sigmas], axis=1),\n            columns=self.sample_submission.columns,\n            index=planet_ids\n        )\n        \n        submission_df.to_csv(\"submission.csv\")\n        return submission_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.3756Z","iopub.execute_input":"2025-09-12T08:41:27.37598Z","iopub.status.idle":"2025-09-12T08:41:27.404299Z","shell.execute_reply.started":"2025-09-12T08:41:27.375948Z","shell.execute_reply":"2025-09-12T08:41:27.403092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nconfig = Config()\n    \nsignal_processor = SignalProcessor(config)\npreprocessed_data = signal_processor.process_all_data()\n\nmodel = TransitModel(config)\npredictions = model.predict_all(preprocessed_data)\n\nsubmission_generator = SubmissionGenerator(config)\nsubmission = submission_generator.create(predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-12T08:41:27.405402Z","iopub.execute_input":"2025-09-12T08:41:27.405805Z","iopub.status.idle":"2025-09-12T08:42:03.203302Z","shell.execute_reply.started":"2025-09-12T08:41:27.405771Z","shell.execute_reply":"2025-09-12T08:42:03.202131Z"}},"outputs":[],"execution_count":null}]}