{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":12846694,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\n\ntrain_df = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train.csv')\nwavelengths_df = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/wavelengths.csv')\ntrain_star_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/train_star_info.csv')\n\nprint(\"Train shape:\", train_df.shape)\nprint(\"Wavelengths shape:\", wavelengths_df.shape)\nprint(\"Star info shape:\", train_star_info.shape)\n\ntrain_df.head()\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-30T09:44:50.506126Z","iopub.execute_input":"2025-06-30T09:44:50.50649Z","iopub.status.idle":"2025-06-30T09:44:53.124578Z","shell.execute_reply.started":"2025-06-30T09:44:50.506454Z","shell.execute_reply":"2025-06-30T09:44:53.123681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplanet_id = train_df['planet_id'].iloc[0]\nspectrum = train_df.iloc[0, 1:].values\nwavelengths = wavelengths_df.values.flatten()\n\nplt.figure(figsize=(10, 5))\nplt.plot(wavelengths, spectrum, color='purple')\nplt.xlabel('Wavelength (micron)')\nplt.ylabel('Normalized Flux')\nplt.title(f'Spectrum for Planet ID: {planet_id}')\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T09:44:53.126545Z","iopub.execute_input":"2025-06-30T09:44:53.126807Z","iopub.status.idle":"2025-06-30T09:44:53.429541Z","shell.execute_reply.started":"2025-06-30T09:44:53.126787Z","shell.execute_reply":"2025-06-30T09:44:53.42816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nplanet_id = train_df['planet_id'].iloc[0]  # First planet ID (e.g., 34983)\n\n# Construct the file path\nsignal_path = f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}/AIRS-CH0_signal_0.parquet'\n\n# Try loading it\ntry:\n    signal_df = pd.read_parquet(signal_path)\n    print(\"Loaded signal shape:\", signal_df.shape)\n\n    # Unflatten the signal data to (time, height, width)\n    signal_images = signal_df.values.reshape(-1, 32, 356)\n\n    # Plot the first frame\n    plt.imshow(signal_images[0], cmap='inferno')\n    plt.title(f'AIRS-CH0 Signal Frame 0 for Planet {planet_id}')\n    plt.colorbar()\n    plt.show()\n\nexcept Exception as e:\n    print(\"Error loading/parsing signal file:\", e)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T09:44:53.430545Z","iopub.execute_input":"2025-06-30T09:44:53.430808Z","iopub.status.idle":"2025-06-30T09:44:56.29795Z","shell.execute_reply.started":"2025-06-30T09:44:53.430788Z","shell.execute_reply":"2025-06-30T09:44:56.296886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Step 1: Load the ADC info\nadc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/adc_info.csv')\nairs_offset = adc_info['AIRS-CH0_adc_offset'].values[0]\nairs_gain = adc_info['AIRS-CH0_adc_gain'].values[0]\n\n# Step 2: Load the AIRS-CH0 signal\nsignal_df = pd.read_parquet('/kaggle/input/ariel-data-challenge-2025/train/34983/AIRS-CH0_signal_0.parquet')\n\n# Step 3: Convert to numpy and reshape to 3D: (11250, 32, 356)\nsignal_images = signal_df.values.reshape(-1, 32, 356).astype(float)\n\n# Step 4: Apply calibration (gain and offset)\ncalibrated_signal = signal_images * airs_gain + airs_offset\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T09:44:56.299118Z","iopub.execute_input":"2025-06-30T09:44:56.299487Z","iopub.status.idle":"2025-06-30T09:44:58.740616Z","shell.execute_reply.started":"2025-06-30T09:44:56.299454Z","shell.execute_reply":"2025-06-30T09:44:58.739516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5: Load and calibrate FGS1 signal\nfgs_offset = adc_info['FGS1_adc_offset'].values[0]\nfgs_gain = adc_info['FGS1_adc_gain'].values[0]\n\nfgs_df = pd.read_parquet('/kaggle/input/ariel-data-challenge-2025/train/34983/FGS1_signal_0.parquet')\n\n# Reshape to (135000, 32, 32)\nfgs_images = fgs_df.values.reshape(-1, 32, 32).astype(float)\n\n# Calibrate the signal\ncalibrated_fgs = fgs_images * fgs_gain + fgs_offset\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T09:44:58.7417Z","iopub.execute_input":"2025-06-30T09:44:58.742013Z","iopub.status.idle":"2025-06-30T09:45:00.763581Z","shell.execute_reply.started":"2025-06-30T09:44:58.741981Z","shell.execute_reply":"2025-06-30T09:45:00.762547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Visualize 1st frame of AIRS-CH0\nplt.figure(figsize=(10, 4))\nplt.subplot(1, 2, 1)\nplt.imshow(calibrated_signal[0], cmap='viridis')\nplt.title(\"AIRS-CH0 (frame 0)\")\nplt.colorbar()\n\n# Visualize 1st frame of FGS1\nplt.subplot(1, 2, 2)\nplt.imshow(calibrated_fgs[0], cmap='plasma')\nplt.title(\"FGS1 (frame 0)\")\nplt.colorbar()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T09:45:00.764621Z","iopub.execute_input":"2025-06-30T09:45:00.764885Z","iopub.status.idle":"2025-06-30T09:45:01.334011Z","shell.execute_reply.started":"2025-06-30T09:45:00.764863Z","shell.execute_reply":"2025-06-30T09:45:01.332797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\n# Calibration parameters\nadc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/adc_info.csv')\nfgs_offset, fgs_gain = adc_info.loc[0, ['FGS1_adc_offset', 'FGS1_adc_gain']]\nairs_offset, airs_gain = adc_info.loc[0, ['AIRS-CH0_adc_offset', 'AIRS-CH0_adc_gain']]\n\n# Training planet IDs\nplanet_ids = train_df['planet_id'].values\n\n# Function to extract features\ndef extract_features(planet_id):\n    features = {}\n    base_path = f'/kaggle/input/ariel-data-challenge-2025/train/{planet_id}'\n    \n    # --- AIRS-CH0 ---\n    try:\n        airs_path = f'{base_path}/AIRS-CH0_signal_0.parquet'\n        airs = pd.read_parquet(airs_path).values.astype(np.float32)\n        airs = airs * airs_gain + airs_offset\n        airs = airs.reshape(-1, 32, 356)\n        features['airs_mean'] = airs.mean()\n        features['airs_std'] = airs.std()\n    except:\n        features['airs_mean'] = np.nan\n        features['airs_std'] = np.nan\n\n    # --- FGS1 ---\n    try:\n        fgs_path = f'{base_path}/FGS1_signal_0.parquet'\n        fgs = pd.read_parquet(fgs_path).values.astype(np.float32)\n        fgs = fgs * fgs_gain + fgs_offset\n        fgs = fgs.reshape(-1, 32, 32)\n        features['fgs_mean'] = fgs.mean()\n        features['fgs_std'] = fgs.std()\n    except:\n        features['fgs_mean'] = np.nan\n        features['fgs_std'] = np.nan\n\n    features['planet_id'] = planet_id\n    return features\n\n# Extract features for all planets\nfeatures_list = [extract_features(pid) for pid in tqdm(planet_ids)]\nfeatures_df = pd.DataFrame(features_list)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T09:45:01.337236Z","iopub.execute_input":"2025-06-30T09:45:01.337605Z","iopub.status.idle":"2025-06-30T11:15:06.082485Z","shell.execute_reply.started":"2025-06-30T09:45:01.33758Z","shell.execute_reply":"2025-06-30T11:15:06.078294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nfrom sklearn.impute import SimpleImputer\n\n# Merge training spectra with features\ntrain_merged = pd.merge(train_df, features_df, on='planet_id')\nX = train_merged[['airs_mean', 'airs_std', 'fgs_mean', 'fgs_std']]\ny = train_merged.drop(columns=['planet_id'])\n\n# Handle NaNs\nimputer = SimpleImputer(strategy='mean')\nX_imputed = imputer.fit_transform(X)\n\n# Train one model per wavelength (very basic!)\nmodels = []\npreds = []\nfor col in y.columns:\n    model = LinearRegression()\n    model.fit(X_imputed, y[col])\n    models.append(model)\n\n# Load test planet features\ntest_star_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2025/test_star_info.csv')\ntest_planet_ids = test_star_info['planet_id'].values\n\ntest_features_list = [extract_features(pid) for pid in tqdm(test_planet_ids)]\ntest_features_df = pd.DataFrame(test_features_list)\nX_test = test_features_df[['airs_mean', 'airs_std', 'fgs_mean', 'fgs_std']]\nX_test_imputed = imputer.transform(X_test)\n\n# Make predictions\npredictions = []\nfor model in models:\n    pred = model.predict(X_test_imputed)\n    predictions.append(pred)\n    \n# Convert to DataFrame\nsubmission = pd.DataFrame(predictions).T\nsubmission.columns = y.columns\nsubmission.insert(0, 'planet_id', test_planet_ids)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T11:15:06.088335Z","iopub.execute_input":"2025-06-30T11:15:06.089098Z","iopub.status.idle":"2025-06-30T11:15:09.271974Z","shell.execute_reply.started":"2025-06-30T11:15:06.089042Z","shell.execute_reply":"2025-06-30T11:15:09.27053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T11:15:09.273359Z","iopub.execute_input":"2025-06-30T11:15:09.273753Z","iopub.status.idle":"2025-06-30T11:15:09.298388Z","shell.execute_reply.started":"2025-06-30T11:15:09.273728Z","shell.execute_reply":"2025-06-30T11:15:09.29691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Finalize the submission DataFrame\nsubmission = pd.DataFrame(predictions).T\nsubmission.columns = [f'wl_{i+1}' for i in range(submission.shape[1])]\nsubmission.insert(0, 'planet_id', test_planet_ids)\n\n# Save to CSV\n\n\n\nsubmission.to_csv('submission.csv', index=False)\n\n# Confirmation\nprint(\"✅ submission.csv created and saved successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T11:15:09.299937Z","iopub.execute_input":"2025-06-30T11:15:09.300268Z","iopub.status.idle":"2025-06-30T11:15:09.326956Z","shell.execute_reply.started":"2025-06-30T11:15:09.300244Z","shell.execute_reply":"2025-06-30T11:15:09.325675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!head submission.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T11:15:09.328854Z","iopub.execute_input":"2025-06-30T11:15:09.329181Z","iopub.status.idle":"2025-06-30T11:15:09.57318Z","shell.execute_reply.started":"2025-06-30T11:15:09.329159Z","shell.execute_reply":"2025-06-30T11:15:09.571904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -lh","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-30T11:15:09.575167Z","iopub.execute_input":"2025-06-30T11:15:09.575675Z","iopub.status.idle":"2025-06-30T11:15:09.78013Z","shell.execute_reply.started":"2025-06-30T11:15:09.575627Z","shell.execute_reply":"2025-06-30T11:15:09.778388Z"}},"outputs":[],"execution_count":null}]}