{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# --- 1. Load the training targets and submission file ---\nprint(\"Loading data...\")\ntrain_df = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/train.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/sample_submission.csv\")\n\n# Isolate the target columns (wavelengths)\nwl_cols = [f'wl_{i}' for i in range(1, 284)]\ntrain_targets = train_df[wl_cols]\n\n# --- 2. Calculate the mean and standard deviation for each wavelength ---\nprint(\"Calculating mean and std dev for each wavelength...\")\nmean_spectrum = train_targets.mean(axis=0)\nstd_spectrum = train_targets.std(axis=0)\n\n# --- 3. Create the submission DataFrame ---\nprint(\"Building submission file...\")\n# Get the planet_id from the sample submission (for the test set)\nsubmission_df = sample_submission[['planet_id']].copy()\n\n# Create columns for all the wl and sigma predictions\n# This is a bit of pandas magic to create the columns in the right order\nwl_sigma_cols = []\nfor i in range(1, 284):\n    wl_sigma_cols.append(f'wl_{i}')\n    wl_sigma_cols.append(f'sigma_{i}')\n\n# Recreate the submission DataFrame with the correct columns, initialized to zero\nfinal_submission = pd.DataFrame(columns=['planet_id'] + wl_sigma_cols)\nfinal_submission['planet_id'] = submission_df['planet_id']\n\n\n# --- 4. Populate the submission file ---\n# Assign the calculated mean to all the 'wl_' columns\nfor i, col in enumerate(wl_cols):\n    final_submission[col] = mean_spectrum[i]\n\n# Assign the calculated standard deviation to all the 'sigma_' columns\nsigma_cols = [f'sigma_{i}' for i in range(1, 284)]\nfor i, col in enumerate(sigma_cols):\n    # We use the std dev of the wl columns as our sigma estimate\n    final_submission[col] = std_spectrum[i]\n\n\n# --- 5. Save the submission file ---\nfinal_submission.to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv created successfully!\")\nprint(\"First 5 columns of submission file:\")\nprint(final_submission.head().iloc[:, :5])","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-05T17:36:03.482765Z","iopub.execute_input":"2025-09-05T17:36:03.483054Z","iopub.status.idle":"2025-09-05T17:36:04.075032Z","shell.execute_reply.started":"2025-09-05T17:36:03.483032Z","shell.execute_reply":"2025-09-05T17:36:04.074112Z"}},"outputs":[],"execution_count":null}]}