{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-15T15:22:17.284636Z","iopub.execute_input":"2024-09-15T15:22:17.285252Z","iopub.status.idle":"2024-09-15T15:22:19.688741Z","shell.execute_reply.started":"2024-09-15T15:22:17.285193Z","shell.execute_reply":"2024-09-15T15:22:19.687289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Necessary Libraries\nimport numpy as np\nimport pandas as pd\nimport pyarrow.parquet as pq\nimport os\nfrom tqdm import tqdm\n\n# Define the base data directory\ndata_dir = '/kaggle/input/ariel-data-challenge-2024'\n\n# Step 1: Verify Data Availability\nprint(\"Verifying data availability...\")\n\n# List contents of the base data directory\nbase_contents = os.listdir(data_dir)\nprint(f\"\\nContents of '{data_dir}':\")\nprint(base_contents)\n\n# Ensure essential files are present\nessential_files = ['train_adc_info.csv', 'test_adc_info.csv', 'train_labels.csv', 'wavelengths.csv', 'axis_info.parquet']\nmissing_files = [f for f in essential_files if f not in base_contents]\nif missing_files:\n    raise FileNotFoundError(f\"The following essential files are missing in '{data_dir}': {missing_files}\")\nelse:\n    print(\"\\nAll essential files are present.\")\n\n# Step 2: Load Metadata\nprint(\"\\nLoading metadata...\")\n\n# Paths to metadata files\ntest_adc_info_path = os.path.join(data_dir, 'test_adc_info.csv')\nwavelengths_path = os.path.join(data_dir, 'wavelengths.csv')\n\n# Load the metadata\ntest_adc_info = pd.read_csv(test_adc_info_path)\nwavelengths = pd.read_csv(wavelengths_path)\n\n# Inspect 'wavelengths.csv' to determine the number of wavelengths\nnum_wavelengths = wavelengths.shape[1] - 1  # Assuming first column is an identifier\nprint(f\"\\nNumber of wavelengths: {num_wavelengths}\")\n\n# Step 3: List Available Test 'planet_id's\nprint(\"\\nListing available test 'planet_id's...\")\n\navailable_test_planet_ids = test_adc_info['planet_id'].unique().tolist()\nprint(f\"Number of available test planets: {len(available_test_planet_ids)}\")\nprint(\"Sample 'planet_id's:\", available_test_planet_ids[:5])\n\n# Select a few test 'planet_id's for minimal submission (e.g., first 2)\nselected_test_planet_ids = available_test_planet_ids[:2]\nprint(f\"\\nSelected test 'planet_id's for submission: {selected_test_planet_ids}\")\n\n# Step 4: Define Helper Functions\n\ndef restore_dynamic_range(signal_df, gain, offset):\n    \"\"\"\n    Restore the original dynamic range of the signal.\n    \"\"\"\n    restored_signal = signal_df * gain + offset\n    return restored_signal\n\ndef apply_calibration(signal, calibration_dir):\n    \"\"\"\n    Apply calibration corrections: subtract dark current and apply flat field correction.\n    \"\"\"\n    # Define calibration file paths\n    dark_path = os.path.join(calibration_dir, 'dark.parquet')\n    flat_path = os.path.join(calibration_dir, 'flat.parquet')\n    \n    # Check if calibration files exist\n    if not os.path.isfile(dark_path):\n        raise FileNotFoundError(f\"Calibration file '{dark_path}' not found.\")\n    if not os.path.isfile(flat_path):\n        raise FileNotFoundError(f\"Calibration file '{flat_path}' not found.\")\n    \n    # Load calibration data\n    dark = pq.read_table(dark_path).to_pandas()\n    flat = pq.read_table(flat_path).to_pandas()\n    \n    # Subtract dark current (mean of dark frames)\n    signal_corrected = signal - dark.mean()\n    \n    # Apply flat field correction (divide by mean of flat frames)\n    signal_corrected /= flat.mean()\n    \n    return signal_corrected\n\n# Step 5: Process Each Selected Test 'planet_id' and Generate Predictions\n\n# Initialize lists to store submission data\nsubmission_planet_ids = []\nsubmission_spectra = []\nsubmission_uncertainties = []\n\nprint(\"\\nProcessing selected test 'planet_id's and generating predictions...\")\n\nfor planet_id in tqdm(selected_test_planet_ids, desc=\"Processing Planets\"):\n    try:\n        # Define paths\n        planet_dir = os.path.join(data_dir, 'test', str(planet_id))\n        airs_signal_path = os.path.join(planet_dir, 'AIRS-CH0_signal.parquet')\n        calibration_dir = os.path.join(planet_dir, 'AIRS-CH0_calibration')\n        \n        # Check if signal file exists\n        if not os.path.isfile(airs_signal_path):\n            print(f\"Signal file for planet_id '{planet_id}' not found. Skipping.\")\n            continue\n        \n        # Load signal data\n        airs_signal_df = pq.read_table(airs_signal_path).to_pandas()\n        print(f\"\\nLoaded 'AIRS-CH0_signal.parquet' for planet_id '{planet_id}' with shape: {airs_signal_df.shape}\")\n        \n        # Retrieve gain and offset from test_adc_info\n        adc_info = test_adc_info[test_adc_info['planet_id'] == planet_id]\n        if adc_info.empty:\n            print(f\"No ADC info found for planet_id '{planet_id}'. Skipping.\")\n            continue\n        gain = adc_info['gain'].values[0]\n        offset = adc_info['offset'].values[0]\n        print(f\"Gain: {gain}, Offset: {offset}\")\n        \n        # Restore dynamic range\n        airs_restored = restore_dynamic_range(airs_signal_df, gain, offset)\n        print(f\"Restored signal shape: {airs_restored.shape}\")\n        \n        # Apply calibration\n        airs_calibrated = apply_calibration(airs_restored, calibration_dir)\n        print(f\"Calibrated signal shape: {airs_calibrated.shape}\")\n        \n        # Aggregate signal by taking the mean over the time axis\n        airs_aggregated = airs_calibrated.mean(axis=0)\n        print(f\"Aggregated signal shape: {airs_aggregated.shape}\")\n        \n        # For demonstration, we'll create a dummy spectrum by normalizing the aggregated signal\n        # In a real scenario, you'd apply a model to predict the spectrum\n        spectrum = airs_aggregated.values[:num_wavelengths]  # Ensure the spectrum length matches wavelengths\n        spectrum_normalized = (spectrum - np.min(spectrum)) / (np.max(spectrum) - np.min(spectrum))\n        \n        # Assign fixed uncertainties (e.g., 10 ppm) as placeholders\n        uncertainties = np.full_like(spectrum_normalized, 10.0)\n        \n        # Append to submission lists\n        submission_planet_ids.append(planet_id)\n        submission_spectra.append(spectrum_normalized.tolist())\n        submission_uncertainties.append(uncertainties.tolist())\n        \n    except Exception as e:\n        print(f\"An error occurred while processing planet_id '{planet_id}': {e}\")\n        continue\n\n# Step 6: Prepare the Submission DataFrame\n\nprint(\"\\nPreparing the submission DataFrame...\")\n\n# Flatten the spectra and uncertainties into separate columns\n# Each spectrum and uncertainty list will be split into individual columns\n\n# Create a DataFrame for spectra\nspectra_df = pd.DataFrame(submission_spectra, columns=[f'spectra_{i+1}' for i in range(num_wavelengths)])\n# Create a DataFrame for uncertainties\nuncertainties_df = pd.DataFrame(submission_uncertainties, columns=[f'uncertainty_{i+1}' for i in range(num_wavelengths)])\n\n# Combine all parts into the final submission DataFrame\nsubmission_df = pd.concat([\n    pd.Series(submission_planet_ids, name='planet_id'),\n    spectra_df,\n    uncertainties_df\n], axis=1)\n\n# Display the submission DataFrame\nprint(\"\\nSample of the submission DataFrame:\")\nprint(submission_df.head())\n\n# Step 7: Save the Submission File\n\nsubmission_file_path = 'submission.csv'\nsubmission_df.to_csv(submission_file_path, index=False)\nprint(f\"\\nSubmission file '{submission_file_path}' created successfully.\")\n\n# Optional: Display the first few lines of the submission file\nprint(\"\\nFirst few lines of the submission file:\")\nprint(submission_df.head().to_csv(index=False))\n","metadata":{"execution":{"iopub.status.busy":"2024-09-15T15:22:19.692076Z","iopub.execute_input":"2024-09-15T15:22:19.692657Z","iopub.status.idle":"2024-09-15T15:22:21.62084Z","shell.execute_reply.started":"2024-09-15T15:22:19.692582Z","shell.execute_reply":"2024-09-15T15:22:21.619698Z"},"trusted":true},"execution_count":null,"outputs":[]}]}