{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-31T21:37:44.883569Z","iopub.execute_input":"2024-10-31T21:37:44.883859Z","iopub.status.idle":"2024-10-31T21:37:52.624284Z","shell.execute_reply.started":"2024-10-31T21:37:44.883826Z","shell.execute_reply":"2024-10-31T21:37:52.623339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport pyarrow.parquet as pq\n\n# Set path to dataset folder\npath_folder = '/kaggle/input/ariel-data-challenge-2024/'","metadata":{"execution":{"iopub.status.busy":"2024-10-31T21:37:52.626397Z","iopub.execute_input":"2024-10-31T21:37:52.62732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Datset Analysis","metadata":{}},{"cell_type":"code","source":"# Load ADC info\nadc_info = pd.read_csv(os.path.join(path_folder, 'train_adc_info.csv'))\nprint(\"ADC Info:\")\ndisplay(adc_info.head())\n\n# Load Ground Truth Labels\ntrain_labels = pd.read_csv(os.path.join(path_folder, 'train_labels.csv'))\nprint(\"\\nTrain Labels:\")\ndisplay(train_labels.head())\n\n# Load Wavelength Data\nwavelength = pd.read_csv(os.path.join(path_folder, 'wavelengths.csv'))\nprint(\"\\nWavelength Data:\")\ndisplay(wavelength.head())\n\n# Load Axis Info\naxis_info = pd.read_parquet(os.path.join(path_folder, 'axis_info.parquet'))\nprint(\"\\nAxis Info:\")\ndisplay(axis_info.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image_data(planet_id, instrument='AIRS-CH0', train=True):\n    folder = 'train' if train else 'test'\n    file_path = os.path.join(path_folder, f'{folder}/{planet_id}/{instrument}_signal.parquet')\n    data = pq.read_table(file_path).to_pandas()\n    return data\n\nsample_planet_id = train_labels['planet_id'].iloc[0]\nsample_image_data = load_image_data(sample_planet_id, instrument='AIRS-CH0')\nprint(f\"\\nSample Image Data for Planet {sample_planet_id} (AIRS-CH0):\")\ndisplay(sample_image_data.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Unflatten and visualize a single image\ndef visualize_sample_image(image_data, index=0, shape=(32, 356)):\n    image = image_data.iloc[index].values.reshape(shape)\n    plt.imshow(image, cmap='viridis')\n    plt.title(f'Sample Image at Index {index}')\n    plt.colorbar(label='Intensity')\n    plt.xlabel('Spectral Axis')\n    plt.ylabel('Spatial Axis')\n    plt.show()\n\n# Visualize a sample image\nvisualize_sample_image(sample_image_data, index=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def restore_dynamic_range(image_data, adc_info, planet_id):\n    # Get gain and offset for the specified planet\n    planet_adc_info = adc_info[adc_info['planet_id'] == planet_id].iloc[0]\n    gain, offset = planet_adc_info['FGS1_adc_gain'], planet_adc_info['FGS1_adc_offset']\n    \n    # Restore the dynamic range\n    restored_data = image_data * gain + offset\n    return restored_data\n\n# Restore dynamic range for the sample image data\nrestored_sample_image_data = restore_dynamic_range(sample_image_data, adc_info, sample_planet_id)\nvisualize_sample_image(restored_sample_image_data, index=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate basic statistics for the sample image data\nprint(\"Sample Image Data - Basic Statistics:\")\ndisplay(sample_image_data.describe())\n\n# Visualize distribution of flux values for the sample data\nplt.figure(figsize=(10, 6))\nsns.histplot(sample_image_data.values.flatten(), bins=50, color='blue')\nplt.title(\"Distribution of Flux Values (Sample Image Data)\")\nplt.xlabel(\"Flux Value\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_flux_over_time(image_data, wavelength_index=100):\n    # Extract flux values for a specific wavelength across all time steps\n    flux_values = image_data.iloc[:, wavelength_index]\n    \n    plt.figure(figsize=(12, 6))\n    plt.plot(flux_values)\n    plt.title(f\"Flux Variation over Time (Wavelength Index {wavelength_index})\")\n    plt.xlabel(\"Time Step\")\n    plt.ylabel(\"Flux\")\n    plt.show()\n\n# Plot flux variation over time for a specific wavelength index\nplot_flux_over_time(sample_image_data, wavelength_index=100)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"def normalize_data(data):\n    \"\"\"Normalize data to range [0, 1]\"\"\"\n    return (data - data.min()) / (data.max() - data.min())\n\n# Apply normalization to the restored sample image data\nnormalized_sample_image_data = normalize_data(restored_sample_image_data)\nvisualize_sample_image(normalized_sample_image_data, index=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.ndimage import gaussian_filter1d\n\ndef smooth_data(data, sigma=2):\n    \"\"\"Apply Gaussian smoothing along the time axis\"\"\"\n    smoothed_data = gaussian_filter1d(data, sigma=sigma, axis=0)\n    return smoothed_data\n\n# Smooth the normalized data for a single planet\nsmoothed_sample_image_data = smooth_data(normalized_sample_image_data.values)\nvisualize_sample_image(pd.DataFrame(smoothed_sample_image_data), index=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def aggregate_flux(data, spatial_axis=0):\n    \"\"\"Aggregate flux across the spatial axis to reduce dimensionality\"\"\"\n    aggregated_data = data.sum(axis=spatial_axis)\n    return aggregated_data\n\n# Apply spatial aggregation to smoothed sample data\naggregated_sample_image_data = aggregate_flux(smoothed_sample_image_data)\nplt.plot(aggregated_sample_image_data)\nplt.title(\"Aggregated Flux Across Spatial Axis (Sample Data)\")\nplt.xlabel(\"Wavelength\")\nplt.ylabel(\"Aggregated Flux\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.signal import detrend\n\ndef detrend_time_series(data):\n    \"\"\"Detrend each wavelength independently along the time axis\"\"\"\n    detrended_data = np.apply_along_axis(detrend, axis=0, arr=data)\n    return detrended_data\n\n# Detrend the aggregated data\ndetrended_sample_image_data = detrend_time_series(aggregated_sample_image_data)\nplt.plot(detrended_sample_image_data)\nplt.title(\"Detrended Aggregated Flux (Sample Data)\")\nplt.xlabel(\"Wavelength\")\nplt.ylabel(\"Detrended Flux\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\n\ndef preprocess_data_pipeline(planet_id, adc_info, instrument='AIRS-CH0'):\n    # Step 1: Load and restore data\n    image_data = load_image_data(planet_id, instrument=instrument)\n    restored_data = restore_dynamic_range(image_data, adc_info, planet_id)\n    \n    # Step 2: Normalize and smooth data\n    normalized_data = normalize_data(restored_data)\n    smoothed_data = smooth_data(normalized_data.values)\n    \n    # Step 3: Aggregate flux across spatial axis\n    aggregated_data = aggregate_flux(smoothed_data)\n    \n    # Step 4: Detrend time-series\n    detrended_data = detrend_time_series(aggregated_data)\n    \n    # Step 5: Save preprocessed data\n    output_file = f'preprocessed_data_{planet_id}.pkl'\n    joblib.dump(detrended_data, output_file)\n    print(f\"Preprocessed data saved to {output_file}\")\n    \n    return detrended_data\n\n# Example usage for a single planet ID\npreprocessed_data = preprocess_data_pipeline(sample_planet_id, adc_info)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pyarrow.parquet as pq\nimport os\n\n# Function to load image data for a given planet ID and restore dynamic range\ndef load_and_restore_data(planet_id, adc_info, path_folder, instrument='AIRS-CH0', train=True):\n    folder = 'train' if train else 'test'\n    file_path = os.path.join(path_folder, f'{folder}/{planet_id}/{instrument}_signal.parquet')\n    data = pq.read_table(file_path).to_pandas()\n\n    # Restore dynamic range using gain and offset\n    planet_adc_info = adc_info[adc_info['planet_id'] == planet_id].iloc[0]\n    gain, offset = planet_adc_info['AIRS-CH0_adc_gain'], planet_adc_info['AIRS-CH0_adc_offset']\n    restored_data = data * gain + offset\n    \n    return restored_data\n\n# Load ADC info metadata\npath_folder = '/kaggle/input/ariel-data-challenge-2024/'\nadc_info = pd.read_csv(os.path.join(path_folder, 'train_adc_info.csv'))\n\ndef aggregate_flux_correctly(data, time_steps=11250, spatial_dim=32, spectral_dim=356):\n    # Reshape the flattened data to (time_steps, spatial, spectral)\n    reshaped_data = data.values.reshape(time_steps, spatial_dim, spectral_dim)\n    # Aggregate along the spatial axis (axis=1)\n    aggregated_data = reshaped_data.sum(axis=1)\n    return aggregated_data\n\nplanet_id = adc_info['planet_id'].iloc[0]\nrestored_data = load_and_restore_data(planet_id, adc_info, path_folder)\naggregated_data = aggregate_flux_correctly(restored_data)\nprint(\"Corrected Aggregated Data Shape:\", aggregated_data.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.signal import detrend\n\ndef preprocess_time_series(data):\n    \"\"\"Normalize and detrend each wavelength's time series\"\"\"\n    # Normalize \n    normalized_data = (data - data.min()) / (data.max() - data.min())\n    # Detrend each wavelength independently along the time axis\n    detrended_data = np.apply_along_axis(detrend, axis=0, arr=normalized_data)\n    return detrended_data\n\n# Preprocess the aggregated data\npreprocessed_data = preprocess_time_series(aggregated_data)\nprint(\"Preprocessed Data Shape:\", preprocessed_data.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\n\ndef preprocess_and_save_all_data(adc_info, path_folder):\n    \"\"\"Preprocess and save data for all planets in the training set.\"\"\"\n    preprocessed_folder = '/kaggle/working/preprocessed_data/'\n    os.makedirs(preprocessed_folder, exist_ok=True)\n    \n    for planet_id in adc_info['planet_id'].unique():\n        # Load and preprocess data\n        restored_data = load_and_restore_data(planet_id, adc_info, path_folder)\n        aggregated_data = aggregate_flux_correctly(restored_data)\n        preprocessed_data = preprocess_time_series(aggregated_data)\n        \n        # Save preprocessed data\n        output_file = os.path.join(preprocessed_folder, f'{planet_id}_preprocessed.pkl')\n        joblib.dump(preprocessed_data, output_file)\n        print(f\"Preprocessed data for planet {planet_id} saved to {output_file}\")\n\n# Run preprocessing for all planets in the training dataset\npreprocess_and_save_all_data(adc_info, path_folder)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Model Training\n","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport joblib\nimport os","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SpectraDataset(Dataset):\n    def __init__(self, data_dir):\n        self.data_dir = data_dir\n        self.planet_files = [f for f in os.listdir(data_dir) if f.endswith('_preprocessed.pkl')]\n    \n    def __len__(self):\n        return len(self.planet_files)\n    \n    def __getitem__(self, idx):\n        planet_file = self.planet_files[idx]\n        data = joblib.load(os.path.join(self.data_dir, planet_file))\n        # Separate into input (time series) and output (spectra)\n        time_series_data = torch.tensor(data, dtype=torch.float32)\n        \n        # For simplicity, let's assume we're only predicting the mean spectra; uncertainty predictions can be added as an extension\n        return time_series_data, time_series_data.mean(dim=0)\n\n# Load the dataset\ndata_dir = '/kaggle/working/preprocessed_data/'  # Directory with preprocessed data\ndataset = SpectraDataset(data_dir)\ndataloader = DataLoader(dataset, batch_size=4, shuffle=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SpectraModel(nn.Module):\n    def __init__(self, input_dim, hidden_dim, output_dim, num_layers=2):\n        super(SpectraModel, self).__init__()\n        self.lstm = nn.LSTM(input_dim, hidden_dim, num_layers, batch_first=True)\n        self.fc = nn.Linear(hidden_dim, output_dim)\n    \n    def forward(self, x):\n        # LSTM forward pass\n        lstm_out, _ = self.lstm(x)\n        # Take only the last hidden state\n        out = self.fc(lstm_out[:, -1, :])\n        return out\n\n# Define model parameters\ninput_dim = dataset[0][0].shape[1]  # Number of wavelengths\nhidden_dim = 64\noutput_dim = input_dim  # Predicting mean spectra for each wavelength\nmodel = SpectraModel(input_dim, hidden_dim, output_dim)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = nn.MSELoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\nnum_epochs = 20\nfor epoch in range(num_epochs):\n    model.train()\n    running_loss = 0.0\n    for inputs, targets in dataloader:\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item() * inputs.size(0)\n    \n    epoch_loss = running_loss / len(dataloader.dataset)\n    print(f\"Epoch {epoch+1}/{num_epochs}, Loss: {epoch_loss:.4f}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()\ntotal_loss = 0.0\nwith torch.no_grad():\n    for inputs, targets in dataloader:\n        outputs = model(inputs)\n        loss = criterion(outputs, targets)\n        total_loss += loss.item() * inputs.size(0)\n\neval_loss = total_loss / len(dataloader.dataset)\nprint(f\"Evaluation Loss: {eval_loss:.4f}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}