{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-27T10:44:11.706591Z","iopub.execute_input":"2024-10-27T10:44:11.707012Z","iopub.status.idle":"2024-10-27T10:44:15.736878Z","shell.execute_reply.started":"2024-10-27T10:44:11.706973Z","shell.execute_reply":"2024-10-27T10:44:15.735612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy.signal import detrend","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:15.743188Z","iopub.execute_input":"2024-10-27T10:44:15.743642Z","iopub.status.idle":"2024-10-27T10:44:16.345178Z","shell.execute_reply.started":"2024-10-27T10:44:15.74358Z","shell.execute_reply":"2024-10-27T10:44:16.343869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\nimport os\n\n# Define the path to the downloaded ZIP file\nzip_file_path = '/kaggle/working/'\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.346996Z","iopub.execute_input":"2024-10-27T10:44:16.347825Z","iopub.status.idle":"2024-10-27T10:44:16.354154Z","shell.execute_reply.started":"2024-10-27T10:44:16.347763Z","shell.execute_reply":"2024-10-27T10:44:16.352915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.357659Z","iopub.execute_input":"2024-10-27T10:44:16.358121Z","iopub.status.idle":"2024-10-27T10:44:16.400398Z","shell.execute_reply.started":"2024-10-27T10:44:16.358071Z","shell.execute_reply":"2024-10-27T10:44:16.398841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.head(10))","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.401883Z","iopub.execute_input":"2024-10-27T10:44:16.402283Z","iopub.status.idle":"2024-10-27T10:44:16.421416Z","shell.execute_reply.started":"2024-10-27T10:44:16.402244Z","shell.execute_reply":"2024-10-27T10:44:16.420088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.tail(10))","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.423054Z","iopub.execute_input":"2024-10-27T10:44:16.423595Z","iopub.status.idle":"2024-10-27T10:44:16.442679Z","shell.execute_reply.started":"2024-10-27T10:44:16.423537Z","shell.execute_reply":"2024-10-27T10:44:16.441197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/train_adc_info.csv')  # Change this to the correct file\n\n# Inspect the dataset\nprint(data.info())  # Check available columns\n\n# Assuming your target variable is named 'target' and your feature column is named 'spectrum'\nif 'spectrum' in data.columns:\n    # Detrending the data to remove spacecraft jitter\n    data['detrended_spectrum'] = detrend(data['spectrum'])\n\n    # Normalize the data\n    scaler = StandardScaler()\n    data_scaled = scaler.fit_transform(data[['detrended_spectrum']])\n\n    # Split the data into train and test sets\n    X_train, X_test, y_train, y_test = train_test_split(data_scaled, data['target'], test_size=0.2, random_state=42)\nelse:\n    print(\"The expected column 'spectrum' is not found in the dataset.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.44439Z","iopub.execute_input":"2024-10-27T10:44:16.444837Z","iopub.status.idle":"2024-10-27T10:44:16.469871Z","shell.execute_reply.started":"2024-10-27T10:44:16.444792Z","shell.execute_reply":"2024-10-27T10:44:16.468482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.head(10))","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.471244Z","iopub.execute_input":"2024-10-27T10:44:16.471747Z","iopub.status.idle":"2024-10-27T10:44:16.482425Z","shell.execute_reply.started":"2024-10-27T10:44:16.471706Z","shell.execute_reply":"2024-10-27T10:44:16.480991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/test_adc_info.csv')  # Change this to the correct file\n\n# Inspect the dataset\nprint(data.info())  # Check available columns\n\n# Assuming your target variable is named 'target' and your feature column is named 'spectrum'\nif 'spectrum' in data.columns:\n    # Detrending the data to remove spacecraft jitter\n    data['detrended_spectrum'] = detrend(data['spectrum'])\n\n    # Normalize the data\n    scaler = StandardScaler()\n    data_scaled = scaler.fit_transform(data[['detrended_spectrum']])\n\n    # Split the data into train and test sets\n    X_train, X_test, y_train, y_test = train_test_split(data_scaled, data['target'], test_size=0.2, random_state=42)\nelse:\n    print(\"The expected column 'spectrum' is not found in the dataset.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.484198Z","iopub.execute_input":"2024-10-27T10:44:16.484731Z","iopub.status.idle":"2024-10-27T10:44:16.5084Z","shell.execute_reply.started":"2024-10-27T10:44:16.484679Z","shell.execute_reply":"2024-10-27T10:44:16.506747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.head(10))","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.509962Z","iopub.execute_input":"2024-10-27T10:44:16.510446Z","iopub.status.idle":"2024-10-27T10:44:16.520794Z","shell.execute_reply.started":"2024-10-27T10:44:16.510397Z","shell.execute_reply":"2024-10-27T10:44:16.519398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hybrid Noise Reduction with Denoising Autoencoder","metadata":{}},{"cell_type":"markdown","source":"### Preprocessing for 1D CNN","metadata":{}},{"cell_type":"code","source":"# Define auxiliary_folder if needed, or remove it if it's not necessary\n# Replace 'path_to_directory' with the correct directory where the dataset is located\nauxiliary_folder = '/kaggle/input/ariel-data-challenge-2024'\n\n# Load the dataset\ntrain_solution = np.loadtxt(f'{auxiliary_folder}/train_labels.csv', delimiter=',', skiprows=1)\n\n# Extract target values from the dataset\ntargets = train_solution[:, 1:] \n\n# Compute the mean of the targets (excluding the FGS point)\ntargets_mean = targets.mean(axis=1)\n\n# Get the number of rows in the targets\nN = targets.shape[0]\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.522523Z","iopub.execute_input":"2024-10-27T10:44:16.523482Z","iopub.status.idle":"2024-10-27T10:44:16.576003Z","shell.execute_reply.started":"2024-10-27T10:44:16.523409Z","shell.execute_reply":"2024-10-27T10:44:16.5748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the train labels dataset\ntrain_solution = np.loadtxt(f'{auxiliary_folder}/train_labels.csv', delimiter=',', skiprows=1)\n\nsignal_AIRS_diff_transposed_binned = train_solution[:, 1:]  # AIRS data (all features except the first column)\nsignal_FGS_diff_transposed_binned = np.random.rand(*signal_AIRS_diff_transposed_binned.shape)  # Placeholder for FGS data\n\nFGS_column = signal_FGS_diff_transposed_binned.sum(axis=1)  # Summing across the second axis\nFGS_column = FGS_column[:, np.newaxis]  # Making it 2D for concatenation\n\ndataset = np.concatenate([signal_AIRS_diff_transposed_binned, FGS_column], axis=1)\nprint(f\"AIRS Data Shape: {signal_AIRS_diff_transposed_binned.shape}\")\nprint(f\"FGS Column Shape: {FGS_column.shape}\")\nprint(f\"Concatenated Dataset Shape: {dataset.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.57749Z","iopub.execute_input":"2024-10-27T10:44:16.577868Z","iopub.status.idle":"2024-10-27T10:44:16.636074Z","shell.execute_reply.started":"2024-10-27T10:44:16.57783Z","shell.execute_reply":"2024-10-27T10:44:16.634767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Dataset Shape: {dataset.shape}\")\ndataset_sum = dataset.sum(axis=0)  \nprint(f\"Summed Dataset Shape: {dataset_sum.shape}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.641123Z","iopub.execute_input":"2024-10-27T10:44:16.641547Z","iopub.status.idle":"2024-10-27T10:44:16.648962Z","shell.execute_reply.started":"2024-10-27T10:44:16.641503Z","shell.execute_reply":"2024-10-27T10:44:16.647506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\ndef norm_star_spectrum(signal): \n    # Calculate the mean spectrum for normalization\n    img_star = signal[:, :50].mean(axis=1) + signal[:, -50:].mean(axis=1)\n    \n    # Reshape img_star to ensure it can be broadcast correctly\n    img_star = img_star[:, np.newaxis]  # This adds a new axis, making it 2D\n    \n    # Perform the normalization\n    normalized_signal = signal / img_star\n    \n    return normalized_signal\nprint(f\"Original dataset shape: {dataset.shape}\")  # Check the shape of dataset\ndataset_norm = norm_star_spectrum(dataset)\nprint(f\"Normalized dataset shape: {dataset_norm.shape}\")\ntry:\n    dataset_norm = np.transpose(dataset_norm, (0, 2, 1))\nexcept ValueError as e:\n    print(f\"Error in transposing: {e}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.650669Z","iopub.execute_input":"2024-10-27T10:44:16.651163Z","iopub.status.idle":"2024-10-27T10:44:16.664183Z","shell.execute_reply.started":"2024-10-27T10:44:16.651108Z","shell.execute_reply":"2024-10-27T10:44:16.662874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Splitting Targets and Observations into Training and Validation Sets","metadata":{}},{"cell_type":"code","source":"import random \n\ncut_inf, cut_sup = 39, 321  \nl = cut_sup - cut_inf + 1 \nwls = np.arange(l)\n\ndef split(data, N): \n    list_planets = random.sample(range(0, data.shape[0]), N)  # Use N instead of N_train\n    list_index_1 = np.zeros(data.shape[0], dtype=bool)\n    for planet in list_planets: \n        list_index_1[planet] = True\n    data_1 = data[list_index_1]\n    data_2 = data[~list_index_1]\n    return data_1, data_2, list_index_1\n\nN_train = 8 * N // 10 \ntrain_obs, valid_obs, list_index_train = split(dataset_norm, N_train)\ntrain_targets, valid_targets = targets[list_index_train], targets[~list_index_train]\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.665939Z","iopub.execute_input":"2024-10-27T10:44:16.666388Z","iopub.status.idle":"2024-10-27T10:44:16.67966Z","shell.execute_reply.started":"2024-10-27T10:44:16.666322Z","shell.execute_reply":"2024-10-27T10:44:16.67845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\n# Sample data for demonstration\n# Replace this with your actual data processing code\ntrain_wc = np.random.rand(200, 100)  # Example: 200 curves, 100 time points each\n\nplt.figure()\nfor i in range(1, 201):  # Start from 1 to avoid an out-of-bounds error for -0\n    plt.plot(train_wc[-i], '-', alpha=0.5)\nplt.title('Light-curves from the train set')\nplt.xlabel('Time')\nplt.ylabel('Normalized flux')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:44:16.681523Z","iopub.execute_input":"2024-10-27T10:44:16.68193Z","iopub.status.idle":"2024-10-27T10:44:17.825986Z","shell.execute_reply.started":"2024-10-27T10:44:16.681885Z","shell.execute_reply":"2024-10-27T10:44:17.82462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import KFold, cross_val_predict\nfrom sklearn.linear_model import LinearRegression  # Replace with the desired model\n\n# Assuming X and Y are your features and target arrays\n# Example initialization (replace with actual dataset loading)\nX = np.random.rand(100, 283)  # 100 samples, 283 features\nY = np.random.rand(100, 1)    # 100 target values\n\nfit_cloned = []  # fit_cloned after cross_val_predict will contain fitted estimators for evaluating test dataset.\n\nntrial = 5\nYP, YVar = None, None\n\nfor k in range(ntrial):\n    print(f\"k={k}\")\n    cv_split = KFold(n_splits=5, random_state=k+1234, shuffle=True).split(X, Y)\n    \n    # Use a standard model like LinearRegression\n    YP1 = cross_val_predict(LinearRegression(), X, Y, cv=cv_split)\n    \n    # Assuming YP1 is structured such that [:, 283:] are variance estimates\n    # Adjust this slicing based on the output shape of your model\n    YSigma1 = YP1[:, 283:]\n    YP1 = YP1[:, :283]\n    \n    if YP is None:\n        YP = YP1.copy()\n        YVar = YSigma1**2\n    else:\n        YP += YP1\n        YVar += YSigma1**2\n\nYP /= ntrial\nYSigma = np.sqrt(YVar / ntrial)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:50:05.717728Z","iopub.execute_input":"2024-10-27T10:50:05.71818Z","iopub.status.idle":"2024-10-27T10:50:05.94825Z","shell.execute_reply.started":"2024-10-27T10:50:05.718135Z","shell.execute_reply":"2024-10-27T10:50:05.947114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(fit_cloned)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:50:25.334493Z","iopub.execute_input":"2024-10-27T10:50:25.33495Z","iopub.status.idle":"2024-10-27T10:50:25.342766Z","shell.execute_reply.started":"2024-10-27T10:50:25.334906Z","shell.execute_reply":"2024-10-27T10:50:25.34155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport random\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Load train_labels.csv\ntrain_labels = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/train_labels.csv')\n\n# Replace 'planet_id' with the actual column name if different\nplanet_ids = train_labels['planet_id']  # Replace 'planet_id' if needed\n\n# Scaling and verification (optional)\nY = np.array(Y)  # Ensure Y, YP, YSigma are numpy arrays for easy indexing\nYP = np.array(YP)\nYSigma = np.array(YSigma)\n\n# Sample plots\nrandom.seed(23871)\n_, axs = plt.subplots(5, 3, figsize=(3 * 4, 5 * 3))\nfor ax in axs.flatten():\n    k = random.randint(0, len(Y) - 1)\n    planet_id = planet_ids.iloc[k]  # Use .iloc to get the correct ID\n    y_true = Y[k]\n    yp_pred = YP[k]\n    y_upper = YP[k] + 2.5 * YSigma[k]\n    y_lower = YP[k] - 2.5 * YSigma[k]\n    \n    # Ensure values are within a reasonable range\n    if np.max(y_true) > 0 and np.max(y_true) < 1e5:\n        ax.plot(y_true, label='ytrue')\n    else:\n        ax.plot(y_true / np.max(y_true), label='ytrue (scaled)')\n\n    ax.plot(y_upper, linewidth=0.5, label='yp+2.5s')\n    ax.plot(yp_pred, label='yp', alpha=0.8)\n    ax.plot(y_lower, linewidth=0.5, label='yp-2.5s')\n    \n    ax.legend()\n    ax.set_title(f'planet={planet_id}')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:55:38.978017Z","iopub.execute_input":"2024-10-27T10:55:38.978483Z","iopub.status.idle":"2024-10-27T10:55:43.296278Z","shell.execute_reply.started":"2024-10-27T10:55:38.978442Z","shell.execute_reply":"2024-10-27T10:55:43.295098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# find sigma multiplier\nimport scipy \n\ndef gll(yp, ysigma, yt):\n    return scipy.stats.norm.logpdf(yt, loc=yp, scale=ysigma).sum(1)\n\nmin_sigma = 1e-5\nL_ideal = gll(Y, min_sigma, Y).mean()  # 2998.0983016896525\nL_ref = gll(Y.flatten().mean(), Y.flatten().std(), Y).mean() # 1398.8395122870256\n\n# find sigma multiplier\nscores = []\nmultipliers = np.linspace(0.2, 2.0, 100)\nfor m in multipliers:\n    score = (gll(YP, np.clip(YSigma * m, min_sigma * 5, None), Y) - L_ref) / (L_ideal - L_ref)\n    score = np.clip(score, 0, None).mean()\n    scores.append(score)\n\nbest_score, multiplier = np.max(scores), multipliers[np.argmax(scores)]\nprint(f'best_score={best_score:.4f}, sigma multiplier={multiplier:.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:58:01.959894Z","iopub.execute_input":"2024-10-27T10:58:01.960348Z","iopub.status.idle":"2024-10-27T10:58:01.989657Z","shell.execute_reply.started":"2024-10-27T10:58:01.960308Z","shell.execute_reply":"2024-10-27T10:58:01.988314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(multipliers, scores)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T10:58:11.451065Z","iopub.execute_input":"2024-10-27T10:58:11.451577Z","iopub.status.idle":"2024-10-27T10:58:11.682281Z","shell.execute_reply.started":"2024-10-27T10:58:11.45153Z","shell.execute_reply":"2024-10-27T10:58:11.681178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}