{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":101849,"databundleVersionId":13093295,"sourceType":"competition"},{"sourceId":12737471,"sourceType":"datasetVersion","datasetId":8051512},{"sourceId":12766627,"sourceType":"datasetVersion","datasetId":8070643},{"sourceId":517229,"sourceType":"modelInstanceVersion","modelInstanceId":407916,"modelId":425782}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install gpytorch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:18:35.502229Z","iopub.execute_input":"2025-08-30T02:18:35.502671Z","iopub.status.idle":"2025-08-30T02:18:40.827778Z","shell.execute_reply.started":"2025-08-30T02:18:35.502637Z","shell.execute_reply":"2025-08-30T02:18:40.826757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport itertools\nfrom astropy.stats import sigma_clip\nimport glob\n# from sklearn.gaussian_process import GaussianProcessRegressor\n# from sklearn.gaussian_process.kernels import RBF, WhiteKernel, ConstantKernel\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nimport random\nimport pickle\nimport time\nfrom numba import njit, prange\nimport torch, gpytorch\nfrom torch.utils.data import DataLoader, TensorDataset\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:18:40.830765Z","iopub.execute_input":"2025-08-30T02:18:40.831557Z","iopub.status.idle":"2025-08-30T02:18:50.52196Z","shell.execute_reply.started":"2025-08-30T02:18:40.831519Z","shell.execute_reply":"2025-08-30T02:18:50.521092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wavelengths = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/wavelengths.csv\")\nadc_info = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/adc_info.csv\")\naxis_info = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2025/axis_info.parquet\")\nstar_info = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/train_star_info.csv\")\ntrain = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/train.csv\")\nsample = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/sample_submission.csv\")\n#train_AIRS_0 = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2025/train/1010375142/AIRS-CH0_signal_0.parquet\")\n#train_AIRS_0_dark = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2025/train/1010375142/AIRS-CH0_calibration_0/dark.parquet\")\n#train_AIRS_0_flat = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2025/train/1010375142/AIRS-CH0_calibration_0/flat.parquet\")\n#train_AIRS_0_dead = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2025/train/1010375142/AIRS-CH0_calibration_0/dead.parquet\")\n#train_linear_corr = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2025/train/1010375142/AIRS-CH0_calibration_0/linear_corr.parquet\")\n#test_AIRS_1 = pd.read_parquet(\"/kaggle/input/ariel-data-challenge-2025/test/1103775/AIRS-CH0_signal_1.parquet\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:18:50.523131Z","iopub.execute_input":"2025-08-30T02:18:50.523827Z","iopub.status.idle":"2025-08-30T02:18:50.955987Z","shell.execute_reply.started":"2025-08-30T02:18:50.523791Z","shell.execute_reply":"2025-08-30T02:18:50.954963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@njit(parallel=True, fastmath=True)\ndef poly_correct_numba_unrolled(signals, coeffs):\n    \"\"\"\n    Numba-parallel Horner's method, unrolled for degree=6, cache-friendly blocking.\n    signals: (t, 32, 282)\n    coeffs:  (6, 32, 282) highest degree first\n    \"\"\"\n    t, nx, ny = signals.shape\n    out = np.empty((t, nx, ny), dtype=np.float64)\n\n    # block the time loop for better cache reuse\n    block_size = 64\n    for ix in prange(nx):\n        for iy in range(ny):\n            c0, c1, c2, c3, c4, c5 = coeffs[:, ix, iy]\n            for tb in range(0, t, block_size):\n                t_end = min(tb + block_size, t)\n                for it in range(tb, t_end):\n                    s = signals[it, ix, iy]\n                    # Horner's method unrolled for degree 6\n                    acc = (((((c0 * s) + c1) * s + c2) * s + c3) * s + c4) * s + c5\n                    out[it, ix, iy] = acc\n    return out\n\ndef data_fast_fast_fast(AIRS, AIRS_dark, AIRS_flat, AIRS_dead, linear_corr, adc_info, axis_info):\n    # Gain + offset correction\n    signals = (AIRS / adc_info[\"AIRS-CH0_adc_gain\"].values[0]) + adc_info[\"AIRS-CH0_adc_offset\"].values[0]\n    signals = signals.values.reshape(signals.shape[0], 32, 356)\n\n    L, U = 39, 321\n    dt_airs = axis_info['AIRS-CH0-integration_time'].dropna().values\n    dt_airs[1::2] += 0.1\n    signals = signals[:, :, L:U]\n\n    # Dark/hot/dead mask → np.nan\n    dark = AIRS_dark.values[:, L:U]\n    hot = sigma_clip(dark, sigma=5, maxiters=5).mask\n    dead = AIRS_dead.values[:, L:U]\n    mask = np.broadcast_to(hot | dead, signals.shape)\n    signals = np.where(mask, signals.mean(axis=(0,1,2)), signals)\n\n    # Polynomial correction (Numba unrolled)\n    coeffs = np.flip(linear_corr.values.reshape(6, 32, 356)[:, :, L:U], axis=0)\n    signals = poly_correct_numba_unrolled(signals, coeffs)\n\n    # Dark subtraction\n    dark_masked = np.where(dead, dark.mean(axis=(0,1)), dark)\n    dark_masked = np.broadcast_to(dark_masked, signals.shape)\n    signals -= dark_masked * dt_airs[:, None, None]\n\n    # CDS subtraction\n    signals = signals[1::2] - signals[::2]\n\n    # Bin sum over 30-frame blocks\n    signals = signals.transpose(0, 2, 1)  # (time, 282, 32)\n    n_blocks = signals.shape[0] // 30\n    cds_binned = signals[:n_blocks * 30].reshape(n_blocks, 30, signals.shape[1], signals.shape[2]).sum(axis=1)\n\n    # Flat-field correction\n    flat = AIRS_flat.values[:, L:U].T\n    dead_flat = AIRS_dead.values[:, L:U].T\n    flat = np.where(dead_flat, flat.mean(axis=(0,1)), flat)\n    flat = np.broadcast_to(flat, cds_binned.shape)\n    cds_binned /= flat\n\n    return cds_binned","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:18:50.958159Z","iopub.execute_input":"2025-08-30T02:18:50.958519Z","iopub.status.idle":"2025-08-30T02:18:51.222488Z","shell.execute_reply.started":"2025-08-30T02:18:50.958493Z","shell.execute_reply":"2025-08-30T02:18:51.221425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# whether to Load or Process the DATA}\nLOAD = 1\n# whether to preprocess data fot Submission\n\nSUBMISSION = 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:18:51.223738Z","iopub.execute_input":"2025-08-30T02:18:51.224103Z","iopub.status.idle":"2025-08-30T02:18:51.228908Z","shell.execute_reply.started":"2025-08-30T02:18:51.224067Z","shell.execute_reply":"2025-08-30T02:18:51.228066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"start_time = time.time()\nRANDOM_SEED = 0\nN = 1100\nif not LOAD:\n    names = []\n    train_cds_binned = np.zeros((int(N*0.8), 187, 282, 32))\n    test_cds_binned = np.zeros((int(N*0.2), 187, 282, 32))\n    for i, j in enumerate(glob.glob(\"/kaggle/input/ariel-data-challenge-2025/train/*\")[:N]):\n        names.append(j[46:])\n        if i < N*0.8:\n            print(\"Train: \" + names[i])\n            start_time1 = time.time()\n            train_AIRS_0 = pd.read_parquet(j + \"/AIRS-CH0_signal_0.parquet\", engine = \"pyarrow\", use_threads = True)\n            train_AIRS_0_dark = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/dark.parquet\", engine = \"pyarrow\", use_threads = True)\n            train_AIRS_0_flat = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/flat.parquet\", engine = \"pyarrow\", use_threads = True)\n            train_AIRS_0_dead = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/dead.parquet\", engine = \"pyarrow\", use_threads = True)\n            train_linear_corr0 = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/linear_corr.parquet\", engine = \"pyarrow\", use_threads = True)\n            end_time1 = time.time()\n            elapsed_time1 = end_time1 - start_time1\n            print(f\"Execution time: {elapsed_time1:.4f} seconds\")\n            train_cds_binned[i] = data_fast_fast_fast(train_AIRS_0, train_AIRS_0_dark, train_AIRS_0_flat, train_AIRS_0_dead, train_linear_corr0, adc_info, axis_info)\n            print(\"(\",i,train_cds_binned.shape[1],train_cds_binned.shape[2],train_cds_binned.shape[3],\")\")\n        else:\n            print(\"Test: \" + names[i] )\n            test_AIRS_0 = pd.read_parquet(j + \"/AIRS-CH0_signal_0.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_AIRS_0_dark = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/dark.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_AIRS_0_flat = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/flat.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_AIRS_0_dead = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/dead.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_linear_corr0 = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/linear_corr.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_cds_binned[i-int(N*0.8)] = data_fast_fast_fast(test_AIRS_0, test_AIRS_0_dark, test_AIRS_0_flat, test_AIRS_0_dead, test_linear_corr0, adc_info, axis_info)\n            print(\"(\",i,test_cds_binned.shape[1],test_cds_binned.shape[2],test_cds_binned.shape[3],\")\")\n    np.save(os.path.join(\"/kaggle/working/\", 'train.npy'), train_cds_binned)\n    np.save(os.path.join(\"/kaggle/working/\", 'test.npy'), test_cds_binned)\n    np.save(os.path.join(\"/kaggle/working/\", 'names.npy'), names)\nelse:\n    if SUBMISSION:\n        names_test = []\n        train_cds_binned = np.load(\"/kaggle/input/ariel-data-100planets/train.npy\")\n        test_cds_binned = np.load(\"/kaggle/input/ariel-data-100planets/test.npy\")\n        train_cds_binned = np.vstack([train_cds_binned, test_cds_binned])\n        del test_cds_binned\n        test_planet_list = glob.glob(\"/kaggle/input/ariel-data-challenge-2025/test/*\")\n        test_cds_binned = np.zeros((len(test_planet_list), 187, 282, 32))\n        names = np.load(\"/kaggle/input/ariel-data-100planets/names.npy\")\n        for i, j in enumerate(test_planet_list):\n            if (i % 100 == 0) or (i == 0):\n                print(i,\"/\",len(test_planet_list))\n            print(os.path.split(j)[1])\n            names_test.append(os.path.split(j)[1])\n            test_AIRS_0 = pd.read_parquet(j + \"/AIRS-CH0_signal_0.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_AIRS_0_dark = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/dark.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_AIRS_0_flat = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/flat.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_AIRS_0_dead = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/dead.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_linear_corr0 = pd.read_parquet(j + \"/AIRS-CH0_calibration_0/linear_corr.parquet\", engine = \"pyarrow\", use_threads = True)\n            test_cds_binned[i] = data_fast_fast_fast(test_AIRS_0, test_AIRS_0_dark, test_AIRS_0_flat, test_AIRS_0_dead, test_linear_corr0, adc_info, axis_info)\n    else:\n        \n        train_cds_binned = np.vstack([np.load(\"/kaggle/input/ariel-data-1100lanets/train_part_1.npy\"), np.load(\"/kaggle/input/ariel-data-1100lanets/train_part_2.npy\"),\n                                      np.load(\"/kaggle/input/ariel-data-1100lanets/train_part_3.npy\"), np.load(\"/kaggle/input/ariel-data-1100lanets/train_part_4.npy\")])\n        test_cds_binned = np.load(\"/kaggle/input/ariel-data-1100lanets/test.npy\")\n        names = np.load(\"/kaggle/input/ariel-data-1100lanets/names.npy\")\nend_time = time.time()\nelapsed_time = end_time - start_time\nprint(f\"Execution time: {elapsed_time:.4f} seconds\")","metadata":{"trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-08-30T02:18:51.229802Z","iopub.execute_input":"2025-08-30T02:18:51.230132Z","iopub.status.idle":"2025-08-30T02:20:17.828458Z","shell.execute_reply.started":"2025-08-30T02:18:51.230107Z","shell.execute_reply":"2025-08-30T02:20:17.827208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if SUBMISSION:\n    test_list = [int(os.path.split(test_planet_list[i])[1]) for i in range(len(test_planet_list))]\n    unsorted_df = pd.DataFrame(test_cds_binned.mean(axis = (1,3)), index = test_list, columns = wavelengths.columns[1:])\n    unsorted_df.index.name = 'planet_id'\n    sorted_df = unsorted_df.sort_values(by=\"planet_id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:20:17.829405Z","iopub.execute_input":"2025-08-30T02:20:17.829719Z","iopub.status.idle":"2025-08-30T02:20:17.835474Z","shell.execute_reply.started":"2025-08-30T02:20:17.829685Z","shell.execute_reply":"2025-08-30T02:20:17.834365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#plt.plot(wavelengths.iloc[-1][1:].values,train[train[\"planet_id\"]==int(names[-1])].values[0,2:])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T00:18:17.164042Z","iopub.execute_input":"2025-08-30T00:18:17.164476Z","iopub.status.idle":"2025-08-30T00:18:17.187146Z","shell.execute_reply.started":"2025-08-30T00:18:17.164431Z","shell.execute_reply":"2025-08-30T00:18:17.185961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#plt.plot(cds_binned.mean(axis=(1,2)), '-', alpha=0.3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T00:18:17.188358Z","iopub.execute_input":"2025-08-30T00:18:17.189029Z","iopub.status.idle":"2025-08-30T00:18:17.20824Z","shell.execute_reply.started":"2025-08-30T00:18:17.188988Z","shell.execute_reply":"2025-08-30T00:18:17.207165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df = pd.DataFrame(cds_binned.mean(axis=(2)))\n#df.columns = wavelengths.iloc[0].values[1:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T22:11:31.67704Z","iopub.status.idle":"2025-08-29T22:11:31.677395Z","shell.execute_reply.started":"2025-08-29T22:11:31.677223Z","shell.execute_reply":"2025-08-29T22:11:31.677237Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Show the image\nplt.imshow(df, aspect='auto', extent = (np.min(df.columns), np.max(df.columns), np.max(df.index), np.min(df.index)))\nplt.xlabel('Wavelength [um]')\nplt.ylabel('Time Step')\nplt.colorbar()  # Optional: show color scale\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-08-12T17:38:52.048325Z","iopub.execute_input":"2025-08-12T17:38:52.048627Z","iopub.status.idle":"2025-08-12T17:38:52.611891Z","shell.execute_reply.started":"2025-08-12T17:38:52.048602Z","shell.execute_reply":"2025-08-12T17:38:52.610848Z"}}},{"cell_type":"markdown","source":"dg = (df - df.mean(axis=0))/df.std(axis=0)","metadata":{"execution":{"iopub.status.busy":"2025-08-11T15:16:48.488947Z","iopub.execute_input":"2025-08-11T15:16:48.489316Z","iopub.status.idle":"2025-08-11T15:16:48.499421Z","shell.execute_reply.started":"2025-08-11T15:16:48.489275Z","shell.execute_reply":"2025-08-11T15:16:48.498498Z"}}},{"cell_type":"markdown","source":"plt.imshow(dg, aspect='auto', extent = (np.min(dg.columns), np.max(dg.columns), np.max(dg.index), np.min(dg.index)))\nplt.xlabel('Wavelength [um]')\nplt.ylabel('Time Step')\nplt.colorbar()  # Optional: show color scale\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-08-11T15:16:48.500334Z","iopub.execute_input":"2025-08-11T15:16:48.500749Z","iopub.status.idle":"2025-08-11T15:16:48.839405Z","shell.execute_reply.started":"2025-08-11T15:16:48.500717Z","shell.execute_reply":"2025-08-11T15:16:48.838305Z"}}},{"cell_type":"markdown","source":"t_ingress, t_egress = 25,125\nfig, ax = plt.subplots(2,1)\nax[0].plot(df[t_egress-1:].mean(axis=0)-df[t_ingress:t_egress].mean(axis=0))\nax[1].plot(df[0:t_ingress-1].mean(axis=0)-df[t_ingress:t_egress].mean(axis=0))","metadata":{"execution":{"iopub.status.busy":"2025-08-11T15:16:48.840556Z","iopub.execute_input":"2025-08-11T15:16:48.840823Z","iopub.status.idle":"2025-08-11T15:16:49.133299Z","shell.execute_reply.started":"2025-08-11T15:16:48.840802Z","shell.execute_reply":"2025-08-11T15:16:49.132371Z"}}},{"cell_type":"code","source":"planet = 879\nnrows = 4\nncols = 8\nfig, axes = plt.subplots(nrows, ncols, figsize=(15, 10)) # Adjust figsize as needed\naxes_flat = axes.flatten()\nfor i in range(32):\n    ax = axes_flat[i]\n    ax.plot(diff[planet][i])\n    ax.set_title(f'Plot {i+1}') # Add a unique title for each plot\n    ax.set_xlabel('X-axis')\n    ax.set_ylabel('Y-axis')\n\n# Adjust layout to prevent overlapping titles and labels\nplt.tight_layout()\n\n# Display the plots\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T00:32:07.954787Z","iopub.execute_input":"2025-08-30T00:32:07.956326Z","iopub.status.idle":"2025-08-30T00:32:12.599933Z","shell.execute_reply.started":"2025-08-30T00:32:07.956288Z","shell.execute_reply":"2025-08-30T00:32:12.598787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"maxvalue, minvalue = train_cds_binned.transpose(0,3,2,1).max(3), train_cds_binned.transpose(0,3,2,1).min(3)\ndifftrain = (maxvalue - minvalue)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:25:16.593786Z","iopub.execute_input":"2025-08-30T02:25:16.594745Z","iopub.status.idle":"2025-08-30T02:25:23.224979Z","shell.execute_reply.started":"2025-08-30T02:25:16.594711Z","shell.execute_reply":"2025-08-30T02:25:23.22384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"maxvalue, minvalue = test_cds_binned.transpose(0,3,2,1).max(3), test_cds_binned.transpose(0,3,2,1).min(3)\ndifftest = (maxvalue - minvalue)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:25:23.226887Z","iopub.execute_input":"2025-08-30T02:25:23.227358Z","iopub.status.idle":"2025-08-30T02:25:24.830768Z","shell.execute_reply.started":"2025-08-30T02:25:23.227297Z","shell.execute_reply":"2025-08-30T02:25:24.829939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"difftrain.shape, difftest.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:25:24.831674Z","iopub.execute_input":"2025-08-30T02:25:24.832002Z","iopub.status.idle":"2025-08-30T02:25:24.839157Z","shell.execute_reply.started":"2025-08-30T02:25:24.831972Z","shell.execute_reply":"2025-08-30T02:25:24.838105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#wavelength = 110\nplanet = 104\nvalue = []\nfor wavelength in range(282):\n value.append((np.max(train_cds_binned[planet].mean(2).transpose(1,0)[wavelength]) - np.min(train_cds_binned[planet].mean(2).transpose(1,0)[wavelength])))\nplt.subplot(3, 1, 1)\nplt.plot(value)\nplt.subplot(3, 1, 2)\nplt.plot(y[planet])\nplt.subplot(3, 1, 3)\nplt.plot(train_cds_binned[planet].mean(2).transpose(1,0).mean(1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T00:18:19.501513Z","iopub.execute_input":"2025-08-30T00:18:19.501974Z","iopub.status.idle":"2025-08-30T00:18:21.550375Z","shell.execute_reply.started":"2025-08-30T00:18:19.501945Z","shell.execute_reply":"2025-08-30T00:18:21.549447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X = train_cds_binned.mean(axis = (1,3))\nX = difftrain.mean(1)\n# X = pd.DataFrame(X.T)\n# X.index = wavelengths.iloc[-1][1:].values\n\ny = []\nif SUBMISSION:\n    for i in names:\n        y.append(train[train[\"planet_id\"]==int(i)].values[0,2:])\nelse:\n    for i in names[:int(N*0.8)]:\n        y.append(train[train[\"planet_id\"]==int(i)].values[0,2:])\ny = pd.DataFrame(y)\n# y.index = wavelengths.iloc[-1][1:].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:24:10.989803Z","iopub.execute_input":"2025-08-30T02:24:10.990283Z","iopub.status.idle":"2025-08-30T02:24:14.112528Z","shell.execute_reply.started":"2025-08-30T02:24:10.99025Z","shell.execute_reply":"2025-08-30T02:24:14.111538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if SUBMISSION:\n    Xtest = sorted_df.values\nif not SUBMISSION:\n    # Xtest = test_cds_binned.mean(axis = (1,3))\n    Xtest = difftest.mean(1)\n    ytest = []\n    for i in names[int(N*0.8):]:\n        ytest.append(train[train[\"planet_id\"]==int(i)].values[0,2:])\n    ytest = pd.DataFrame(ytest).transpose()\n    ytest.index = wavelengths.iloc[-1][1:].values\n# Xtest = pd.DataFrame(Xtest.T)\n# Xtest.index = wavelengths.iloc[-1][1:].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:26:06.682736Z","iopub.execute_input":"2025-08-30T02:26:06.683672Z","iopub.status.idle":"2025-08-30T02:26:06.790778Z","shell.execute_reply.started":"2025-08-30T02:26:06.683638Z","shell.execute_reply":"2025-08-30T02:26:06.789769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X = X.T\n# y = y.T\n# Xtest = Xtest.T\nif SUBMISSION:\n    print(X.shape, Xtest.shape, y.shape)\nelse:\n    ytest = ytest.T\n    print(X.shape, y.shape, Xtest.shape, ytest.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:26:07.796442Z","iopub.execute_input":"2025-08-30T02:26:07.796816Z","iopub.status.idle":"2025-08-30T02:26:07.803235Z","shell.execute_reply.started":"2025-08-30T02:26:07.796789Z","shell.execute_reply":"2025-08-30T02:26:07.802326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = X.reshape(X.shape[0], -1)\nXtest = Xtest.reshape(Xtest.shape[0], -1)\nscaler_flux = StandardScaler(with_mean=True, with_std=True)\nX_train_scaled = scaler_flux.fit_transform(X)\nX_test_scaled  = scaler_flux.transform(Xtest)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:26:10.515792Z","iopub.execute_input":"2025-08-30T02:26:10.516144Z","iopub.status.idle":"2025-08-30T02:26:10.52957Z","shell.execute_reply.started":"2025-08-30T02:26:10.516113Z","shell.execute_reply":"2025-08-30T02:26:10.528536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_components = 60   # try 10-50; you can tune this\npca = PCA(n_components=n_components)\nZ_train = pca.fit_transform(X_train_scaled)   # (n_train, n_components)\nZ_test  = pca.transform(X_test_scaled)        # (n_test,  n_components)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T01:10:38.307575Z","iopub.execute_input":"2025-08-30T01:10:38.307891Z","iopub.status.idle":"2025-08-30T01:10:39.605686Z","shell.execute_reply.started":"2025-08-30T01:10:38.30787Z","shell.execute_reply":"2025-08-30T01:10:39.604553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import umap\nimport hdbscan\ndef cluster_planets(R, n_neighbors=32, min_dist=0.1, n_components=282, min_cluster_size=32, random_state=42):\n    umap_model = umap.UMAP(\n        n_neighbors=n_neighbors,\n        min_dist=min_dist,\n        n_components=n_components,\n        random_state=random_state\n    )\n    embedding = umap_model.fit_transform(R)\n    \n    # 3. Apply HDBSCAN for clustering\n    clusterer = hdbscan.HDBSCAN(\n        min_cluster_size=min_cluster_size,\n        metric='euclidean'\n    )\n    labels = clusterer.fit_predict(embedding)\n    \n    return labels, embedding","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:41:37.531758Z","iopub.execute_input":"2025-08-30T02:41:37.532178Z","iopub.status.idle":"2025-08-30T02:41:37.539138Z","shell.execute_reply.started":"2025-08-30T02:41:37.532146Z","shell.execute_reply":"2025-08-30T02:41:37.538142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels, embedding = cluster_planets(X_train_scaled)\n\nprint(\"Cluster counts:\", np.unique(labels, return_counts=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:41:37.768152Z","iopub.execute_input":"2025-08-30T02:41:37.768555Z","iopub.status.idle":"2025-08-30T02:41:46.11674Z","shell.execute_reply.started":"2025-08-30T02:41:37.768525Z","shell.execute_reply":"2025-08-30T02:41:46.1158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.scatter(embedding[:,0], embedding[:,1], c=labels, cmap='Spectral', s=10)\nplt.colorbar(label='Cluster')\nplt.title(\"UMAP + HDBSCAN clustering of planets\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T02:41:46.118335Z","iopub.execute_input":"2025-08-30T02:41:46.118624Z","iopub.status.idle":"2025-08-30T02:41:46.385638Z","shell.execute_reply.started":"2025-08-30T02:41:46.118601Z","shell.execute_reply":"2025-08-30T02:41:46.384641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import pairwise_distances\nd = pairwise_distances(Z_test, Z_train)  # shape (n_test, 1100)\nprint(\"min dist per test -> train:\", d.min(axis=1))\nprint(\"mean train-train min dist:\", pairwise_distances(Z_train, Z_train).min(axis=1).mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T01:10:39.606866Z","iopub.execute_input":"2025-08-30T01:10:39.607173Z","iopub.status.idle":"2025-08-30T01:10:39.642584Z","shell.execute_reply.started":"2025-08-30T01:10:39.607149Z","shell.execute_reply":"2025-08-30T01:10:39.641816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_scaler = StandardScaler(with_mean=True, with_std=True)\nY_train_scaled = y_scaler.fit_transform(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T01:10:39.643558Z","iopub.execute_input":"2025-08-30T01:10:39.644089Z","iopub.status.idle":"2025-08-30T01:10:39.665241Z","shell.execute_reply.started":"2025-08-30T01:10:39.644064Z","shell.execute_reply":"2025-08-30T01:10:39.664136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nXt = torch.from_numpy(Z_train).float().to(device)\nYt = torch.from_numpy(Y_train_scaled).float().to(device)\nXttest = torch.from_numpy(Z_test).float().to(device)\n\nN, d = Xt.shape\nP = Yt.shape[1]\n\ntrain_loader = DataLoader(TensorDataset(Xt, Yt), batch_size=256, shuffle=False)\nM = min(128, N)   # inducing points\nprint(M)\ninducing = Xt[torch.randperm(N)[:M]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T01:10:39.667101Z","iopub.execute_input":"2025-08-30T01:10:39.667445Z","iopub.status.idle":"2025-08-30T01:10:39.679574Z","shell.execute_reply.started":"2025-08-30T01:10:39.667422Z","shell.execute_reply":"2025-08-30T01:10:39.67863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class IndependentMTGP(gpytorch.models.ApproximateGP):\n    def __init__(self, inducing_points, num_tasks, ard_dims):\n        # one variational distribution for all tasks (independent across tasks)\n        variational_distribution = gpytorch.variational.CholeskyVariationalDistribution(inducing_points.size(0))\n        variational_strategy = gpytorch.variational.VariationalStrategy(\n            self, inducing_points, variational_distribution, learn_inducing_locations=True\n        )\n        # wrap with multitask independence\n        mt_strategy = gpytorch.variational.IndependentMultitaskVariationalStrategy(\n            variational_strategy, num_tasks=num_tasks\n        )\n        super().__init__(mt_strategy)\n\n        self.mean_module = gpytorch.means.ConstantMean(batch_shape=torch.Size([num_tasks]))\n        self.covar_module = gpytorch.kernels.ScaleKernel(\n            gpytorch.kernels.RBFKernel(ard_num_dims=ard_dims, batch_shape=torch.Size([num_tasks])),\n            batch_shape=torch.Size([num_tasks])\n        )\n\n    def forward(self, x):\n        mean_x = self.mean_module(x)\n        covar_x = self.covar_module(x)\n        return gpytorch.distributions.MultivariateNormal(mean_x, covar_x)\n\nmodel = IndependentMTGP(inducing, num_tasks=P, ard_dims=d).to(device)\nlikelihood = gpytorch.likelihoods.MultitaskGaussianLikelihood(num_tasks=P).to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T01:10:43.594879Z","iopub.execute_input":"2025-08-30T01:10:43.595216Z","iopub.status.idle":"2025-08-30T01:10:43.607839Z","shell.execute_reply.started":"2025-08-30T01:10:43.595192Z","shell.execute_reply":"2025-08-30T01:10:43.606635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.train(); likelihood.train()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.03)\nmll = gpytorch.mlls.VariationalELBO(likelihood, model, num_data=N)\n\nEPOCHS = 10\nfor epoch in range(EPOCHS):\n    total_loss = 0\n    for xb, yb in train_loader:\n        optimizer.zero_grad(set_to_none=True)\n        out = model(xb)\n        loss = -mll(out, yb)\n        loss.backward()\n        optimizer.step()\n        total_loss += loss.item()\n    print(f\"epoch {epoch+1} loss {total_loss:.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T01:10:43.924708Z","iopub.execute_input":"2025-08-30T01:10:43.92506Z","iopub.status.idle":"2025-08-30T01:11:54.262173Z","shell.execute_reply.started":"2025-08-30T01:10:43.925032Z","shell.execute_reply":"2025-08-30T01:11:54.26125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval(); likelihood.eval()\nwith torch.inference_mode(), gpytorch.settings.fast_pred_var():\n    pred = likelihood(model(Xttest))\n    mean = pred.mean.reshape(-1, P)        # (M, P)\n    sigma = pred.variance.sqrt().reshape(-1, P)\n\nY_pred = mean.cpu().numpy()\nY_sigma = sigma.cpu().numpy()\nY_pred = y_scaler.inverse_transform(Y_pred)   # (n_test, L)\nY_sigma = Y_sigma * y_scaler.scale_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T01:11:54.263954Z","iopub.execute_input":"2025-08-30T01:11:54.26424Z","iopub.status.idle":"2025-08-30T01:11:54.744595Z","shell.execute_reply.started":"2025-08-30T01:11:54.264217Z","shell.execute_reply":"2025-08-30T01:11:54.743787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not SUBMISSION:\n    dfy = pd.DataFrame(Y_pred.T)\n    dfy.index = wavelengths.iloc[-1][1:].values\n    a = 0\n    fig, axs = plt.subplots(2, 1)\n    axs[0].plot(dfy.index, dfy.values[:,a], label='Line 1')\n    axs[0].set_title('Predited Values')\n    axs[1].plot(dfy.index, ytest.T.values[:,a], label='Line 2', linestyle='--')\n    axs[1].set_title('Actual Values')\n\n\n# Plot the second line\n#plt.plot(dfy.index, ytest.T.values[:,a], label='Line 2', linestyle='--')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T01:11:54.745485Z","iopub.execute_input":"2025-08-30T01:11:54.745748Z","iopub.status.idle":"2025-08-30T01:11:55.102924Z","shell.execute_reply.started":"2025-08-30T01:11:54.745727Z","shell.execute_reply":"2025-08-30T01:11:55.101895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not SUBMISSION:\n    from sklearn.metrics import mean_absolute_error\n\n    # y_true and y_pred should be arrays of the same shape\n    mae = mean_absolute_error(ytest.T.values, dfy.values)\n    print(\"MAE:\", mae)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-30T01:12:58.449608Z","iopub.execute_input":"2025-08-30T01:12:58.449996Z","iopub.status.idle":"2025-08-30T01:12:58.457971Z","shell.execute_reply.started":"2025-08-30T01:12:58.449968Z","shell.execute_reply":"2025-08-30T01:12:58.457144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def make_dataframe(pred_array, sigma_pred):\n#     df = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/sample_submission.csv\")\n#     df.drop(0, inplace=True)\n#     typed = df[\"planet_id\"].dtype\n#     df[\"planet_id\"] = names_test\n#     df[\"planet_id\"] = df[\"planet_id\"].astype(typed)\n#     for i in range(Y_pred.shape[0]):\n#         df.loc[i,df.columns[1:].tolist()] = np.hstack((Y_pred[i][0],Y_pred[i],Y_sigma[i][0],Y_sigma[i])).T\n#     return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.77204Z","iopub.status.idle":"2025-08-28T02:55:02.772333Z","shell.execute_reply.started":"2025-08-28T02:55:02.772178Z","shell.execute_reply":"2025-08-28T02:55:02.772191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_dataframe(Y_pred, Y_sigma):\n    df = pd.read_csv(\"/kaggle/input/ariel-data-challenge-2025/sample_submission.csv\", index_col=\"planet_id\")\n    dg = pd.DataFrame(index=df.index, columns=df.columns)\n    for i, j in enumerate(df.index):\n        dg.loc[j] = np.hstack((Y_pred[i][0],Y_pred[i],Y_sigma[i][0],Y_sigma[i])).T\n    df.reset_index(inplace=True)\n    dg.reset_index(inplace=True)\n    for col in dg.columns:\n        dg[col] = dg[col].astype(df[col].dtype)\n    return dg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.773611Z","iopub.status.idle":"2025-08-28T02:55:02.773809Z","shell.execute_reply.started":"2025-08-28T02:55:02.773716Z","shell.execute_reply":"2025-08-28T02:55:02.773725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if SUBMISSION:\n    sub = make_dataframe(Y_pred, Y_sigma)\n    sub.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.774943Z","iopub.status.idle":"2025-08-28T02:55:02.775159Z","shell.execute_reply.started":"2025-08-28T02:55:02.775051Z","shell.execute_reply":"2025-08-28T02:55:02.77506Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Only run these when downloading the files and managing file size, otherwise leave commented!","metadata":{}},{"cell_type":"code","source":"#train_cds_binned = np.load(\"/kaggle/working/train.npy\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.776239Z","iopub.status.idle":"2025-08-28T02:55:02.776607Z","shell.execute_reply.started":"2025-08-28T02:55:02.77643Z","shell.execute_reply":"2025-08-28T02:55:02.776445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from IPython.display import FileLink\n#FileLink(r'names.npy')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.777699Z","iopub.status.idle":"2025-08-28T02:55:02.777939Z","shell.execute_reply.started":"2025-08-28T02:55:02.777837Z","shell.execute_reply":"2025-08-28T02:55:02.777846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from IPython.display import FileLink\n#FileLink(r'test.npy')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.779254Z","iopub.status.idle":"2025-08-28T02:55:02.779521Z","shell.execute_reply.started":"2025-08-28T02:55:02.779378Z","shell.execute_reply":"2025-08-28T02:55:02.779389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#os.remove(\"/kaggle/working/train.npy\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.780504Z","iopub.status.idle":"2025-08-28T02:55:02.780789Z","shell.execute_reply.started":"2025-08-28T02:55:02.780672Z","shell.execute_reply":"2025-08-28T02:55:02.780684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#np.save((\"/kaggle/working/train_part_1.npy\"), train_cds_binned[0:220])\n#np.save((\"/kaggle/working/train_part_2.npy\"), train_cds_binned[220:440])\n#np.save((\"/kaggle/working/train_part_3.npy\"), train_cds_binned[440:660])\n#np.save((\"/kaggle/working/train_part_4.npy\"), train_cds_binned[660:880])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.781595Z","iopub.status.idle":"2025-08-28T02:55:02.781847Z","shell.execute_reply.started":"2025-08-28T02:55:02.781739Z","shell.execute_reply":"2025-08-28T02:55:02.78175Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from IPython.display import FileLink\n#FileLink(r'train_part_1.npy')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.782703Z","iopub.status.idle":"2025-08-28T02:55:02.783363Z","shell.execute_reply.started":"2025-08-28T02:55:02.783198Z","shell.execute_reply":"2025-08-28T02:55:02.783214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from IPython.display import FileLink\n#FileLink(r'train_part_2.npy')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.784228Z","iopub.status.idle":"2025-08-28T02:55:02.78451Z","shell.execute_reply.started":"2025-08-28T02:55:02.784366Z","shell.execute_reply":"2025-08-28T02:55:02.784379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from IPython.display import FileLink\n#FileLink(r'train_part_3.npy')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.785427Z","iopub.status.idle":"2025-08-28T02:55:02.785728Z","shell.execute_reply.started":"2025-08-28T02:55:02.785579Z","shell.execute_reply":"2025-08-28T02:55:02.785594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from IPython.display import FileLink\n#FileLink(r'train_part_4.npy')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T02:55:02.786512Z","iopub.status.idle":"2025-08-28T02:55:02.786798Z","shell.execute_reply.started":"2025-08-28T02:55:02.786666Z","shell.execute_reply":"2025-08-28T02:55:02.786678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}