{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"path_folder = '/kaggle/input/ariel-data-challenge-2024/' # path to the folder containing the data\npath_out = '/kaggle/tmp/data_light_raw/' # path to the folder to store the light data\noutput_dir = '/kaggle/tmp/data_light_raw/' # path for the output directory","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-12T16:03:41.523227Z","iopub.execute_input":"2024-09-12T16:03:41.524322Z","iopub.status.idle":"2024-09-12T16:03:41.5292Z","shell.execute_reply.started":"2024-09-12T16:03:41.524253Z","shell.execute_reply":"2024-09-12T16:03:41.527986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os","metadata":{"execution":{"iopub.status.busy":"2024-09-12T16:03:41.852387Z","iopub.execute_input":"2024-09-12T16:03:41.853452Z","iopub.status.idle":"2024-09-12T16:03:41.857719Z","shell.execute_reply.started":"2024-09-12T16:03:41.853409Z","shell.execute_reply":"2024-09-12T16:03:41.856697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(path_out):\n    os.makedirs(path_out)\n    print(f\"Directory {path_out} created.\")\nelse:\n    print(f\"Directory {path_out} already exists.\")","metadata":{"execution":{"iopub.status.busy":"2024-09-12T16:03:42.222566Z","iopub.execute_input":"2024-09-12T16:03:42.223413Z","iopub.status.idle":"2024-09-12T16:03:42.229399Z","shell.execute_reply.started":"2024-09-12T16:03:42.223371Z","shell.execute_reply":"2024-09-12T16:03:42.228306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport os\nimport glob\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nCHUNKS_SIZE = 1\n\ndef get_index(files, chunk_size=CHUNKS_SIZE):\n    \"\"\"\n    Extracts and returns the index of relevant files based on their filenames.\n    \n    Parameters:\n    files (list): List of file paths.\n    chunk_size (int): Size of each chunk.\n    \n    Returns:\n    list: A list of indices split by chunk size.\n    \"\"\"\n    index = []\n    for file in files:\n        file_name = file.split('/')[-1]\n        if file_name.startswith('AIRS-CH0') and file_name.endswith('signal.parquet'):\n            file_index = os.path.basename(os.path.dirname(file))\n            index.append(int(file_index))\n    index = np.sort(np.array(index))\n    index = np.array_split(index, len(index) // chunk_size)\n    return index\n\ndef load_data(file: str, chunk_size: int, nb_files: int) -> np.ndarray:\n    \"\"\"\n    Loads and returns data from .npy files.\n    \n    Parameters:\n    file (str): Base file path (without index) for loading .npy files.\n    chunk_size (int): Number of samples per chunk.\n    nb_files (int): Number of files to load.\n    \n    Returns:\n    np.ndarray: A NumPy array containing the loaded dataset.\n    \"\"\"\n    try:\n        data0 = np.load(file + '_0.npy')\n    except FileNotFoundError as e:\n        raise FileNotFoundError(f\"File {file}_0.npy not found.\") from e\n    \n    data_all = np.zeros((nb_files * chunk_size, data0.shape[1], data0.shape[2], data0.shape[3]))\n    data_all[:chunk_size] = data0\n\n    for i in tqdm(range(1, nb_files), desc=\"Loading Data\"):\n        try:\n            data_chunk = np.load(file + f'_{i}.npy')\n            data_all[i * chunk_size:(i + 1) * chunk_size] = data_chunk\n        except FileNotFoundError as e:\n            raise FileNotFoundError(f\"File {file}_{i}.npy not found.\") from e\n    \n    return data_all\n\ndef save_data(file_name: str, data: np.ndarray):\n    \"\"\"\n    Saves the NumPy array data to disk with a given file name.\n    \n    Parameters:\n    file_name (str): Name of the file to save.\n    data (np.ndarray): The data to save.\n    \"\"\"\n    save_path = os.path.join('/kaggle/working/', file_name + '.npy')\n    np.save(save_path, data)\n    print(f\"Saved data to {save_path}\")\n\n# Example usage\ndef load_data (file, chunk_size, nb_files) : \n    data0 = np.load(file + '_0.npy')\n    data_all = np.zeros((nb_files*chunk_size, data0.shape[1], data0.shape[2], data0.shape[3]))\n    data_all[:chunk_size] = data0\n    for i in range (1, nb_files) : \n        data_all[i*chunk_size:(i+1)*chunk_size] = np.load(file + '_{}.npy'.format(i))\n    return data_all \n\n# Load datasets\ndata_train = load_data(path_out + 'AIRS_clean_train', CHUNKS_SIZE, len(index)) \ndata_train_FGS = load_data(path_out + 'FGS1_train', CHUNKS_SIZE, len(index))\n\n# Save datasets\nsave_data('data_train', data_train)\nsave_data('data_train_FGS', data_train_FGS)\n\n# Print dataset shapes\nprint('Training Dataset Shapes:')\nprint(f'AIRS-CH0 Shape: {data_train.shape}')\nprint(f'FGS1 Shape: {data_train_FGS.shape}')\n\n# Optionally, visualize some data using matplotlib\nplt.figure(figsize=(10, 5))\nplt.subplot(1, 2, 1)\nplt.title('AIRS-CH0 Sample')\nplt.imshow(data_train[0, :, :, 0], cmap='gray')\n\nplt.subplot(1, 2, 2)\nplt.title('FGS1 Sample')\nplt.imshow(data_train_FGS[0, :, :, 0], cmap='gray')\n\nplt.show()\n\nfor i in range(len(data_train)) : \n    light_curve = data_train[i,:,:,:].sum(axis=(1,2))\n    plt.plot(light_curve/light_curve.mean(), '-', alpha=0.3)\n\nplt.xlabel('Time (frame index)')\nplt.ylabel('Normalized flux in the frame')","metadata":{"execution":{"iopub.status.busy":"2024-09-12T16:03:42.623064Z","iopub.execute_input":"2024-09-12T16:03:42.623494Z","iopub.status.idle":"2024-09-12T16:03:43.730135Z","shell.execute_reply.started":"2024-09-12T16:03:42.623451Z","shell.execute_reply":"2024-09-12T16:03:43.729077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}