{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30762,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <div align=\"center\" style=\"padding: 20px; background-color: #d2e4dc; color: #f99393d2; border-radius: 80px;\">Imports</div>","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport dask.dataframe as dd\nimport dask.array as da\nfrom dask.distributed import Client, LocalCluster","metadata":{"execution":{"iopub.status.busy":"2024-08-25T12:15:57.163903Z","iopub.execute_input":"2024-08-25T12:15:57.164272Z","iopub.status.idle":"2024-08-25T12:15:57.169985Z","shell.execute_reply.started":"2024-08-25T12:15:57.164238Z","shell.execute_reply":"2024-08-25T12:15:57.168956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div align=\"center\" style=\"padding: 20px; background-color: #d2e4dc; color: #f99393d2; border-radius: 80px;\">Load Dataset</div>","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Set up Dask client with local cluster\ncluster = LocalCluster(n_workers=4, threads_per_worker=2)  # Adjust based on your CPU\nclient = Client(cluster)\n\n# Define base path\nBASE_PATH = '/kaggle/input/ariel-data-challenge-2024'\n\n# Function to load metadata files\ndef load_metadata():\n    metadata = {\n        'axis_info': dd.read_parquet(f'{BASE_PATH}/axis_info.parquet'),\n        'train_adc_info': dd.read_csv(f'{BASE_PATH}/train_adc_info.csv'),\n        'train_labels': dd.read_csv(f'{BASE_PATH}/train_labels.csv'),\n        'wavelengths': dd.read_csv(f'{BASE_PATH}/wavelengths.csv')\n    }\n    return metadata\n\n# Function to load signal data for a specific planet\ndef load_signal_data(planet_id):\n    signals = {\n        'FGS1': dd.read_parquet(f'{BASE_PATH}/train/{planet_id}/FGS1_signal.parquet'),\n        'AIRS-CH0': dd.read_parquet(f'{BASE_PATH}/train/{planet_id}/AIRS-CH0_signal.parquet')\n    }\n    return signals\n\n# Function to load calibration data\ndef load_calibration_data(planet_id, instrument):\n    calibration_files = ['dead', 'linear_corr', 'read', 'flat', 'dark']\n    calibration_data = {}\n    for file in calibration_files:\n        path = f'{BASE_PATH}/train/{planet_id}/{instrument}_calibration/{file}.parquet'\n        calibration_data[file] = dd.read_parquet(path)\n    return calibration_data\n\n# Function to get all planet IDs\ndef get_planet_ids():\n    return [os.path.basename(f) for f in glob.glob(f'{BASE_PATH}/train/*') if os.path.isdir(f)]\n\n# Function to process a single planet's data\ndef process_planet_data(planet_id):\n    signals = load_signal_data(planet_id)\n    \n    # Example processing: calculate mean and std for each signal\n    results = {}\n    for instrument, signal in signals.items():\n        # Process only a subset of columns for quicker computation\n        subset = signal.iloc[:, :10]  # First 10 columns\n        mean = subset.mean().compute()\n        std = subset.std().compute()\n        results[instrument] = {'mean': mean.mean(), 'std': std.mean()}\n    \n    return results\n\n# Main function to process all planets\ndef process_all_planets():\n    planet_ids = get_planet_ids()\n    results = {}\n    \n    for planet_id in planet_ids[:3]:  # Process only first 3 planets for testing\n        results[planet_id] = process_planet_data(planet_id)\n    \n    return results","metadata":{"execution":{"iopub.status.busy":"2024-08-25T11:45:16.144235Z","iopub.execute_input":"2024-08-25T11:45:16.144892Z","iopub.status.idle":"2024-08-25T11:45:23.034379Z","shell.execute_reply.started":"2024-08-25T11:45:16.144852Z","shell.execute_reply":"2024-08-25T11:45:23.033358Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Dask Client Setup:\n```python\ncluster = LocalCluster(n_workers=4, threads_per_worker=2)\nclient = Client(cluster)\n```\n### Dask is used to handle large datasets efficiently. The LocalCluster sets up a distributed computing environment on your local machine. We're using 4 workers with 2 threads each, which allows for parallel processing and can significantly speed up computations. This setup can be adjusted based on your CPU's capabilities.\n\n## 2. Base Path Definition:\n```python\nBASE_PATH = '/kaggle/input/ariel-data-challenge-2024'\n```\n### This centralizes the location of all data files, making it easier to manage and update paths if needed.\n\n## 3. Metadata Loading Function:\n```python\ndef load_metadata():\n    metadata = {\n        'axis_info': dd.read_parquet(f'{BASE_PATH}/axis_info.parquet'),\n        'train_adc_info': dd.read_csv(f'{BASE_PATH}/train_adc_info.csv'),\n        'train_labels': dd.read_csv(f'{BASE_PATH}/train_labels.csv'),\n        'wavelengths': dd.read_csv(f'{BASE_PATH}/wavelengths.csv')\n    }\n    return metadata\n```\n### This function loads all metadata files at once. It uses Dask (dd) to read both CSV and Parquet files, which allows for lazy loading of large files. This approach keeps memory usage low until the data is actually needed.\n\n## 4. Signal Data Loading Function:\n```python\ndef load_signal_data(planet_id):\n    signals = {\n        'FGS1': dd.read_parquet(f'{BASE_PATH}/train/{planet_id}/FGS1_signal.parquet'),\n        'AIRS-CH0': dd.read_parquet(f'{BASE_PATH}/train/{planet_id}/AIRS-CH0_signal.parquet')\n    }\n    return signals\n```\n### This function loads signal data for a specific planet. It uses Dask to read Parquet files, which is efficient for large datasets. Loading is done per planet to manage memory usage and allow for parallel processing.\n\n## 5. Calibration Data Loading Function:\n```python\ndef load_calibration_data(planet_id, instrument):\n    calibration_files = ['dead', 'linear_corr', 'read', 'flat', 'dark']\n    calibration_data = {}\n    for file in calibration_files:\n        path = f'{BASE_PATH}/train/{planet_id}/{instrument}_calibration/{file}.parquet'\n        calibration_data[file] = dd.read_parquet(path)\n    return calibration_data\n```\n### Reasoning: Similar to signal data, calibration data is loaded per planet and instrument. This modular approach allows for flexible processing and memory management.\n\n## 6. Planet ID Retrieval Function:\n```python\ndef get_planet_ids():\n    return [os.path.basename(f) for f in glob.glob(f'{BASE_PATH}/train/*') if os.path.isdir(f)]\n```\n### This function dynamically retrieves all planet IDs by looking at the directory structure. It's flexible and will work even if new planets are added to the dataset.\n\n## 7. Planet Data Processing Function:\n```python\ndef process_planet_data(planet_id):\n    signals = load_signal_data(planet_id)\n    \n    results = {}\n    for instrument, signal in signals.items():\n        subset = signal.iloc[:, :10]  # First 10 columns\n        mean = subset.mean().compute()\n        std = subset.std().compute()\n        results[instrument] = {'mean': mean.mean(), 'std': std.mean()}\n    \n    return results\n```\n### This function processes data for a single planet. It:\n- Loads signal data for the planet\n- Processes only the first 10 columns as a subset to speed up initial analysis\n- Computes mean and standard deviation, which are basic but informative statistics\n- Uses Dask's lazy evaluation (only computing when .compute() is called) to manage memory efficiently\n\n## 8. All Planets Processing Function:\n```python\ndef process_all_planets():\n    planet_ids = get_planet_ids()\n    results = {}\n    \n    for planet_id in planet_ids[:3]:  # Process only first 3 planets for testing\n        results[planet_id] = process_planet_data(planet_id)\n    \n    return results\n```\n### This function orchestrates the processing of multiple planets. It:\n- Retrieves all planet IDs\n- Processes only the first 3 planets for initial testing (to manage processing time and resource usage)\n- Calls process_planet_data for each planet, collecting results in a dictionary","metadata":{}},{"cell_type":"code","source":"metadata = load_metadata()\nresults = process_all_planets()\n\nprint(\"Processing complete. Results summary:\")\nfor planet_id, data in results.items():\n    print(f\"Planet {planet_id}:\")\n    for instrument, stats in data.items():\n        print(f\"  {instrument}: Mean = {stats['mean']:.2f}, Std = {stats['std']:.2f}\")\n\nclient.close()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T11:45:28.834495Z","iopub.execute_input":"2024-08-25T11:45:28.83544Z","iopub.status.idle":"2024-08-25T11:46:03.977035Z","shell.execute_reply.started":"2024-08-25T11:45:28.83539Z","shell.execute_reply":"2024-08-25T11:46:03.976126Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dataset_structure():\n    print(\"Dataset Structure:\")\n    for root, dirs, files in os.walk(BASE_PATH):\n        level = root.replace(BASE_PATH, '').count(os.sep)\n        indent = ' ' * 4 * level\n        print(f\"{indent}{os.path.basename(root)}/\")\n        if level < 2:  # Limit depth to keep output manageable\n            for file in files:\n                print(f\"{indent}    {file}\")","metadata":{"execution":{"iopub.status.busy":"2024-08-25T12:12:13.986563Z","iopub.execute_input":"2024-08-25T12:12:13.986936Z","iopub.status.idle":"2024-08-25T12:12:13.99379Z","shell.execute_reply.started":"2024-08-25T12:12:13.986901Z","shell.execute_reply":"2024-08-25T12:12:13.992445Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_dataset_structure()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T12:12:16.523428Z","iopub.execute_input":"2024-08-25T12:12:16.523806Z","iopub.status.idle":"2024-08-25T12:12:18.193328Z","shell.execute_reply.started":"2024-08-25T12:12:16.523771Z","shell.execute_reply":"2024-08-25T12:12:18.19244Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def analyze_metadata():\n    print(\"\\nAnalyzing Metadata Files:\")\n    metadata_files = ['axis_info.parquet', 'train_adc_info.csv', 'train_labels.csv', 'wavelengths.csv']\n    \n    for file in metadata_files:\n        path = os.path.join(BASE_PATH, file)\n        if file.endswith('.csv'):\n            df = pd.read_csv(path)\n        else:\n            df = pd.read_parquet(path)\n        \n        print(f\"\\n{file}:\")\n        print(f\"Shape: {df.shape}\")\n        print(f\"Columns: {df.columns.tolist()}\")\n        print(df.describe())","metadata":{"execution":{"iopub.status.busy":"2024-08-25T12:14:08.000189Z","iopub.execute_input":"2024-08-25T12:14:08.000568Z","iopub.status.idle":"2024-08-25T12:14:08.008785Z","shell.execute_reply.started":"2024-08-25T12:14:08.000513Z","shell.execute_reply":"2024-08-25T12:14:08.007733Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"analyze_metadata()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T12:14:09.22548Z","iopub.execute_input":"2024-08-25T12:14:09.22627Z","iopub.status.idle":"2024-08-25T12:14:10.584354Z","shell.execute_reply.started":"2024-08-25T12:14:09.226231Z","shell.execute_reply":"2024-08-25T12:14:10.583125Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def analyze_signal_data(num_planets=3):\n    print(\"\\nAnalyzing Signal Data (Sample):\")\n    planet_dirs = glob.glob(os.path.join(BASE_PATH, 'train', '*'))[:num_planets]\n    \n    for planet_dir in planet_dirs:\n        planet_id = os.path.basename(planet_dir)\n        print(f\"\\nPlanet ID: {planet_id}\")\n        \n        for instrument in ['FGS1', 'AIRS-CH0']:\n            file_path = os.path.join(planet_dir, f\"{instrument}_signal.parquet\")\n            df = dd.read_parquet(file_path)\n            \n            print(f\"  {instrument}:\")\n            print(f\"    Shape: {df.shape[0].compute()} rows, {df.shape[1]} columns\")\n            \n            # Compute basic stats on the first column (assuming it's representative)\n            stats = df.iloc[:, 0].describe().compute()\n            print(f\"    Stats of first column:\")\n            print(stats)\n            \n            # Plot histogram of the first column\n            plt.figure(figsize=(10, 5))\n            df.iloc[:, 0].sample(frac=0.01).compute().hist(bins=50)\n            plt.title(f\"{instrument} Signal Distribution (First Column)\")\n            plt.xlabel(\"Signal Value\")\n            plt.ylabel(\"Frequency\")\n            plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T12:14:36.867104Z","iopub.execute_input":"2024-08-25T12:14:36.867458Z","iopub.status.idle":"2024-08-25T12:14:36.877528Z","shell.execute_reply.started":"2024-08-25T12:14:36.867423Z","shell.execute_reply":"2024-08-25T12:14:36.876375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"analyze_signal_data()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T12:14:37.677611Z","iopub.execute_input":"2024-08-25T12:14:37.67795Z","iopub.status.idle":"2024-08-25T12:15:14.289652Z","shell.execute_reply.started":"2024-08-25T12:14:37.677915Z","shell.execute_reply":"2024-08-25T12:15:14.288716Z"},"trusted":true},"execution_count":null,"outputs":[]}]}