{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. Introduction\n\nBasic first day exploratory analysis. \nWriting functions to show images over time and load data. \n\nQuestions raised\n1. What's up with nulls in axis_info.parquet?\n2. Each planet's 32x32 image seem to have large uninformative background. Removing this background may help(?) \n3. Each planet is recored over 3.75hrs (135,000x0.1second). Similar to point 2, 'background' data (where it's just the star) may be uninformative. \n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nfrom plotly.subplots import make_subplots\nimport numpy as np\nimport seaborn as sns\nimport scipy.stats\nfrom tqdm import tqdm\nimport pickle\nimport plotly.graph_objects as go\n\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import r2_score, mean_squared_error","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-02T20:44:06.745501Z","iopub.execute_input":"2024-09-02T20:44:06.74595Z","iopub.status.idle":"2024-09-02T20:44:06.754394Z","shell.execute_reply.started":"2024-09-02T20:44:06.745909Z","shell.execute_reply":"2024-09-02T20:44:06.75315Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/train_adc_info.csv',\n                           index_col='planet_id')\n# test_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/test_adc_info.csv',\n#                            index_col='planet_id')\ntrain_labels = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/train_labels.csv',\n                           index_col='planet_id')\nwavelengths = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/wavelengths.csv')\naxis_info = pd.read_parquet('/kaggle/input/ariel-data-challenge-2024/axis_info.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-09-02T19:43:49.351936Z","iopub.execute_input":"2024-09-02T19:43:49.352677Z","iopub.status.idle":"2024-09-02T19:43:49.71377Z","shell.execute_reply.started":"2024-09-02T19:43:49.352614Z","shell.execute_reply":"2024-09-02T19:43:49.712652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_null_distributions(df_train, top_n=30):\n    # Calculate null counts\n    nan_count = df_train.isna().sum()\n    nan_count = nan_count[nan_count > 0].sort_values(ascending=False)\n    \n    if nan_count.empty:\n        print(\"There are no null values in the dataset.\")\n        return\n    \n    # Print feature with most NaNs\n    print(f\"Feature with most NaNs: {nan_count.index[0]} with {nan_count.iloc[0]}\")\n    \n    # Determine the number of features to display\n    n_features = min(top_n, len(nan_count))\n    \n    # Create the plot\n    fig, axs = plt.subplots(figsize=(12, 10))\n    \n    # Plot the bar chart\n    bars = sns.barplot(y=nan_count.index[:n_features], \n                       x=nan_count.values[:n_features], \n                       alpha=0.8)\n    \n    # Add labels to the bars\n    for i, v in enumerate(nan_count.values[:n_features]):\n        axs.text(v + 3, i, str(v), va='center')\n    \n    # Set title and labels\n    plt.title('Number of NaNs')\n    plt.xlabel('Number of NaNs')\n    plt.tight_layout()\n    \n    # Show the plot\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-02T19:49:13.976164Z","iopub.execute_input":"2024-09-02T19:49:13.976595Z","iopub.status.idle":"2024-09-02T19:49:13.985069Z","shell.execute_reply.started":"2024-09-02T19:49:13.976553Z","shell.execute_reply":"2024-09-02T19:49:13.983984Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.1 null distribution","metadata":{}},{"cell_type":"code","source":"file_list = train_adc_info, train_labels, wavelengths, axis_info\nfor file in file_list:\n    \n    plot_null_distributions(file)","metadata":{"execution":{"iopub.status.busy":"2024-09-02T19:56:29.482652Z","iopub.execute_input":"2024-09-02T19:56:29.483127Z","iopub.status.idle":"2024-09-02T19:56:29.855642Z","shell.execute_reply.started":"2024-09-02T19:56:29.483087Z","shell.execute_reply":"2024-09-02T19:56:29.854449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for file in file_list:\n    print('len(file)', len(file))\n    print(file.head())","metadata":{"execution":{"iopub.status.busy":"2024-09-02T19:59:28.495972Z","iopub.execute_input":"2024-09-02T19:59:28.496505Z","iopub.status.idle":"2024-09-02T19:59:28.528517Z","shell.execute_reply.started":"2024-09-02T19:59:28.496463Z","shell.execute_reply":"2024-09-02T19:59:28.526988Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like a lot of numerical features.","metadata":{}},{"cell_type":"markdown","source":"# 2. Visualisations","metadata":{}},{"cell_type":"code","source":"# take some sample planet_id\nsample_planet_id = train_adc_info.index[0:2].tolist()+train_adc_info.index[27:30].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-09-02T20:05:03.882241Z","iopub.execute_input":"2024-09-02T20:05:03.882764Z","iopub.status.idle":"2024-09-02T20:05:03.890918Z","shell.execute_reply.started":"2024-09-02T20:05:03.882721Z","shell.execute_reply":"2024-09-02T20:05:03.889528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1 Simple show 32x32 image","metadata":{}},{"cell_type":"code","source":"def show_image(FGS1_signal):\n    _, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 4))\n    sns.heatmap(FGS1_signal.iloc[6666].values.reshape(32, 32), ax=ax1, vmin=0, vmax=52000)\n    ax1.set_aspect('equal')\n    sns.heatmap(FGS1_signal.iloc[6667].values.reshape(32, 32), ax=ax2, vmin=0, vmax=52000)\n    ax2.set_aspect('equal')\n    plt.suptitle('A pair of FGS1 images')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-02T20:26:50.280654Z","iopub.execute_input":"2024-09-02T20:26:50.281234Z","iopub.status.idle":"2024-09-02T20:26:50.290516Z","shell.execute_reply.started":"2024-09-02T20:26:50.281186Z","shell.execute_reply":"2024-09-02T20:26:50.288631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for planet_id in sample_planet_id[:3]:\n    f_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/FGS1_signal.parquet')\n    show_image(f_signal)","metadata":{"execution":{"iopub.status.busy":"2024-09-02T20:26:50.478023Z","iopub.execute_input":"2024-09-02T20:26:50.478548Z","iopub.status.idle":"2024-09-02T20:26:54.954306Z","shell.execute_reply.started":"2024-09-02T20:26:50.478499Z","shell.execute_reply":"2024-09-02T20:26:54.952812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Stars seem to be centered. I wonder if this was postprocess or camera being centered automatically. Additionally, most of the image seem to be near zero outside the center, giving zero information(?). Background removing algorithms may help.\n","metadata":{}},{"cell_type":"markdown","source":"## 2.2 Simple play 32x32 image video","metadata":{}},{"cell_type":"code","source":"## Visualize 32x32 image over time(rows) \n## downsample_factor = 1 loads all 135000 images. NOT RECOMMENDED.\n## I recommend downsample_factor > 1000 which loads 135000/1000=135 images\ndef visualize_images_over_time(f_images, downsample_factor=1000):\n    # Downsample the data\n    if downsample_factor > 1:\n        f_images = f_images.iloc[::downsample_factor, :]\n    \n    def create_figure(frame):\n        img = f_images.iloc[frame].values.reshape(32, 32)\n        fig = go.Figure(data=go.Heatmap(z=img, colorscale='Viridis'))\n        fig.update_layout(\n            title=f'Image at time t={frame * downsample_factor}',\n            xaxis=dict(showticklabels=False),\n            yaxis=dict(showticklabels=False),\n        )\n        return fig\n\n    # Create the initial figure\n    fig = create_figure(0)\n\n    # Create and add frames\n    frames = [go.Frame(data=[go.Heatmap(z=f_images.iloc[i].values.reshape(32, 32))],\n                       layout=go.Layout(title=f'Image at time t={i * downsample_factor}'),\n                       name=str(i))\n              for i in range(len(f_images))]\n    fig.frames = frames\n\n    # Add slider\n    sliders = [dict(\n        active=0,\n        currentvalue=dict(prefix=\"Time: \", visible=True),\n        pad=dict(t=50),\n        steps=[dict(\n            method='animate',\n            args=[[str(i)], dict(mode='immediate', frame=dict(duration=0, redraw=True), transition=dict(duration=0))],\n            label=str(i * downsample_factor)\n        ) for i in range(len(f_images))]\n    )]\n\n    # Add play and pause buttons\n    updatemenus = [dict(\n        type='buttons',\n        showactive=False,\n        buttons=[dict(label='Play',\n                      method='animate',\n                      args=[None, dict(frame=dict(duration=50, redraw=True), fromcurrent=True, transition=dict(duration=0))]),\n                 dict(label='Pause',\n                      method='animate',\n                      args=[[None], dict(frame=dict(duration=0, redraw=False), mode='immediate', transition=dict(duration=0))])]\n    )]\n\n    # Update layout\n    fig.update_layout(\n        updatemenus=updatemenus,\n        sliders=sliders\n    )\n\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-02T20:45:31.734285Z","iopub.execute_input":"2024-09-02T20:45:31.7347Z","iopub.status.idle":"2024-09-02T20:45:31.747172Z","shell.execute_reply.started":"2024-09-02T20:45:31.734663Z","shell.execute_reply":"2024-09-02T20:45:31.745958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"planet_id = sample_planet_id[0]\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/FGS1_signal.parquet')\nvisualize_images_over_time(f_signal, downsample_factor=1000)","metadata":{"execution":{"iopub.status.busy":"2024-09-02T20:45:31.913261Z","iopub.execute_input":"2024-09-02T20:45:31.914205Z","iopub.status.idle":"2024-09-02T20:45:32.58919Z","shell.execute_reply.started":"2024-09-02T20:45:31.914161Z","shell.execute_reply":"2024-09-02T20:45:32.587984Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This visualisation doesn't really give much information other than observing the max brightness increase and decrease overtime. It can be improved by decreasing downsample_factor AND focusing around transit duration. Therefore TODO","metadata":{}},{"cell_type":"markdown","source":"# Reference\n- [ADC24 Intro training ⭐️⭐️⭐️⭐️⭐️](https://www.kaggle.com/code/ambrosm/adc24-intro-training/notebook)","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}