{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\nimport scipy.stats\nfrom tqdm import tqdm\nimport pickle\n\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.linear_model import Ridge\nfrom sklearn.metrics import r2_score, mean_squared_error","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-28T09:30:25.884581Z","iopub.execute_input":"2024-09-28T09:30:25.885295Z","iopub.status.idle":"2024-09-28T09:30:28.93206Z","shell.execute_reply.started":"2024-09-28T09:30:25.885253Z","shell.execute_reply":"2024-09-28T09:30:28.930771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 任务分析：\n\n**训练数据**：\n* 行星数量：673颗行星。\n* 预测label：需要预测的目标数量为283个(光谱分解)——多输出回归。\n\n**测试数据**：\n* 约800颗隐藏的行星用于测试。","metadata":{}},{"cell_type":"code","source":"train_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/train_adc_info.csv',index_col='planet_id')\n# test_adc_info = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/test_adc_info.csv',index_col='planet_id')\ntrain_labels = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/train_labels.csv',index_col='planet_id')\nwavelengths = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/wavelengths.csv')\naxis_info = pd.read_parquet('/kaggle/input/ariel-data-challenge-2024/axis_info.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:30:28.934513Z","iopub.execute_input":"2024-09-28T09:30:28.935421Z","iopub.status.idle":"2024-09-28T09:30:29.264905Z","shell.execute_reply.started":"2024-09-28T09:30:28.935304Z","shell.execute_reply":"2024-09-28T09:30:29.26367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## FGS1数据观察（暂时不用校准文件）\n\n**数据描述**\n\n* 每个文件包含135,000行图像，图像以0.1秒的时间间隔拍摄。每行是一个32x32的单波长图像。","metadata":{}},{"cell_type":"markdown","source":"取100468857号行星的FGS1数据进行观察","metadata":{}},{"cell_type":"code","source":"planet_id = 100468857\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/FGS1_signal.parquet')\nf_signal","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:30:29.266466Z","iopub.execute_input":"2024-09-28T09:30:29.266956Z","iopub.status.idle":"2024-09-28T09:30:31.572761Z","shell.execute_reply.started":"2024-09-28T09:30:29.266905Z","shell.execute_reply":"2024-09-28T09:30:31.571648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"将每一行数据恢复成32\\*32的像素点观察","metadata":{}},{"cell_type":"code","source":"# 取100468857号行星0时刻和1时刻的FGS1数据进行比较\n_, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 4))\nsns.heatmap(f_signal.iloc[0].values.reshape(32, 32), ax=ax1, vmin=0, vmax=52000)\nax1.set_aspect('equal')\nsns.heatmap(f_signal.iloc[1].values.reshape(32, 32), ax=ax2, vmin=0, vmax=52000)\nax2.set_aspect('equal')\nplt.suptitle('A pair of FGS1 images')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:30:31.57501Z","iopub.execute_input":"2024-09-28T09:30:31.575363Z","iopub.status.idle":"2024-09-28T09:30:32.484924Z","shell.execute_reply.started":"2024-09-28T09:30:31.575327Z","shell.execute_reply":"2024-09-28T09:30:32.48383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"观察时序变化","metadata":{}},{"cell_type":"code","source":"_, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2, 2, figsize=(12, 4))\n\n#变化比较明显的一个行星\nplanet_id = 100468857\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/FGS1_signal.parquet')\n\n#直接对每行图像取平均？是否合理？怎么改进？\nmean_signal = f_signal.values.mean(axis=1)\n# 奇数帧-偶数帧（观察数据好像两帧之间会有跳变）\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\n\n#滑动窗口平滑数据\nwindow=800\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\nax1.set_title('FGS1: time series of planet with strong signal')\nax1.plot(net_signal, label='raw signal')\nax1.legend()\nax3.plot(smooth_signal, color='c', label='smoothened signal')\nax3.legend()\nax3.set_xlabel('time step')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax3.axvline(time_step, color='gray')\n\n#变化没有那么明显的一个行星\nplanet_id = 4249337798\nf_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/FGS1_signal.parquet')\n\nmean_signal = f_signal.values.mean(axis=1)\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\nwindow=800\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\nax2.set_title('FGS1: time series of planet with weak signal')\nax2.plot(net_signal, label='raw signal')\nax2.legend()\nax4.plot(smooth_signal, color='c', label='smoothened signal')\nax4.legend()\nax4.set_xlabel('time step')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax4.axvline(time_step, color='gray')\n\n# plt.suptitle('FGS1 time series', y=0.96)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:30:32.486217Z","iopub.execute_input":"2024-09-28T09:30:32.486614Z","iopub.status.idle":"2024-09-28T09:30:36.960856Z","shell.execute_reply.started":"2024-09-28T09:30:32.486574Z","shell.execute_reply":"2024-09-28T09:30:36.959614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## AIRS 数据观察（暂时未校准）\n\n**数据描述**\n\n* 每个文件包含11,250行图像，图像以0.1秒的时间间隔拍摄。每行是一个32 x 356的单波长图像。","metadata":{}},{"cell_type":"markdown","source":"还是来观察100468857","metadata":{}},{"cell_type":"code","source":"planet_id = 100468857\na_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/AIRS-CH0_signal.parquet')\na_signal","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-09-28T11:10:58.825269Z","iopub.execute_input":"2024-09-28T11:10:58.825807Z","iopub.status.idle":"2024-09-28T11:11:00.259981Z","shell.execute_reply.started":"2024-09-28T11:10:58.825764Z","shell.execute_reply":"2024-09-28T11:11:00.258596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 取100468857号行星0时刻和1时刻的AIRS数据进行比较\n_, (ax1, ax2) = plt.subplots(2, 1, figsize=(15, 10))\n\nsns.heatmap(a_signal.iloc[0].values.reshape(32, 356), ax=ax1, vmin=0, vmax=52000)\nax1.set_title('AIRS Image at Time 0')\nax1.set_aspect('equal')\nax1.set_ylim(32, 0)\nax1.set_aspect('auto')\n\nsns.heatmap(a_signal.iloc[1].values.reshape(32, 356), ax=ax2, vmin=0, vmax=52000)\nax2.set_title('AIRS Image at Time 1')\nax2.set_aspect('equal')\nax2.set_ylim(32, 0)\nax2.set_aspect('auto') \n\nplt.suptitle('A Pair of AIRS Images')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T10:47:46.837316Z","iopub.execute_input":"2024-09-28T10:47:46.837789Z","iopub.status.idle":"2024-09-28T10:47:48.598902Z","shell.execute_reply.started":"2024-09-28T10:47:46.837746Z","shell.execute_reply":"2024-09-28T10:47:48.597752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import InputLayer, Conv2D\nfrom astropy.stats import sigma_clip\n# 构建 CNN 模型，用于对死点进行插值\ndef build_cnn_model():\n    model = Sequential()\n    model.add(InputLayer(input_shape=(None, None, 1)))\n    model.add(Conv2D(32, kernel_size=(3, 3), padding='same', activation='relu'))\n    model.add(Conv2D(64, kernel_size=(3, 3), padding='same', activation='relu'))\n    model.add(Conv2D(1, kernel_size=(3, 3), padding='same'))\n    model.compile(optimizer='adam', loss='mean_squared_error')\n    return model\n\ndef mask_hot_dead_single_frame(frame, dead, dark):\n    \"\"\"\n    针对单个帧（即二维图像数据）进行热点和死点的掩盖与插值。\n\n    参数：\n    - frame: ndarray, 单个帧数据，形状为 (32, 356) 或类似\n    - dead: ndarray, 死点掩码，形状与 frame 相同\n    - dark: ndarray, 暗噪声数据，形状与 frame 相同\n    \n    返回值：\n    - 填充插值后的帧数据，形状与输入帧一致\n    \"\"\"\n    # 识别热点和死点，并将其转换为布尔类型\n    hot = sigma_clip(dark, sigma=5, maxiters=5).mask.astype(bool)  # 将 `hot` 转换为布尔类型\n    dead = dead.astype(bool)  # 将 `dead` 转换为布尔类型\n\n    # 合并热点与死点掩码\n    combined_mask = dead\n\n    # 使用掩码遮盖信号中的热点和死点位置\n    frame_masked = np.ma.masked_where(combined_mask, frame).filled(np.nan)  # 用 `NaN` 表示缺失值\n\n    # 使用周围点均值进行插值\n    filled_frame = np.copy(frame_masked)\n    nan_indices = np.argwhere(np.isnan(filled_frame))  # 找出所有的 `NaN` 位置\n\n    for (i, j) in nan_indices:\n        # 找出邻近的四个点并计算均值（跳过超出边界的点）\n        neighbors = []\n        if j - 1 >= 0:  # 左\n            neighbors.append(filled_frame[i, j - 1])\n        if j + 1 < frame.shape[1]:  # 右\n            neighbors.append(filled_frame[i, j + 1])\n        \n        # 计算邻居均值，排除 `NaN` 值\n        neighbors = [val for val in neighbors if not np.isnan(val)]\n        if len_neighbors := len(neighbors):  # 如果邻居存在非 NaN 的值\n            filled_frame[i, j] = sum(neighbors) / len_neighbors\n\n    return filled_frame","metadata":{"execution":{"iopub.status.busy":"2024-09-28T11:34:46.697769Z","iopub.execute_input":"2024-09-28T11:34:46.698252Z","iopub.status.idle":"2024-09-28T11:34:46.711341Z","shell.execute_reply.started":"2024-09-28T11:34:46.698213Z","shell.execute_reply":"2024-09-28T11:34:46.710089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/FGS1_signal.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:40:13.564943Z","iopub.execute_input":"2024-09-28T09:40:13.565473Z","iopub.status.idle":"2024-09-28T09:40:15.789264Z","shell.execute_reply.started":"2024-09-28T09:40:13.565431Z","shell.execute_reply":"2024-09-28T09:40:15.788106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dark = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/100468857/AIRS-CH0_calibration/dark.parquet')\ndead = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/100468857/AIRS-CH0_calibration/dead.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-09-28T10:47:59.578206Z","iopub.execute_input":"2024-09-28T10:47:59.578689Z","iopub.status.idle":"2024-09-28T10:47:59.654693Z","shell.execute_reply.started":"2024-09-28T10:47:59.578626Z","shell.execute_reply":"2024-09-28T10:47:59.65345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dark","metadata":{"execution":{"iopub.status.busy":"2024-09-28T10:48:01.615524Z","iopub.execute_input":"2024-09-28T10:48:01.617065Z","iopub.status.idle":"2024-09-28T10:48:01.659739Z","shell.execute_reply.started":"2024-09-28T10:48:01.616993Z","shell.execute_reply":"2024-09-28T10:48:01.657997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dead","metadata":{"execution":{"iopub.status.busy":"2024-09-28T10:48:05.222296Z","iopub.execute_input":"2024-09-28T10:48:05.222762Z","iopub.status.idle":"2024-09-28T10:48:05.261421Z","shell.execute_reply.started":"2024-09-28T10:48:05.22272Z","shell.execute_reply":"2024-09-28T10:48:05.259946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dead.iloc[14:17,210:230]","metadata":{"execution":{"iopub.status.busy":"2024-09-28T10:48:11.801025Z","iopub.execute_input":"2024-09-28T10:48:11.801488Z","iopub.status.idle":"2024-09-28T10:48:11.822757Z","shell.execute_reply.started":"2024-09-28T10:48:11.801443Z","shell.execute_reply":"2024-09-28T10:48:11.821335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"p1=a_signal[1:2].values.reshape(32, 356)\np1 = p1.astype(float)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T11:34:52.917844Z","iopub.execute_input":"2024-09-28T11:34:52.918728Z","iopub.status.idle":"2024-09-28T11:34:52.927662Z","shell.execute_reply.started":"2024-09-28T11:34:52.91868Z","shell.execute_reply":"2024-09-28T11:34:52.926447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from astropy.stats import sigma_clip\np1=mask_hot_dead_single_frame(p1, dead, dark)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T11:34:54.604029Z","iopub.execute_input":"2024-09-28T11:34:54.604491Z","iopub.status.idle":"2024-09-28T11:34:54.615013Z","shell.execute_reply.started":"2024-09-28T11:34:54.604448Z","shell.execute_reply":"2024-09-28T11:34:54.613776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 取100468857号行星0时刻和1时刻的AIRS数据进行比较\n_, (ax1, ax2) = plt.subplots(2, 1, figsize=(15, 10))\n\nsns.heatmap(p1.reshape(32, 356), ax=ax1, vmin=0, vmax=52000)\nax1.set_title('AIRS Image at Time 0')\nax1.set_aspect('equal')\nax1.set_ylim(32, 0)\nax1.set_aspect('auto')\n\nsns.heatmap(p1.reshape(32, 356), ax=ax2, vmin=0, vmax=52000)\nax2.set_title('AIRS Image at Time 1')\nax2.set_aspect('equal')\nax2.set_ylim(32, 0)\nax2.set_aspect('auto') \n\nplt.suptitle('A Pair of AIRS Images')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T11:34:56.971694Z","iopub.execute_input":"2024-09-28T11:34:56.97216Z","iopub.status.idle":"2024-09-28T11:34:58.758918Z","shell.execute_reply.started":"2024-09-28T11:34:56.972118Z","shell.execute_reply":"2024-09-28T11:34:58.757666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"这个图里面似乎就有一个坏点（）","metadata":{}},{"cell_type":"markdown","source":"和上面fgs1的数据处理基本一致，观察时序图","metadata":{}},{"cell_type":"code","source":"_, ((ax1, ax2), (ax3, ax4)) = plt.subplots(2, 2, figsize=(12, 4))\n\n#还是取上面两个行星\nplanet_id = 100468857\na_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/AIRS-CH0_signal.parquet')\n\nmean_signal = a_signal.values.mean(axis=1)\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\nwindow=80\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\nax1.set_title('AIRS-CH0 time series of planet 100468857')\nax1.plot(net_signal, label='raw signal')\nax1.legend()\nax3.plot(smooth_signal, color='c', label='smoothened signal')\nax3.legend()\nax3.set_xlabel('time step')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax3.axvline(time_step* 11250 // 135000, color='gray')\n    \n\nplanet_id = 4249337798\na_signal = pd.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/train/{planet_id}/AIRS-CH0_signal.parquet')\n\nmean_signal = a_signal.values.mean(axis=1)\nnet_signal = mean_signal[1::2] - mean_signal[0::2]\ncum_signal = net_signal.cumsum()\nwindow=80\nsmooth_signal = (cum_signal[window:] - cum_signal[:-window]) / window\n\nax2.set_title('AIRS-CH0 time series of planet 4249337798')\nax2.plot(net_signal, label='raw signal')\nax2.legend()\nax4.plot(smooth_signal, color='c', label='smoothened signal')\nax4.legend()\nax4.set_xlabel('time step')\nfor time_step in [20500, 23500, 44000, 47000]:\n    ax4.axvline(time_step* 11250 // 135000, color='gray')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:30:40.952303Z","iopub.execute_input":"2024-09-28T09:30:40.952714Z","iopub.status.idle":"2024-09-28T09:30:45.489776Z","shell.execute_reply.started":"2024-09-28T09:30:40.952673Z","shell.execute_reply":"2024-09-28T09:30:45.488583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"为所有673个训练行星读取FGS1数据和AIRS-CH0数据。\n\n由于数据集无法完全放入RAM，我们仅保留每个行星的两个**一维时间序列**。即：\n\n* 从FGS1数据中提取的每个行星67500步的时间序列\n\n* 从AIRS-CH0数据中提取的每个行星5625步的时间序列\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nfrom tqdm import tqdm\n\ndef f_read_and_preprocess(dataset, adc_info, planet_ids):\n    \n#     读取所有行星ID的FGS1文件并提取时间序列。\n\n#     参数：\n#     dataset：'train' 或 'test'\n#     adc_info：元数据数据框，可能是 train_adc_info 或 test_adc_info\n#     planet_ids：行星ID列表\n\n#     返回：\n#     每个行星ID对应一行的 数据框，每行包含67500个值\n\n    f_raw_train = np.full((len(planet_ids), 67500), np.nan, dtype=np.float32)\n    for i, planet_id in tqdm(list(enumerate(planet_ids))):\n        f_signal = pl.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/{dataset}/{planet_id}/FGS1_signal.parquet')\n        mean_signal = f_signal.cast(pl.Int32).sum_horizontal().cast(pl.Float32).to_numpy() / 1024 \n        net_signal = mean_signal[1::2] - mean_signal[0::2]\n        f_raw_train[i] = net_signal\n    return f_raw_train","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:30:45.491729Z","iopub.execute_input":"2024-09-28T09:30:45.492652Z","iopub.status.idle":"2024-09-28T09:30:45.502536Z","shell.execute_reply.started":"2024-09-28T09:30:45.492594Z","shell.execute_reply":"2024-09-28T09:30:45.501283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nf_raw_train = f_read_and_preprocess('train', train_adc_info, train_labels.index)\nwith open('f_raw_train.pickle', 'wb') as f:\n    pickle.dump(f_raw_train, f)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:30:45.506642Z","iopub.execute_input":"2024-09-28T09:30:45.507986Z","iopub.status.idle":"2024-09-28T09:32:47.478858Z","shell.execute_reply.started":"2024-09-28T09:30:45.507921Z","shell.execute_reply":"2024-09-28T09:32:47.477423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nfrom tqdm import tqdm\n\ndef a_read_and_preprocess(dataset, adc_info, planet_ids):\n    \n#     读取所有行星ID的AIRS-CH0文件并提取时间序列。\n#     参数：\n#     dataset：'train' 或 'test'\n#     adc_info：元数据数据框，可能是 train_adc_info 或 test_adc_info\n#     planet_ids：行星ID列表\n\n#     返回：\n#     每个行星ID对应一行的 数据框，每行包含5625个值\n\n    a_raw_train = np.full((len(planet_ids), 5625), np.nan, dtype=np.float32)\n    for i, planet_id in tqdm(list(enumerate(planet_ids))):\n        signal = pl.read_parquet(f'/kaggle/input/ariel-data-challenge-2024/{dataset}/{planet_id}/AIRS-CH0_signal.parquet')\n        mean_signal = signal.cast(pl.Int32).sum_horizontal().cast(pl.Float32).to_numpy() / (32*356) # 对32*356个像素求均值\n        net_signal = mean_signal[1::2] - mean_signal[0::2]\n        a_raw_train[i] = net_signal\n    return a_raw_train","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:32:47.480748Z","iopub.execute_input":"2024-09-28T09:32:47.481137Z","iopub.status.idle":"2024-09-28T09:32:47.489656Z","shell.execute_reply.started":"2024-09-28T09:32:47.481098Z","shell.execute_reply":"2024-09-28T09:32:47.488331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\na_raw_train = a_read_and_preprocess('train', train_adc_info, train_labels.index)\nwith open('a_raw_train.pickle', 'wb') as f:\n    pickle.dump(a_raw_train, f)","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:32:47.491533Z","iopub.execute_input":"2024-09-28T09:32:47.492059Z","iopub.status.idle":"2024-09-28T09:32:48.598911Z","shell.execute_reply.started":"2024-09-28T09:32:47.492008Z","shell.execute_reply":"2024-09-28T09:32:48.597774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"全数据观察","metadata":{}},{"cell_type":"code","source":"f_raw_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:32:48.600323Z","iopub.execute_input":"2024-09-28T09:32:48.600807Z","iopub.status.idle":"2024-09-28T09:32:48.775337Z","shell.execute_reply.started":"2024-09-28T09:32:48.600755Z","shell.execute_reply":"2024-09-28T09:32:48.773253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(6, 2))\nplt.plot(f_raw_train.mean(axis=0))\nfor time_step in [20500, 23500, 44000, 47000]:\n    plt.axvline(time_step, color='gray')\nplt.xlabel('time step')\nplt.title('FGS1: Overall mean')\nplt.show()\n\nplt.figure(figsize=(6, 2))\nplt.plot(a_raw_train.mean(axis=0))\nfor time_step in [20500, 23500, 44000, 47000]:\n    plt.axvline(time_step * 11250 // 135000, color='gray')\nplt.xlabel('time step')\nplt.title('AIRS-CH0: Overall mean')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T09:32:48.776625Z","iopub.status.idle":"2024-09-28T09:32:48.777072Z","shell.execute_reply.started":"2024-09-28T09:32:48.776869Z","shell.execute_reply":"2024-09-28T09:32:48.776889Z"},"trusted":true},"execution_count":null,"outputs":[]}]}