{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":70367,"databundleVersionId":9188054,"sourceType":"competition"},{"sourceId":200710889,"sourceType":"kernelVersion"},{"sourceId":200732670,"sourceType":"kernelVersion"},{"sourceId":200739204,"sourceType":"kernelVersion"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q --no-index --find-links /kaggle/input/astropy astropy\n!python /kaggle/input/preprocessing-for-adc/main.py test","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:32:07.830962Z","iopub.execute_input":"2024-10-14T10:32:07.831826Z","iopub.status.idle":"2024-10-14T10:32:37.769914Z","shell.execute_reply.started":"2024-10-14T10:32:07.831752Z","shell.execute_reply":"2024-10-14T10:32:37.76875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.modules[__name__].__dict__.clear()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:32:53.038299Z","iopub.execute_input":"2024-10-14T10:32:53.039054Z","iopub.status.idle":"2024-10-14T10:32:53.048697Z","shell.execute_reply.started":"2024-10-14T10:32:53.039008Z","shell.execute_reply":"2024-10-14T10:32:53.047743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\nimport scipy.stats\nfrom tqdm import tqdm\n\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.metrics import r2_score, mean_squared_error","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-14T10:32:54.197785Z","iopub.execute_input":"2024-10-14T10:32:54.198226Z","iopub.status.idle":"2024-10-14T10:32:55.448457Z","shell.execute_reply.started":"2024-10-14T10:32:54.198175Z","shell.execute_reply":"2024-10-14T10:32:55.447474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"airs = np.load(\"/kaggle/working/data_test.npy\").reshape(-1, 187, 9024)\nfgs = np.load(\"/kaggle/working/data_test_FGS.npy\").reshape(-1, 187, 1024)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:40:47.759829Z","iopub.execute_input":"2024-10-14T10:40:47.760563Z","iopub.status.idle":"2024-10-14T10:40:47.770194Z","shell.execute_reply.started":"2024-10-14T10:40:47.760519Z","shell.execute_reply":"2024-10-14T10:40:47.769274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fgs_max, fgs_min, airs_max, airs_min = fgs.max(axis=None), fgs.min(axis=None), airs.max(axis=None), airs.min(axis=None)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:40:47.957322Z","iopub.execute_input":"2024-10-14T10:40:47.957875Z","iopub.status.idle":"2024-10-14T10:40:47.965313Z","shell.execute_reply.started":"2024-10-14T10:40:47.957828Z","shell.execute_reply":"2024-10-14T10:40:47.964293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fgs = 0.30 + ((fgs - fgs_min)/(fgs_max - fgs_min))*0.50\nairs = 1.95 + ((airs - airs_min)/(airs_max - airs_min))*1.95\ndataset = np.concatenate((fgs, airs), axis=2)\ndel airs, fgs","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:40:51.452405Z","iopub.execute_input":"2024-10-14T10:40:51.452832Z","iopub.status.idle":"2024-10-14T10:40:51.467569Z","shell.execute_reply.started":"2024-10-14T10:40:51.452772Z","shell.execute_reply":"2024-10-14T10:40:51.46658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = (dataset - 0.30)/(3.90-0.30)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:40:57.982406Z","iopub.execute_input":"2024-10-14T10:40:57.98323Z","iopub.status.idle":"2024-10-14T10:40:57.991422Z","shell.execute_reply.started":"2024-10-14T10:40:57.98319Z","shell.execute_reply":"2024-10-14T10:40:57.990347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(dataset[0])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:00.894974Z","iopub.execute_input":"2024-10-14T10:41:00.895366Z","iopub.status.idle":"2024-10-14T10:41:03.206031Z","shell.execute_reply.started":"2024-10-14T10:41:00.895329Z","shell.execute_reply":"2024-10-14T10:41:03.205035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # detect and init the TPU\n    resolver = tf.distribute.cluster_resolver.TPUClusterResolver(tpu='local')\n    tf.config.experimental_connect_to_cluster(resolver)\n    # This is the TPU initialization code that has to be at the beginning.\n    tf.tpu.experimental.initialize_tpu_system(resolver)\n    print(\"All devices: \", tf.config.list_logical_devices('TPU'))\n    strategy = tf.distribute.TPUStrategy(resolver)\n    print(\"Running on TPU\")\n    print(\"REPLICAS: \", strategy.num_replicas_in_sync)\n    device = \"tpu\"\nexcept tf.errors.NotFoundError:\n    print(\"Not on TPU\")\n    try:\n        strategy = tf.distribute.MirroredStrategy()\n        print(\"Number of devices: {}\".format(strategy.num_replicas_in_sync))\n        print(\"Running on GPU\")\n        device = \"gpu\"\n    except:\n        print(\"On CPU\")","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:35:03.586254Z","iopub.execute_input":"2024-10-14T10:35:03.58664Z","iopub.status.idle":"2024-10-14T10:35:15.346353Z","shell.execute_reply.started":"2024-10-14T10:35:03.586601Z","shell.execute_reply":"2024-10-14T10:35:15.345387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Input, Conv1D, MaxPooling1D, Dense, Dropout, BatchNormalization, Concatenate, AveragePooling1D, SpatialDropout1D\nfrom keras.models import Model, load_model\nfrom keras.layers import GlobalAveragePooling1D, Layer\nfrom tensorflow.keras.optimizers import Adam, SGD\nfrom tensorflow.keras.callbacks import LearningRateScheduler, ModelCheckpoint\n    \nwith strategy.scope():\n    input_wc = Input((10048, 187))\n    x = Conv1D(32, 7, 2, activation='relu', padding='same')(input_wc)\n    x = Conv1D(32, 3, activation='relu', padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Conv1D(64, 7, 2,activation='relu', padding='same')(x)\n    x = Conv1D(64, 3, activation='relu', padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Conv1D(128, 7, 2,activation='relu', padding='same')(x)\n    x = Conv1D(128, 3, activation='relu', padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Conv1D(256, 7, 2,activation='relu', padding='same')(x)\n    x = Conv1D(256, 3, activation='relu', padding='same')(x)\n    x = BatchNormalization()(x)\n    x = GlobalAveragePooling1D()(x)\n    output_wc = Dense(2, activation='linear')(x)\n    model_wc = Model(inputs=input_wc, outputs=output_wc)\n    model_wc.summary()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:06.025182Z","iopub.execute_input":"2024-10-14T10:41:06.025573Z","iopub.status.idle":"2024-10-14T10:41:06.635279Z","shell.execute_reply.started":"2024-10-14T10:41:06.025536Z","shell.execute_reply":"2024-10-14T10:41:06.634277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_wc.load_weights(\"/kaggle/input/my-adc-training/model_1dcnn.keras\")\nmean_and_std = model_wc.predict(dataset.transpose(0, 2, 1))","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:06.637267Z","iopub.execute_input":"2024-10-14T10:41:06.637963Z","iopub.status.idle":"2024-10-14T10:41:08.50916Z","shell.execute_reply.started":"2024-10-14T10:41:06.637915Z","shell.execute_reply":"2024-10-14T10:41:08.507941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = dataset[:, 0:186]\ndataset = dataset.mean(axis=2)\ndataset = dataset[:, 1::2] - dataset[:, 0::2]\ndataset = dataset.cumsum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:19.328778Z","iopub.execute_input":"2024-10-14T10:41:19.329695Z","iopub.status.idle":"2024-10-14T10:41:19.335819Z","shell.execute_reply.started":"2024-10-14T10:41:19.329653Z","shell.execute_reply":"2024-10-14T10:41:19.334845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(dataset[0])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:20.116599Z","iopub.execute_input":"2024-10-14T10:41:20.117483Z","iopub.status.idle":"2024-10-14T10:41:20.272811Z","shell.execute_reply.started":"2024-10-14T10:41:20.117439Z","shell.execute_reply":"2024-10-14T10:41:20.271701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frame = pd.DataFrame(data = \n                     np.concatenate((dataset, mean_and_std), axis=1),\n                     columns=[f'col_{i}' for i in range(dataset.shape[1]+2)]\n                    )\ndel dataset, mean_and_std","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:23.623014Z","iopub.execute_input":"2024-10-14T10:41:23.623901Z","iopub.status.idle":"2024-10-14T10:41:23.630828Z","shell.execute_reply.started":"2024-10-14T10:41:23.623848Z","shell.execute_reply":"2024-10-14T10:41:23.629641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostRegressor","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:24.775658Z","iopub.execute_input":"2024-10-14T10:41:24.776672Z","iopub.status.idle":"2024-10-14T10:41:25.090536Z","shell.execute_reply.started":"2024-10-14T10:41:24.776614Z","shell.execute_reply":"2024-10-14T10:41:25.089534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = CatBoostRegressor(task_type='GPU', iterations=200, loss_function='MultiRMSE', verbose=50)\nmodel.load_model('/kaggle/input/my-adc-training/ADC_model.cbm')\npredictions = model.predict(frame)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:27.694515Z","iopub.execute_input":"2024-10-14T10:41:27.695244Z","iopub.status.idle":"2024-10-14T10:41:28.126431Z","shell.execute_reply.started":"2024-10-14T10:41:27.695201Z","shell.execute_reply":"2024-10-14T10:41:28.125437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sigma_pred = np.full_like(predictions, 0.00002439372)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:31.010094Z","iopub.execute_input":"2024-10-14T10:41:31.01049Z","iopub.status.idle":"2024-10-14T10:41:31.015399Z","shell.execute_reply.started":"2024-10-14T10:41:31.01045Z","shell.execute_reply":"2024-10-14T10:41:31.014413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/ariel-data-challenge-2024/sample_submission.csv')\nsample_submission.loc[:, sample_submission.columns != 'planet_id'] = np.concatenate((predictions, sigma_pred), axis=1)\nsample_submission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T10:41:43.585776Z","iopub.execute_input":"2024-10-14T10:41:43.586172Z","iopub.status.idle":"2024-10-14T10:41:43.625441Z","shell.execute_reply.started":"2024-10-14T10:41:43.586136Z","shell.execute_reply":"2024-10-14T10:41:43.624682Z"},"trusted":true},"execution_count":null,"outputs":[]}]}