{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"dockerImageVersionId":30700,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-12T11:19:13.826387Z","iopub.execute_input":"2024-05-12T11:19:13.827191Z","iopub.status.idle":"2024-05-12T11:19:14.170498Z","shell.execute_reply.started":"2024-05-12T11:19:13.827156Z","shell.execute_reply":"2024-05-12T11:19:14.169493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.callbacks import LearningRateScheduler, ReduceLROnPlateau\nfrom tensorflow.keras import backend as K\nimport tensorflow as tf\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nimport os\nimport polars as pl\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import KFold\n\n\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:19:16.957661Z","iopub.execute_input":"2024-05-12T11:19:16.958097Z","iopub.status.idle":"2024-05-12T11:19:29.114263Z","shell.execute_reply.started":"2024-05-12T11:19:16.958071Z","shell.execute_reply":"2024-05-12T11:19:29.11328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:19:29.115743Z","iopub.execute_input":"2024-05-12T11:19:29.116246Z","iopub.status.idle":"2024-05-12T11:19:29.312564Z","shell.execute_reply.started":"2024-05-12T11:19:29.11621Z","shell.execute_reply":"2024-05-12T11:19:29.311066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_chunk = pd.read_csv('/kaggle/input/leap-atmospheric-physics-ai-climsim/train.csv', chunksize=500000)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:19:43.783362Z","iopub.execute_input":"2024-05-12T11:19:43.783723Z","iopub.status.idle":"2024-05-12T11:19:43.793806Z","shell.execute_reply.started":"2024-05-12T11:19:43.783695Z","shell.execute_reply":"2024-05-12T11:19:43.792795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_chunk","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:19:46.278842Z","iopub.execute_input":"2024-05-12T11:19:46.279427Z","iopub.status.idle":"2024-05-12T11:19:46.285075Z","shell.execute_reply.started":"2024-05-12T11:19:46.279398Z","shell.execute_reply":"2024-05-12T11:19:46.284181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize chunk_1 to None\nchunk_1 = None\n\n# Loop over chunks and break after the first\nfor chunk in train_df_chunk:\n    chunk_1 = chunk  # Assign the first chunk to chunk_1\n    break  # Exit the loop after assigning the first chunk","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:19:48.893472Z","iopub.execute_input":"2024-05-12T11:19:48.894128Z","iopub.status.idle":"2024-05-12T11:22:55.515796Z","shell.execute_reply.started":"2024-05-12T11:19:48.894096Z","shell.execute_reply":"2024-05-12T11:22:55.514973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunk_1.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:23:43.617986Z","iopub.execute_input":"2024-05-12T11:23:43.618759Z","iopub.status.idle":"2024-05-12T11:23:43.637861Z","shell.execute_reply.started":"2024-05-12T11:23:43.618726Z","shell.execute_reply":"2024-05-12T11:23:43.63697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #storing the chunks after iterating from chunk object\n# chunk_data=[chunk for chunk in train_df_chunk]\n\n# #concatnating dataframes to make it a complete dataset\n# train=pd.concat(chunk_data)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:23:44.142637Z","iopub.execute_input":"2024-05-12T11:23:44.142967Z","iopub.status.idle":"2024-05-12T11:23:44.147063Z","shell.execute_reply.started":"2024-05-12T11:23:44.142935Z","shell.execute_reply":"2024-05-12T11:23:44.146124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunk_1.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:23:45.072044Z","iopub.execute_input":"2024-05-12T11:23:45.072403Z","iopub.status.idle":"2024-05-12T11:23:45.10506Z","shell.execute_reply.started":"2024-05-12T11:23:45.072376Z","shell.execute_reply":"2024-05-12T11:23:45.10425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunk_1.describe","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:23:46.106082Z","iopub.execute_input":"2024-05-12T11:23:46.106945Z","iopub.status.idle":"2024-05-12T11:23:46.172472Z","shell.execute_reply.started":"2024-05-12T11:23:46.10691Z","shell.execute_reply":"2024-05-12T11:23:46.171569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEAT_COLS = chunk_1.columns[1:557]\nTARGET_COLS = chunk_1.columns[557:]\nprint(f'Number of features: {len(FEAT_COLS)}')\nprint(f'Number of targets: {len(TARGET_COLS)}')","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:23:47.45267Z","iopub.execute_input":"2024-05-12T11:23:47.453024Z","iopub.status.idle":"2024-05-12T11:23:47.458609Z","shell.execute_reply.started":"2024-05-12T11:23:47.452997Z","shell.execute_reply":"2024-05-12T11:23:47.457753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom R² metric\ndef r2_metric(y_true, y_pred):\n    ss_res = K.sum(K.square(y_true - y_pred))\n    ss_tot = K.sum(K.square(y_true - K.mean(y_true)))\n    r2 = 1 - ss_res / (ss_tot + K.epsilon())\n    return r2\n\n# Create a simple neural network for multi-output regression\ndef build_regression_model(input_shape, output_shape):\n    model = keras.Sequential([\n        layers.Dense(2048, activation='relu', input_shape=input_shape),\n        layers.BatchNormalization(),\n        layers.Dropout(0.2),\n        layers.Dense(512, activation='relu'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.2),\n        layers.Dense(256, activation='relu'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.2),\n        layers.Dense(128, activation='relu'),\n        layers.BatchNormalization(),\n        layers.Dropout(0.2),\n        layers.Dense(output_shape[0]),  # Output layer for multi-output regression\n    ])\n    model.compile(\n        optimizer='adam',\n        loss='mean_squared_error',  # Loss function for regression\n        metrics=[r2_metric]  # Using R² as an evaluation metric\n    )\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:23:48.706641Z","iopub.execute_input":"2024-05-12T11:23:48.706978Z","iopub.status.idle":"2024-05-12T11:23:48.715705Z","shell.execute_reply.started":"2024-05-12T11:23:48.706953Z","shell.execute_reply":"2024-05-12T11:23:48.714762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_epochs = 50  # Number of epochs to train\nbatch_size = 64  # Batch size for training\ninput_shape = (556,)  # Number of input features\noutput_shape = (368,)  # Number of target columns\n\n# Build the regression model\nmodel = build_regression_model(input_shape, output_shape)\n\n# To collect training history\ntraining_history = []\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:24:09.164944Z","iopub.execute_input":"2024-05-12T11:24:09.165641Z","iopub.status.idle":"2024-05-12T11:24:09.269551Z","shell.execute_reply.started":"2024-05-12T11:24:09.165603Z","shell.execute_reply":"2024-05-12T11:24:09.26874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    chunk_1.iloc[:, 1:557],  # Features\n    chunk_1.iloc[:, 557:],    # Labels\n    epochs=num_epochs,\n    batch_size=batch_size,\n    verbose=1\n)\n\n# Convert the training history to a DataFrame\ntraining_history = {\n    \"epoch\": list(range(1, num_epochs + 1)),\n    \"loss\": history.history[\"loss\"],\n    \"r2\": history.history[\"r2_metric\"]\n}\ntraining_df = pd.DataFrame(training_history)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:24:10.663964Z","iopub.execute_input":"2024-05-12T11:24:10.664846Z","iopub.status.idle":"2024-05-12T11:41:22.329222Z","shell.execute_reply.started":"2024-05-12T11:24:10.664813Z","shell.execute_reply":"2024-05-12T11:41:22.328392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for epoch in range(num_epochs):\n#     print(f\"Training epoch {epoch + 1}/{num_epochs}\")\n\n#     # Loop over each chunk and train the model\n#     for i, chunk in enumerate(train_df_chunk):\n#         # Extract features and labels from the chunk\n#         features = chunk.iloc[:,1:557]  # First 556 columns as features less sample_id\n#         labels = chunk.iloc[:, 557:]  # Remaining 368 columns as targets\n        \n#         # Train the model on this chunk and collect history\n#         history = model.fit(\n#             features,\n#             labels,\n#             epochs=5,  # Train one epoch for each chunk\n#             batch_size=batch_size,\n#             verbose=1  # Change to 0 to suppress output\n#         )\n        \n#         # Store the history\n#         epoch_history = {\n#             \"epoch\": epoch + 1,\n#             \"chunk\": i + 1,\n#             \"loss\": history.history[\"loss\"][0],\n#             \"r2\": history.history[\"r2_metric\"][0]\n#         }\n        \n#         training_history.append(epoch_history)\n\n# # Convert the training history to a DataFrame\n# training_df = pd.DataFrame(training_history)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:41:41.391781Z","iopub.execute_input":"2024-05-12T11:41:41.392142Z","iopub.status.idle":"2024-05-12T11:41:41.397379Z","shell.execute_reply.started":"2024-05-12T11:41:41.392113Z","shell.execute_reply":"2024-05-12T11:41:41.3964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(10, 4))\n\n# Plot 'column1' in the first subplot\nax1.plot(training_df['r2'])\nax1.set_title('R2')\n\n# Plot 'column2' in the second subplot\nax2.plot(training_df['loss'])\nax2.set_title('loss')\n\n# Display the plots\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:41:46.194342Z","iopub.execute_input":"2024-05-12T11:41:46.195189Z","iopub.status.idle":"2024-05-12T11:41:46.765937Z","shell.execute_reply.started":"2024-05-12T11:41:46.195146Z","shell.execute_reply":"2024-05-12T11:41:46.76502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test dataset and sample prediction file paths\ntest_file_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/test.csv'  # Path to the test dataset\nsample_prediction_file_path = '/kaggle/input/leap-atmospheric-physics-ai-climsim/sample_submission.csv'  # Path to the sample prediction file\n\n# Define the chunksize for reading the test and sample datasets\nchunksize = 50000  # Chunk size for processing\n\n# Open the file in write mode to store predictions\nprediction_file_path = 'final_submission.csv'\n\n\n# Read both test and sample prediction data in chunks\ntest_df_chunk = pd.read_csv(test_file_path, chunksize=chunksize)\nsample_prediction_chunk = pd.read_csv(sample_prediction_file_path, chunksize=chunksize)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:41:52.512495Z","iopub.execute_input":"2024-05-12T11:41:52.512852Z","iopub.status.idle":"2024-05-12T11:41:52.531048Z","shell.execute_reply.started":"2024-05-12T11:41:52.512822Z","shell.execute_reply":"2024-05-12T11:41:52.530139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize test_chunk_1 to None\ntest_chunk_1 = None\n\n# Loop over chunks and break after the first\nfor chunk in test_df_chunk:\n    test_chunk_1 = chunk  # Assign the first chunk to test_chunk_1\n    break  # Exit the loop after assigning the first chunk\n    \ntest_chunk_1.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:41:55.736355Z","iopub.execute_input":"2024-05-12T11:41:55.736707Z","iopub.status.idle":"2024-05-12T11:42:07.072697Z","shell.execute_reply.started":"2024-05-12T11:41:55.736679Z","shell.execute_reply":"2024-05-12T11:42:07.071719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize sample_chunk_1 to None\nsample_chunk_1 = None\n\n# Loop over chunks and break after the first\nfor chunk in sample_prediction_chunk:\n    sample_chunk_1 = chunk  # Assign the first chunk to sample_chunk_1\n    break  # Exit the loop after assigning the first chunk\n    \nsample_chunk_1.info()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:42:07.074679Z","iopub.execute_input":"2024-05-12T11:42:07.075039Z","iopub.status.idle":"2024-05-12T11:42:12.965461Z","shell.execute_reply.started":"2024-05-12T11:42:07.075006Z","shell.execute_reply":"2024-05-12T11:42:12.964458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read both test and sample prediction data in chunks\ntest_df = pd.read_csv(test_file_path)\nsample_prediction = pd.read_csv(sample_prediction_file_path)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:42:12.966722Z","iopub.execute_input":"2024-05-12T11:42:12.967004Z","iopub.status.idle":"2024-05-12T11:45:44.726418Z","shell.execute_reply.started":"2024-05-12T11:42:12.966978Z","shell.execute_reply":"2024-05-12T11:45:44.725558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:45:44.728492Z","iopub.execute_input":"2024-05-12T11:45:44.728796Z","iopub.status.idle":"2024-05-12T11:45:44.734993Z","shell.execute_reply.started":"2024-05-12T11:45:44.72877Z","shell.execute_reply":"2024-05-12T11:45:44.733977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_prediction.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:45:44.736142Z","iopub.execute_input":"2024-05-12T11:45:44.736484Z","iopub.status.idle":"2024-05-12T11:45:44.751757Z","shell.execute_reply.started":"2024-05-12T11:45:44.736451Z","shell.execute_reply":"2024-05-12T11:45:44.750909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = model.predict(test_df.iloc[:, 1:557])\nprediction.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:45:44.752904Z","iopub.execute_input":"2024-05-12T11:45:44.753518Z","iopub.status.idle":"2024-05-12T11:46:32.145004Z","shell.execute_reply.started":"2024-05-12T11:45:44.753484Z","shell.execute_reply":"2024-05-12T11:46:32.144243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_prediction.drop(['sample_id'], axis=1).to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:46:32.146182Z","iopub.execute_input":"2024-05-12T11:46:32.146478Z","iopub.status.idle":"2024-05-12T11:46:33.184612Z","shell.execute_reply.started":"2024-05-12T11:46:32.146453Z","shell.execute_reply":"2024-05-12T11:46:33.183499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multiplied_predictions = np.multiply(prediction,sample_prediction.drop(['sample_id'], axis=1).to_numpy())\nmultiplied_predictions","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:46:33.187591Z","iopub.execute_input":"2024-05-12T11:46:33.188204Z","iopub.status.idle":"2024-05-12T11:46:35.583875Z","shell.execute_reply.started":"2024-05-12T11:46:33.188177Z","shell.execute_reply":"2024-05-12T11:46:35.582982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(sample_prediction.iloc[:, 0] == test_df.iloc[:, 0]).sum()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:46:35.585426Z","iopub.execute_input":"2024-05-12T11:46:35.585724Z","iopub.status.idle":"2024-05-12T11:46:35.741788Z","shell.execute_reply.started":"2024-05-12T11:46:35.5857Z","shell.execute_reply":"2024-05-12T11:46:35.740688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_df = pd.DataFrame(multiplied_predictions)\ncolumn_names = sample_prediction.columns[1:]\npredictions_df.columns = column_names\npredictions_df","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:46:35.743143Z","iopub.execute_input":"2024-05-12T11:46:35.743515Z","iopub.status.idle":"2024-05-12T11:46:36.038816Z","shell.execute_reply.started":"2024-05-12T11:46:35.743482Z","shell.execute_reply":"2024-05-12T11:46:36.03788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add the `target_id` column from the test data\npredictions_df['sample_id'] = sample_prediction.iloc[:, 0]\npredictions_df","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:46:36.040064Z","iopub.execute_input":"2024-05-12T11:46:36.040369Z","iopub.status.idle":"2024-05-12T11:46:36.393031Z","shell.execute_reply.started":"2024-05-12T11:46:36.040344Z","shell.execute_reply":"2024-05-12T11:46:36.392025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Write the predictions to the CSV file, append with index=False\npredictions_df.to_csv(prediction_file_path, index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:46:36.39447Z","iopub.execute_input":"2024-05-12T11:46:36.394949Z","iopub.status.idle":"2024-05-12T11:53:07.590602Z","shell.execute_reply.started":"2024-05-12T11:46:36.394913Z","shell.execute_reply":"2024-05-12T11:53:07.589663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}