{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":56537,"databundleVersionId":8015876,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I tried creating a NN with Keras, but I couldn't make predictions well. I would appreciate it if you could give me some findings.\n\n# Index\n01.Import  \n02.Check data  \n03.Seperate dataframe(train vs test)  \n04.Normalization  \n05.Make a model  \n06.Conduct  \n07.Forecast  \n08.Evaluation  \n09.Submit","metadata":{}},{"cell_type":"markdown","source":"### 01.Import","metadata":{}},{"cell_type":"code","source":"# Data analysis\nimport numpy as np\nimport pandas as pd\n# Avoid omissions\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None) \n# Set pandas display options\npd.set_option('display.float_format', '{:.30f}'.format)\n\n# Plot\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# ML\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf # for setting random state\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Activation, Input\nfrom keras.optimizers import SGD\nfrom keras.optimizers import Adam\nfrom keras.callbacks import EarlyStopping, LearningRateScheduler\n\n# Evaluation function\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import mean_squared_error\nfrom math import sqrt\nfrom sklearn.metrics import r2_score\n\n# Others\nimport os\nimport polars as pl\nimport time\n\n# Decide random state\nrandom_state=4","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:27:44.936718Z","iopub.execute_input":"2024-05-25T21:27:44.937062Z","iopub.status.idle":"2024-05-25T21:27:49.61141Z","shell.execute_reply.started":"2024-05-25T21:27:44.93703Z","shell.execute_reply":"2024-05-25T21:27:49.610619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#　Check each path\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:27:49.61628Z","iopub.execute_input":"2024-05-25T21:27:49.616562Z","iopub.status.idle":"2024-05-25T21:27:49.622252Z","shell.execute_reply.started":"2024-05-25T21:27:49.616536Z","shell.execute_reply":"2024-05-25T21:27:49.621346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import each file\n# temporarily set n_rows for saving time\nn_rows = 100000\nfolder = 'leap-atmospheric-physics-ai-climsim'\nt1 = time.time()\ndf_train = pl.read_csv('../input/'+folder+'/train.csv', n_rows=n_rows).to_pandas()\ndf_test = pl.read_csv('../input/'+folder+'/test.csv').to_pandas()\ndf_sub = pl.read_csv('../input/'+folder+'/sample_submission.csv').to_pandas()\n#df_train = pl.read_csv('../input/'+folder+'/train.csv').to_pandas()\n#df_test = pl.read_csv('../input/'+folder+'/test.csv').to_pandas()\n#df_sub = pl.read_csv('../input/'+folder+'/sample_submission.csv').to_pandas()\nt2 = time.time()\nprint('Elapsed time [s]: ', np.round(t2-t1,2))","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:27:49.623291Z","iopub.execute_input":"2024-05-25T21:27:49.623567Z","iopub.status.idle":"2024-05-25T21:28:13.130455Z","shell.execute_reply.started":"2024-05-25T21:27:49.623544Z","shell.execute_reply":"2024-05-25T21:28:13.129448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Issue1: Importing full data is difficult due to memory limitations.\n#### answer1: ","metadata":{}},{"cell_type":"markdown","source":"### 02.Check data","metadata":{}},{"cell_type":"code","source":"# Check each shape\ndf_train.shape, df_test.shape, df_sub.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:13.134045Z","iopub.execute_input":"2024-05-25T21:28:13.134356Z","iopub.status.idle":"2024-05-25T21:28:13.141771Z","shell.execute_reply.started":"2024-05-25T21:28:13.13433Z","shell.execute_reply":"2024-05-25T21:28:13.140691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check each columns quickly\ndf_train.columns","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:13.143222Z","iopub.execute_input":"2024-05-25T21:28:13.143626Z","iopub.status.idle":"2024-05-25T21:28:13.155511Z","shell.execute_reply.started":"2024-05-25T21:28:13.143592Z","shell.execute_reply":"2024-05-25T21:28:13.154472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check missing data\nmissing_data = df_train.isna().any().any()\nmissing_data\n","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:13.156685Z","iopub.execute_input":"2024-05-25T21:28:13.156993Z","iopub.status.idle":"2024-05-25T21:28:13.287741Z","shell.execute_reply.started":"2024-05-25T21:28:13.156968Z","shell.execute_reply":"2024-05-25T21:28:13.286746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check data overview\ndf_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:13.289102Z","iopub.execute_input":"2024-05-25T21:28:13.289502Z","iopub.status.idle":"2024-05-25T21:28:13.665372Z","shell.execute_reply.started":"2024-05-25T21:28:13.289463Z","shell.execute_reply":"2024-05-25T21:28:13.664382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check data overview\ndf_train.describe()","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:13.666461Z","iopub.execute_input":"2024-05-25T21:28:13.666746Z","iopub.status.idle":"2024-05-25T21:28:18.691659Z","shell.execute_reply.started":"2024-05-25T21:28:13.666722Z","shell.execute_reply":"2024-05-25T21:28:18.690752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### issue2: In pbuf_CH4_27, the average value may be weird, if all data is the same value.\t\n#### answer2: There are some different values. Check the min value in pbuf_CH4_27, for example.","metadata":{}},{"cell_type":"markdown","source":"### 03.Seperate dataframe(train vs test)","metadata":{}},{"cell_type":"code","source":"# Seperate data to train and validation\n(train, val) = train_test_split(df_train, test_size=0.2, shuffle=True, random_state=random_state)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:18.69294Z","iopub.execute_input":"2024-05-25T21:28:18.693552Z","iopub.status.idle":"2024-05-25T21:28:19.333187Z","shell.execute_reply.started":"2024-05-25T21:28:18.693515Z","shell.execute_reply":"2024-05-25T21:28:19.332274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seperate the iput data and target data \n# train data\nx_train = train.iloc[:, 1:557]\ny_train = train.iloc[:, 557:]\n# validation data\nx_val = val.iloc[:, 1:557]\ny_val = val.iloc[:, 557:]","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:19.334418Z","iopub.execute_input":"2024-05-25T21:28:19.334776Z","iopub.status.idle":"2024-05-25T21:28:19.581909Z","shell.execute_reply.started":"2024-05-25T21:28:19.334741Z","shell.execute_reply":"2024-05-25T21:28:19.581042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 04.Normalization","metadata":{}},{"cell_type":"code","source":"# Set normalization function\ndef safe_normalize(df):\n    normalized_df = df.copy()\n    for column in df.columns:\n        min_val = df[column].min()\n        max_val = df[column].max()\n        range_val = max_val - min_val\n        normalized_df[column] = (df[column] - min_val) / (range_val + 1e-20)  # Add small values to avoid generating nan\n    return normalized_df","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:19.5831Z","iopub.execute_input":"2024-05-25T21:28:19.583432Z","iopub.status.idle":"2024-05-25T21:28:19.589864Z","shell.execute_reply.started":"2024-05-25T21:28:19.583404Z","shell.execute_reply":"2024-05-25T21:28:19.588808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Conduct the normalization\nx_train_normalized = safe_normalize(x_train)\nx_val_normalized = safe_normalize(x_val)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:19.591004Z","iopub.execute_input":"2024-05-25T21:28:19.591297Z","iopub.status.idle":"2024-05-25T21:28:21.068515Z","shell.execute_reply.started":"2024-05-25T21:28:19.59127Z","shell.execute_reply":"2024-05-25T21:28:21.067275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check pbuf_CH4_27 if there are different values.\nx_train_normalized.describe()","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:21.072653Z","iopub.execute_input":"2024-05-25T21:28:21.072982Z","iopub.status.idle":"2024-05-25T21:28:24.423646Z","shell.execute_reply.started":"2024-05-25T21:28:21.072953Z","shell.execute_reply":"2024-05-25T21:28:24.422741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 05.Make a model","metadata":{}},{"cell_type":"code","source":"# Check input data shape and output data shape\nprint(f\"x_train shape: {x_train.shape}\")\nprint(f\"y_train shape: {y_train.shape}\")\nprint(f\"x_val shape: {x_val.shape}\")\nprint(f\"y_val shape: {y_val.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:24.424841Z","iopub.execute_input":"2024-05-25T21:28:24.425149Z","iopub.status.idle":"2024-05-25T21:28:24.430682Z","shell.execute_reply.started":"2024-05-25T21:28:24.425122Z","shell.execute_reply":"2024-05-25T21:28:24.42978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a model\nn_in = 556 # Number of input columns\nn_hidden = 800 # hidden node\nn_out = 368 # # Number of output columns \nepochs = 100 # Number of learning\n#batch_size = 1000 # Batch size -> val loss is around 12.XX\n#batch_size = 10 # -> val loss is around 7.XX\nbatch_size = 5 # -> val loss is around 5.XX","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:24.431734Z","iopub.execute_input":"2024-05-25T21:28:24.432017Z","iopub.status.idle":"2024-05-25T21:28:24.441184Z","shell.execute_reply.started":"2024-05-25T21:28:24.431993Z","shell.execute_reply":"2024-05-25T21:28:24.440261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set a model\ntf.random.set_seed(random_state) # Set random state\nmodel = Sequential()\nmodel.add(Input(shape=(n_in,)))\nmodel.add(Dense(n_hidden, activation= 'relu'))\nmodel.add(Dense(n_hidden, activation= 'relu'))\nmodel.add(Dense(n_hidden, activation= 'relu'))\nmodel.add(Dense(units=n_out))\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:24.442238Z","iopub.execute_input":"2024-05-25T21:28:24.442497Z","iopub.status.idle":"2024-05-25T21:28:25.072303Z","shell.execute_reply.started":"2024-05-25T21:28:24.442474Z","shell.execute_reply":"2024-05-25T21:28:25.071374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#　Set optimization algorithm\noptimizer = SGD(learning_rate=0.0001, momentum=0.9)\nmodel.compile(loss='mean_squared_error', optimizer=optimizer)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:25.07335Z","iopub.execute_input":"2024-05-25T21:28:25.073664Z","iopub.status.idle":"2024-05-25T21:28:25.088664Z","shell.execute_reply.started":"2024-05-25T21:28:25.073637Z","shell.execute_reply":"2024-05-25T21:28:25.087837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#　Set early stopping\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:25.089659Z","iopub.execute_input":"2024-05-25T21:28:25.089952Z","iopub.status.idle":"2024-05-25T21:28:25.09408Z","shell.execute_reply.started":"2024-05-25T21:28:25.089928Z","shell.execute_reply":"2024-05-25T21:28:25.09308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 06.Conduct","metadata":{}},{"cell_type":"code","source":"# Conducting\nt1 = time.time()\nhistory = model.fit(x_train_normalized, y_train, epochs=epochs, batch_size=batch_size, validation_data=(x_val_normalized, y_val), callbacks=[early_stopping])\nt2 = time.time()\nprint('Elapsed time [s]: ', np.round(t2-t1,2))","metadata":{"execution":{"iopub.status.busy":"2024-05-25T21:28:25.095298Z","iopub.execute_input":"2024-05-25T21:28:25.09558Z","iopub.status.idle":"2024-05-25T22:04:05.105944Z","shell.execute_reply.started":"2024-05-25T21:28:25.095555Z","shell.execute_reply":"2024-05-25T22:04:05.104856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 07.Forecast ","metadata":{}},{"cell_type":"code","source":"# 構築したモデルで予測\nval_predict = model.predict(x_val_normalized)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T22:04:05.107171Z","iopub.execute_input":"2024-05-25T22:04:05.107493Z","iopub.status.idle":"2024-05-25T22:04:06.88082Z","shell.execute_reply.started":"2024-05-25T22:04:05.107465Z","shell.execute_reply":"2024-05-25T22:04:06.879748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_predict","metadata":{"execution":{"iopub.status.busy":"2024-05-25T22:04:06.882598Z","iopub.execute_input":"2024-05-25T22:04:06.882924Z","iopub.status.idle":"2024-05-25T22:04:06.88973Z","shell.execute_reply.started":"2024-05-25T22:04:06.882897Z","shell.execute_reply":"2024-05-25T22:04:06.888801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 08.Evaluation","metadata":{}},{"cell_type":"code","source":"#Accuracy - coefficient of determination (close 1 is better)\nR2 = r2_score(y_val, val_predict)\nprint('R2：', R2)","metadata":{"execution":{"iopub.status.busy":"2024-05-25T22:04:06.891071Z","iopub.execute_input":"2024-05-25T22:04:06.891412Z","iopub.status.idle":"2024-05-25T22:04:07.00326Z","shell.execute_reply.started":"2024-05-25T22:04:06.891381Z","shell.execute_reply":"2024-05-25T22:04:07.002256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot loss function\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs = len(loss)\nplt.plot(range(epochs), loss, marker='.', label='loss')\nplt.plot(range(epochs), val_loss, marker='.', label='val_loss')\nplt.legend(loc='best')\nplt.grid()\nplt.xlabel('epoch')\nplt.ylabel('loss')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-25T22:04:07.004576Z","iopub.execute_input":"2024-05-25T22:04:07.005282Z","iopub.status.idle":"2024-05-25T22:04:07.282973Z","shell.execute_reply.started":"2024-05-25T22:04:07.005242Z","shell.execute_reply":"2024-05-25T22:04:07.282061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Observed-Predicted plot\nplt.figure()\nplt.scatter(y_val, val_predict, c='blue', alpha=0.8)\nplt.ylim(plt.ylim())\nplt.grid()\nplt.xlabel('observed_data')\nplt.ylabel('predicted_data')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-25T22:04:07.284021Z","iopub.execute_input":"2024-05-25T22:04:07.28429Z","iopub.status.idle":"2024-05-25T22:04:23.870902Z","shell.execute_reply.started":"2024-05-25T22:04:07.284267Z","shell.execute_reply":"2024-05-25T22:04:23.869964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 09.Submit","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}