{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Exploratory Data Analysis**","metadata":{}},{"cell_type":"markdown","source":"**I will explore the data to gain insights about the data.**","metadata":{}},{"cell_type":"code","source":"# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import libraries\nimport pandas as pd #data processing\nimport matplotlib.pyplot as plt # data visualization\nimport seaborn as sns #visualizations\nimport os\nfrom path import Path\nimport numpy as np #linear algebra","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:16:23.038597Z","iopub.execute_input":"2023-07-06T07:16:23.039482Z","iopub.status.idle":"2023-07-06T07:16:24.318836Z","shell.execute_reply.started":"2023-07-06T07:16:23.039437Z","shell.execute_reply":"2023-07-06T07:16:24.317679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#define the root directory\nroot_dir = Path(\"/kaggle/input/google-research-identify-contrails-reduce-global-warming\")\n#define the path for train,validatin and test folders\ntrain_band = os.listdir(root_dir + '/train/')\nval_band = os.listdir(root_dir + '/validation/')\ntest_band = os.listdir(root_dir + '/test/')","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:16:28.082104Z","iopub.execute_input":"2023-07-06T07:16:28.082459Z","iopub.status.idle":"2023-07-06T07:16:28.347523Z","shell.execute_reply.started":"2023-07-06T07:16:28.082434Z","shell.execute_reply":"2023-07-06T07:16:28.346411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check number of samples in train,validation and test folders\nprint(f\"samples in train folder :{len(train_band)}\")\nprint(f\"samples in validation frolder :{len(val_band)}\")\nprint(f\"samples in test folder :{len(test_band)}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:16:39.990486Z","iopub.execute_input":"2023-07-06T07:16:39.990907Z","iopub.status.idle":"2023-07-06T07:16:39.997362Z","shell.execute_reply.started":"2023-07-06T07:16:39.990854Z","shell.execute_reply":"2023-07-06T07:16:39.996271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load the meta_data_set using pandas\ntrain_df = pd.read_json('/kaggle/input/google-research-identify-contrails-reduce-global-warming/train_metadata.json')\nval_df = pd.read_json('/kaggle/input/google-research-identify-contrails-reduce-global-warming/validation_metadata.json')","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:16:42.065846Z","iopub.execute_input":"2023-07-06T07:16:42.066257Z","iopub.status.idle":"2023-07-06T07:16:42.484291Z","shell.execute_reply.started":"2023-07-06T07:16:42.066228Z","shell.execute_reply":"2023-07-06T07:16:42.483315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#view diemensions of the dataset\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:16:46.804353Z","iopub.execute_input":"2023-07-06T07:16:46.804908Z","iopub.status.idle":"2023-07-06T07:16:46.814731Z","shell.execute_reply.started":"2023-07-06T07:16:46.804845Z","shell.execute_reply":"2023-07-06T07:16:46.813524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**It has 20529 instances and 7 attributes in training dataset**","metadata":{}},{"cell_type":"code","source":"val_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:16:49.164537Z","iopub.execute_input":"2023-07-06T07:16:49.165674Z","iopub.status.idle":"2023-07-06T07:16:49.173165Z","shell.execute_reply.started":"2023-07-06T07:16:49.165628Z","shell.execute_reply":"2023-07-06T07:16:49.171863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**It has 1856 instances and 7 attributes in validation dataset**","metadata":{}},{"cell_type":"code","source":"#preview the dataset of training\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:16:51.187911Z","iopub.execute_input":"2023-07-06T07:16:51.18875Z","iopub.status.idle":"2023-07-06T07:16:51.218078Z","shell.execute_reply.started":"2023-07-06T07:16:51.188707Z","shell.execute_reply":"2023-07-06T07:16:51.216958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preview the dataset of validating\nval_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T13:57:42.081185Z","iopub.execute_input":"2023-05-27T13:57:42.081616Z","iopub.status.idle":"2023-05-27T13:57:42.096722Z","shell.execute_reply.started":"2023-05-27T13:57:42.081582Z","shell.execute_reply":"2023-05-27T13:57:42.095904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.size","metadata":{"execution":{"iopub.status.busy":"2023-05-27T13:58:20.561758Z","iopub.execute_input":"2023-05-27T13:58:20.562203Z","iopub.status.idle":"2023-05-27T13:58:20.569697Z","shell.execute_reply.started":"2023-05-27T13:58:20.562166Z","shell.execute_reply":"2023-05-27T13:58:20.568432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#view summary of train dataset\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T13:58:28.379516Z","iopub.execute_input":"2023-05-27T13:58:28.379972Z","iopub.status.idle":"2023-05-27T13:58:28.416274Z","shell.execute_reply.started":"2023-05-27T13:58:28.379933Z","shell.execute_reply":"2023-05-27T13:58:28.41478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Findings**\n* The datset has 1 character variable and 6 numerical variables\n* It has no missing values or null values","metadata":{}},{"cell_type":"code","source":"#view summary of validation dataset\nval_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T13:58:46.312541Z","iopub.execute_input":"2023-05-27T13:58:46.312962Z","iopub.status.idle":"2023-05-27T13:58:46.328646Z","shell.execute_reply.started":"2023-05-27T13:58:46.312925Z","shell.execute_reply":"2023-05-27T13:58:46.327745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the data type of a particular column\ntrain_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-05-27T13:52:26.420775Z","iopub.execute_input":"2023-05-27T13:52:26.421157Z","iopub.status.idle":"2023-05-27T13:52:26.429503Z","shell.execute_reply.started":"2023-05-27T13:52:26.421126Z","shell.execute_reply":"2023-05-27T13:52:26.428432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the data type of a particular column in validation dataset\nval_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:18:56.900364Z","iopub.execute_input":"2023-05-31T04:18:56.900749Z","iopub.status.idle":"2023-05-31T04:18:56.908464Z","shell.execute_reply.started":"2023-05-31T04:18:56.900719Z","shell.execute_reply":"2023-05-31T04:18:56.907403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# View statistical properties of dataset","metadata":{}},{"cell_type":"code","source":"#presents statistical properties in vertical form, It shows properties only for numerical variables. It excludes character variables.\ntrain_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T05:51:37.955881Z","iopub.execute_input":"2023-05-24T05:51:37.956242Z","iopub.status.idle":"2023-05-24T05:51:37.992492Z","shell.execute_reply.started":"2023-05-24T05:51:37.956216Z","shell.execute_reply":"2023-05-24T05:51:37.991202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#presents statistical properties in vertical form, It shows properties only for numerical variables. It excludes character variables.\nval_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:18:48.481495Z","iopub.execute_input":"2023-05-31T04:18:48.482082Z","iopub.status.idle":"2023-05-31T04:18:48.535449Z","shell.execute_reply.started":"2023-05-31T04:18:48.482033Z","shell.execute_reply":"2023-05-31T04:18:48.534354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#view the statistical properties in horizontal form\ntrain_df.describe().T","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:19:08.038701Z","iopub.execute_input":"2023-05-31T04:19:08.039139Z","iopub.status.idle":"2023-05-31T04:19:08.07395Z","shell.execute_reply.started":"2023-05-31T04:19:08.039102Z","shell.execute_reply":"2023-05-31T04:19:08.07314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#view the statistical properties in horizontal form\nval_df.describe().T","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#view the statistical properties of character variables\ntrain_df.describe(include=['object'])","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:19:18.937011Z","iopub.execute_input":"2023-05-31T04:19:18.937378Z","iopub.status.idle":"2023-05-31T04:19:18.964857Z","shell.execute_reply.started":"2023-05-31T04:19:18.937347Z","shell.execute_reply":"2023-05-31T04:19:18.963738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df.describe(include=['object'])","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:19:14.066032Z","iopub.execute_input":"2023-05-31T04:19:14.066441Z","iopub.status.idle":"2023-05-31T04:19:14.086266Z","shell.execute_reply.started":"2023-05-31T04:19:14.066405Z","shell.execute_reply":"2023-05-31T04:19:14.085273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#view the statistical properties of all the variables\ntrain_df.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2023-05-24T05:54:33.69377Z","iopub.execute_input":"2023-05-24T05:54:33.694173Z","iopub.status.idle":"2023-05-24T05:54:33.750971Z","shell.execute_reply.started":"2023-05-24T05:54:33.694143Z","shell.execute_reply":"2023-05-24T05:54:33.749944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#view the statistical properties of all the variables\nval_df.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:19:29.073607Z","iopub.execute_input":"2023-05-31T04:19:29.074031Z","iopub.status.idle":"2023-05-31T04:19:29.119313Z","shell.execute_reply.started":"2023-05-31T04:19:29.073994Z","shell.execute_reply":"2023-05-31T04:19:29.118217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#total number of missing values in each column in the dataframe\ntrain_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T05:58:09.029879Z","iopub.execute_input":"2023-05-24T05:58:09.030299Z","iopub.status.idle":"2023-05-24T05:58:09.048115Z","shell.execute_reply.started":"2023-05-24T05:58:09.030266Z","shell.execute_reply":"2023-05-24T05:58:09.046765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#total number of missing values in each column in the dataframe\nval_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:19:37.857497Z","iopub.execute_input":"2023-05-31T04:19:37.857902Z","iopub.status.idle":"2023-05-31T04:19:37.868047Z","shell.execute_reply.started":"2023-05-31T04:19:37.857869Z","shell.execute_reply":"2023-05-31T04:19:37.866998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Interpretation**\n\nWe can see that there are no missing values in the train and validation dataset.","metadata":{}},{"cell_type":"code","source":"#assert that there are no missing values in the dataframe\nassert pd.notnull(train_df).all().all()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:20:02.419021Z","iopub.execute_input":"2023-05-31T04:20:02.419381Z","iopub.status.idle":"2023-05-31T04:20:02.433641Z","shell.execute_reply.started":"2023-05-31T04:20:02.419354Z","shell.execute_reply":"2023-05-31T04:20:02.432415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#assert that there are no missing values in the dataframe\nassert pd.notnull(val_df).all().all()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:20:05.787156Z","iopub.execute_input":"2023-05-31T04:20:05.787553Z","iopub.status.idle":"2023-05-31T04:20:05.795431Z","shell.execute_reply.started":"2023-05-31T04:20:05.787504Z","shell.execute_reply":"2023-05-31T04:20:05.794494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* All the data values are grater than or equal to zero excluding character values","metadata":{}},{"cell_type":"code","source":"#visualizing heatmap for checking null values\nplt.subplots(figsize=(10,10))\nsns.heatmap(train_df.isnull(), yticklabels=False, cbar=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T13:53:34.444171Z","iopub.execute_input":"2023-05-27T13:53:34.445351Z","iopub.status.idle":"2023-05-27T13:53:34.904538Z","shell.execute_reply.started":"2023-05-27T13:53:34.445297Z","shell.execute_reply":"2023-05-27T13:53:34.902899Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**In the graph, there are no missing values displayed**","metadata":{}},{"cell_type":"code","source":"#check duplicate values in dataframe\ntrain_df.duplicated()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T06:08:07.880636Z","iopub.execute_input":"2023-05-24T06:08:07.881015Z","iopub.status.idle":"2023-05-24T06:08:07.912561Z","shell.execute_reply.started":"2023-05-24T06:08:07.880987Z","shell.execute_reply":"2023-05-24T06:08:07.911681Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check duplicate values in dataframe\nval_df.duplicated()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preview Categorical Variables","metadata":{}},{"cell_type":"code","source":"categorical1 = [var for var in train_df.columns if train_df[var].dtype == 'object']\ntrain_df[categorical1].head()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:53:28.495552Z","iopub.execute_input":"2023-05-31T04:53:28.496207Z","iopub.status.idle":"2023-05-31T04:53:28.524935Z","shell.execute_reply.started":"2023-05-31T04:53:28.496171Z","shell.execute_reply":"2023-05-31T04:53:28.523581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical2 = [var for var in val_df.columns if val_df[var].dtype == 'object']\nval_df[categorical2].head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#frequency distribution of categorical variable\nfor var in categorical1:\n    \n    print(train_df[var].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:53:33.477448Z","iopub.execute_input":"2023-05-31T04:53:33.477856Z","iopub.status.idle":"2023-05-31T04:53:33.487865Z","shell.execute_reply.started":"2023-05-31T04:53:33.477822Z","shell.execute_reply":"2023-05-31T04:53:33.486704Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#frequency distribution of categorical variable\nfor var in categorical2:\n    \n    print(val_df[var].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:22:06.630889Z","iopub.execute_input":"2023-05-31T04:22:06.631439Z","iopub.status.idle":"2023-05-31T04:22:06.640189Z","shell.execute_reply.started":"2023-05-31T04:22:06.631397Z","shell.execute_reply":"2023-05-31T04:22:06.638863Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore the variables in train_metadata","metadata":{}},{"cell_type":"code","source":"#frequency distribution of numerical variables\ntrain_df['row_min'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T14:00:39.622254Z","iopub.execute_input":"2023-05-27T14:00:39.622645Z","iopub.status.idle":"2023-05-27T14:00:39.636402Z","shell.execute_reply.started":"2023-05-27T14:00:39.622615Z","shell.execute_reply":"2023-05-27T14:00:39.635201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['row_size'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T14:01:09.362498Z","iopub.execute_input":"2023-05-27T14:01:09.362905Z","iopub.status.idle":"2023-05-27T14:01:09.374611Z","shell.execute_reply.started":"2023-05-27T14:01:09.362874Z","shell.execute_reply":"2023-05-27T14:01:09.373333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['col_min'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T14:01:22.939406Z","iopub.execute_input":"2023-05-27T14:01:22.939834Z","iopub.status.idle":"2023-05-27T14:01:22.951671Z","shell.execute_reply.started":"2023-05-27T14:01:22.939797Z","shell.execute_reply":"2023-05-27T14:01:22.950491Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['col_size'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T14:01:56.374323Z","iopub.execute_input":"2023-05-27T14:01:56.37474Z","iopub.status.idle":"2023-05-27T14:01:56.385544Z","shell.execute_reply.started":"2023-05-27T14:01:56.374709Z","shell.execute_reply":"2023-05-27T14:01:56.384345Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore the variables in validation_metadata","metadata":{}},{"cell_type":"code","source":"#assign all numerical attributes into a variable\nnumeric_var = [var for var in val_df.columns if val_df[var].dtype == 'float64']\nval_df[numeric_var].head()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:58:36.686097Z","iopub.execute_input":"2023-05-31T04:58:36.686461Z","iopub.status.idle":"2023-05-31T04:58:36.701149Z","shell.execute_reply.started":"2023-05-31T04:58:36.686434Z","shell.execute_reply":"2023-05-31T04:58:36.700072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#frequency distribution of numerical attributes in validation_dataset\nfor var in numeric_var:\n    \n    print(val_df[var].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-05-31T04:58:45.206889Z","iopub.execute_input":"2023-05-31T04:58:45.207586Z","iopub.status.idle":"2023-05-31T04:58:45.219629Z","shell.execute_reply.started":"2023-05-31T04:58:45.207551Z","shell.execute_reply":"2023-05-31T04:58:45.218703Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop the record_id and draw a countplot\ntrain_df = train_df.drop(columns= 'record_id')\nsns.countplot(data=train_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T06:09:27.797669Z","iopub.execute_input":"2023-05-31T06:09:27.798099Z","iopub.status.idle":"2023-05-31T06:09:27.986294Z","shell.execute_reply.started":"2023-05-31T06:09:27.798068Z","shell.execute_reply":"2023-05-31T06:09:27.985213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df = val_df.drop(columns= 'record_id')\nsns.countplot(data=val_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:19:55.10653Z","iopub.execute_input":"2023-05-31T05:19:55.106953Z","iopub.status.idle":"2023-05-31T05:19:55.29458Z","shell.execute_reply.started":"2023-05-31T05:19:55.106917Z","shell.execute_reply":"2023-05-31T05:19:55.293801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x= val_df['row_min'],data=val_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T12:35:43.108688Z","iopub.execute_input":"2023-05-31T12:35:43.109134Z","iopub.status.idle":"2023-05-31T12:35:45.861966Z","shell.execute_reply.started":"2023-05-31T12:35:43.1091Z","shell.execute_reply":"2023-05-31T12:35:45.860538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot the row_min frequency distribution\n#Length of this column is very high, can't clearly see plots\nf, ax = plt.subplots(figsize=(100, 60))\nax = train_df['row_min'].value_counts().plot(kind=\"bar\", color=\"green\")\nax.set_title(\"Frequency distribution of row_min variable\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T14:02:33.743761Z","iopub.execute_input":"2023-05-27T14:02:33.744219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Relationship between attributes in train_metadata","metadata":{}},{"cell_type":"code","source":"#drop record_id and timestamp columns and create new dataframe\ntrain_df = train_df.drop(columns = ['timestamp'])\n#draw the pairplots to check distribution and relationship between each attributes\nsns.pairplot(train_df)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T06:09:59.847447Z","iopub.execute_input":"2023-05-31T06:09:59.847919Z","iopub.status.idle":"2023-05-31T06:10:04.141102Z","shell.execute_reply.started":"2023-05-31T06:09:59.847876Z","shell.execute_reply":"2023-05-31T06:10:04.140086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Relationship between numeric attributes in validation_metadata","metadata":{}},{"cell_type":"code","source":"#drop record_id and timestamp columns and create new dataframe\nval_df = val_df.drop(columns = ['timestamp'])\n#draw the pairplots to check distribution and relationship between each attributes\nsns.pairplot(val_df)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlation","metadata":{}},{"cell_type":"code","source":"# check relationships between variables\ncorr_matrix_train = train_df.corr()\nprint(corr_matrix_train)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T06:00:42.322832Z","iopub.execute_input":"2023-05-31T06:00:42.323216Z","iopub.status.idle":"2023-05-31T06:00:42.332757Z","shell.execute_reply.started":"2023-05-31T06:00:42.323188Z","shell.execute_reply":"2023-05-31T06:00:42.331688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualize the correlations\nsns.heatmap(corr_matrix_train, annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T06:01:24.983671Z","iopub.execute_input":"2023-05-31T06:01:24.984085Z","iopub.status.idle":"2023-05-31T06:01:25.243818Z","shell.execute_reply.started":"2023-05-31T06:01:24.984053Z","shell.execute_reply":"2023-05-31T06:01:25.242243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check relationships between variables\ncorr_matrix_val = val_df.corr()\nprint(corr_matrix_val)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T06:00:46.306934Z","iopub.execute_input":"2023-05-31T06:00:46.307586Z","iopub.status.idle":"2023-05-31T06:00:46.318995Z","shell.execute_reply.started":"2023-05-31T06:00:46.307541Z","shell.execute_reply":"2023-05-31T06:00:46.318171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualize the correlations\nsns.heatmap(corr_matrix_val, annot = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T06:01:39.052234Z","iopub.execute_input":"2023-05-31T06:01:39.052612Z","iopub.status.idle":"2023-05-31T06:01:39.30564Z","shell.execute_reply.started":"2023-05-31T06:01:39.052583Z","shell.execute_reply":"2023-05-31T06:01:39.304491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore the train list folder to review a sample","metadata":{}},{"cell_type":"code","source":"#get images from the folder 1000216489776414077 in train \n#create 2 arrays , assisgn band number and assign colurs for band numbers\nbands = ['08', '09', '10', '11', '12', '13', '14', '15', '16']\ncolors = ['gray', 'blue', 'pink', 'orange', 'purple', 'green', 'magenta', 'yellow', 'black']","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:18:31.631909Z","iopub.execute_input":"2023-07-06T07:18:31.63228Z","iopub.status.idle":"2023-07-06T07:18:31.637631Z","shell.execute_reply.started":"2023-07-06T07:18:31.632248Z","shell.execute_reply":"2023-07-06T07:18:31.636847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Check different wave length for bands**","metadata":{}},{"cell_type":"code","source":"for i, band in enumerate(bands):\n    #load the specific folder into a variable\n    plot = np.load(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/train/1000216489776414077/band_{band}.npy')\n    plt.plot(plot[0], color=colors[i])","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:21:04.333038Z","iopub.execute_input":"2023-07-06T07:21:04.333478Z","iopub.status.idle":"2023-07-06T07:21:04.728849Z","shell.execute_reply.started":"2023-07-06T07:21:04.333441Z","shell.execute_reply":"2023-07-06T07:21:04.727825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the shape\nplot.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:51:00.215041Z","iopub.execute_input":"2023-05-31T11:51:00.215589Z","iopub.status.idle":"2023-05-31T11:51:00.222543Z","shell.execute_reply.started":"2023-05-31T11:51:00.215546Z","shell.execute_reply":"2023-05-31T11:51:00.221687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Each band files contain 8 images**","metadata":{}},{"cell_type":"markdown","source":"*  # Understanding bands in train_list","metadata":{}},{"cell_type":"code","source":"#display images in different band.npy files in folder called 1000603527582775543 in train list\n#example : band_08.npy file containes 8 images , and each folders have 9 band files\nfig, axs = plt.subplots(8, len(bands), figsize=(16, 16)) \n\nfor j, band in enumerate(bands):\n    img = np.load(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/train/1000603527582775543/band_{band}.npy')\n    for i in range(8):\n        axs[i,j].imshow(img[..., i]) \n        axs[i,j].set_title(f\"Band {band}\") \n\nplt.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:51:08.810229Z","iopub.execute_input":"2023-05-31T11:51:08.810592Z","iopub.status.idle":"2023-05-31T11:51:19.654779Z","shell.execute_reply.started":"2023-05-31T11:51:08.810565Z","shell.execute_reply":"2023-05-31T11:51:19.653829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* # Understanding masks in train list folder","metadata":{}},{"cell_type":"code","source":"#load the masks files\nhuman_in_mask = np.load(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/train/1000603527582775543/human_individual_masks.npy')\nhuman_pi_mask = np.load(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/train/1000603527582775543/human_pixel_masks.npy')","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:51:56.819322Z","iopub.execute_input":"2023-05-31T11:51:56.820227Z","iopub.status.idle":"2023-05-31T11:51:56.829692Z","shell.execute_reply.started":"2023-05-31T11:51:56.820191Z","shell.execute_reply":"2023-05-31T11:51:56.828863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the shape of human_in_mask file\nhuman_in_mask.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:36:38.538832Z","iopub.execute_input":"2023-05-31T05:36:38.539245Z","iopub.status.idle":"2023-05-31T05:36:38.546041Z","shell.execute_reply.started":"2023-05-31T05:36:38.539212Z","shell.execute_reply":"2023-05-31T05:36:38.544925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len_in_mask = len(human_in_mask[0,0,0])","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:51:49.220181Z","iopub.execute_input":"2023-05-31T11:51:49.22059Z","iopub.status.idle":"2023-05-31T11:51:49.226073Z","shell.execute_reply.started":"2023-05-31T11:51:49.220556Z","shell.execute_reply":"2023-05-31T11:51:49.2249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"human_pi_mask.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:36:58.61251Z","iopub.execute_input":"2023-05-31T05:36:58.612923Z","iopub.status.idle":"2023-05-31T05:36:58.618714Z","shell.execute_reply.started":"2023-05-31T05:36:58.612887Z","shell.execute_reply":"2023-05-31T05:36:58.617962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show the masks.npy using imshow()\nfig, axs = plt.subplots(1, len(human_in_mask[0,0,0])+1) \n    \nfor i in range(len_in_mask):\n    axs[i].imshow(human_in_mask[..., i])\n    axs[i].set_title(\"Individual\")\naxs[i+1].imshow(human_pi_mask)\naxs[i+1].set_title(\"Pixel\")\n\nplt.tight_layout() \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:52:04.347156Z","iopub.execute_input":"2023-05-31T11:52:04.34753Z","iopub.status.idle":"2023-05-31T11:52:05.04612Z","shell.execute_reply.started":"2023-05-31T11:52:04.347502Z","shell.execute_reply":"2023-05-31T11:52:05.044809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore the validation list folder to review a sample","metadata":{}},{"cell_type":"code","source":"for i, band in enumerate(bands):\n    plot_val = np.load(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/validation/1000834164244036115/band_{band}.npy')\n    plt.plot(plot_val[0], color=colors[i], label=f'Band {band}')","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:52:12.971314Z","iopub.execute_input":"2023-05-31T11:52:12.971701Z","iopub.status.idle":"2023-05-31T11:52:14.004475Z","shell.execute_reply.started":"2023-05-31T11:52:12.971657Z","shell.execute_reply":"2023-05-31T11:52:14.003317Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_val.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:52:23.453764Z","iopub.execute_input":"2023-05-31T11:52:23.454138Z","iopub.status.idle":"2023-05-31T11:52:23.462054Z","shell.execute_reply.started":"2023-05-31T11:52:23.454111Z","shell.execute_reply":"2023-05-31T11:52:23.460727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* # Displaying bands in a specific folder in validation list","metadata":{}},{"cell_type":"code","source":"#display images in different band.npy files in folder called 1000834164244036115 in validation list\n#example : band_08.npy file containes 8 images , and each folders have 9 band files\nfig, axs = plt.subplots(8, len(bands), figsize=(16, 16)) \n\nfor j, band in enumerate(bands):\n    images = np.load(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/validation/1000834164244036115/band_{band}.npy')\n    for i in range(8):\n        axs[i,j].imshow(images[..., i]) \n        axs[i,j].set_title(f\"Band {band}\") \n\nplt.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:52:32.337429Z","iopub.execute_input":"2023-05-31T11:52:32.337832Z","iopub.status.idle":"2023-05-31T11:52:42.698751Z","shell.execute_reply.started":"2023-05-31T11:52:32.3378Z","shell.execute_reply":"2023-05-31T11:52:42.697883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* # Understanding the masks in specific folder in validation_list","metadata":{}},{"cell_type":"code","source":"val_masks = np.load(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/validation/1000834164244036115/human_pixel_masks.npy')","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:53:08.815948Z","iopub.execute_input":"2023-05-31T11:53:08.816379Z","iopub.status.idle":"2023-05-31T11:53:08.823547Z","shell.execute_reply.started":"2023-05-31T11:53:08.816345Z","shell.execute_reply":"2023-05-31T11:53:08.822238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shape of human_pixel_masks.npy and it has one image to show\nval_masks.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-31T11:53:10.482168Z","iopub.execute_input":"2023-05-31T11:53:10.482541Z","iopub.status.idle":"2023-05-31T11:53:10.490353Z","shell.execute_reply.started":"2023-05-31T11:53:10.482511Z","shell.execute_reply":"2023-05-31T11:53:10.489063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#to show the human_pixel_masks\nplt.imshow(val_masks)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T12:15:36.158399Z","iopub.execute_input":"2023-05-31T12:15:36.158805Z","iopub.status.idle":"2023-05-31T12:15:36.407813Z","shell.execute_reply.started":"2023-05-31T12:15:36.158767Z","shell.execute_reply":"2023-05-31T12:15:36.406654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore the test list folder to review a sample","metadata":{}},{"cell_type":"markdown","source":"**Check different wave lenghts for bands in test**","metadata":{}},{"cell_type":"code","source":"for i, band in enumerate(bands):\n    plot_test = np.load(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/test/1000834164244036115/band_{band}.npy')\n    plt.plot(plot_test[0], color=colors[i], label=f'Band {band}')","metadata":{"execution":{"iopub.status.busy":"2023-05-31T12:24:06.638164Z","iopub.execute_input":"2023-05-31T12:24:06.638597Z","iopub.status.idle":"2023-05-31T12:24:07.067321Z","shell.execute_reply.started":"2023-05-31T12:24:06.638562Z","shell.execute_reply":"2023-05-31T12:24:07.06633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-31T12:25:59.726211Z","iopub.execute_input":"2023-05-31T12:25:59.72662Z","iopub.status.idle":"2023-05-31T12:25:59.733028Z","shell.execute_reply.started":"2023-05-31T12:25:59.726591Z","shell.execute_reply":"2023-05-31T12:25:59.732158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Contains 8 images in a band","metadata":{}},{"cell_type":"markdown","source":"* # Understanding bands in specific folder in test_list ","metadata":{}},{"cell_type":"code","source":"#display images in different band.npy files in folder called 1000834164244036115 in test list\n#example : band_08.npy file containes 8 images , and each folders have 9 band files\nfig, axs = plt.subplots(8, len(bands), figsize=(16, 16)) \n\nfor j, band in enumerate(bands):\n    test_img = np.load(f'/kaggle/input/google-research-identify-contrails-reduce-global-warming/test/1000834164244036115/band_{band}.npy')\n    for i in range(8):\n        axs[i,j].imshow(test_img[..., i]) \n        axs[i,j].set_title(f\"Band {band}\") \n\nplt.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T12:31:12.372229Z","iopub.execute_input":"2023-05-31T12:31:12.373947Z","iopub.status.idle":"2023-05-31T12:31:23.177535Z","shell.execute_reply.started":"2023-05-31T12:31:12.373897Z","shell.execute_reply":"2023-05-31T12:31:23.175461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Relevant libraries are imported.**\n* Pandas  - creating dataframe\n* Matplotlib and seaborn libraries – data visualizations\n* Numpy – creating array \n* Os  and path – defining path, root or directory\n\nThe root directory of input data is defined.There are 3 main folders in the dataset namely, train,test and validation.The path for these 3 directories are defined and assigned into variables separately.\nI checked the number of samples in train,test and validation folders and print them.\n\nThere are 2 metadata files in json format.The data in json files are read and loaded into dataframe using “pd.read_json”.\nThen, preview the training and validation dataset.\n\n* View dimensions of dataset using “shape” keyword\n* View first 5 rows in dataset using “head()” keyword\n* character variables and numerical variable are identified getting summary of dataset\n* statistical properties are identified such as count,mean,median and IQR.\n* The heatmap is drawn for checking null values.Null values and duplicate values are not in there.\n\nThen, categorical variables and numerical variable are analyzed  for their frequency distribution seperately.\n\nFor seeing relationships between each attributes, draw the pairplot to check distribution after dropping attributes which are irrelevant for getting insights.(drop record_id and timestamp columns)\n\n* In separate 3 folders(train,test,validation), there are images.Assign bands file numbers into an array.Each band file contain 8 images.\n\n* View relationships between attributes checking correlations and draw heatmap\n\n* Explore the train list folder to review a sample and get images from the specific folder (1000216489776414077) in train\n* Check the wave lengths for bands by plots\n* Display images in different band.npy files in folder called 1000603527582775543 in train list\n**Example : band_08.npy file contains 8 images , and each folders have 9 band files**\n\nFor understanding masks in train list folder, masks files are loaded into variables and images are shown in tight_layout.\nSame things are done for understanding bands and masks in validation and test global folders\n","metadata":{}}]}