{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:14.689484Z","iopub.execute_input":"2023-07-01T03:30:14.689836Z","iopub.status.idle":"2023-07-01T03:30:14.695052Z","shell.execute_reply.started":"2023-07-01T03:30:14.689811Z","shell.execute_reply":"2023-07-01T03:30:14.694081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta = pd.read_csv('/kaggle/input/asl-fingerspelling/supplemental_metadata.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:14.706066Z","iopub.execute_input":"2023-07-01T03:30:14.706794Z","iopub.status.idle":"2023-07-01T03:30:14.811313Z","shell.execute_reply.started":"2023-07-01T03:30:14.706758Z","shell.execute_reply":"2023-07-01T03:30:14.810341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:14.813431Z","iopub.execute_input":"2023-07-01T03:30:14.8137Z","iopub.status.idle":"2023-07-01T03:30:14.823811Z","shell.execute_reply.started":"2023-07-01T03:30:14.813672Z","shell.execute_reply":"2023-07-01T03:30:14.823072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/asl-fingerspelling/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:14.824815Z","iopub.execute_input":"2023-07-01T03:30:14.825722Z","iopub.status.idle":"2023-07-01T03:30:14.95069Z","shell.execute_reply.started":"2023-07-01T03:30:14.825653Z","shell.execute_reply":"2023-07-01T03:30:14.949833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:14.952325Z","iopub.execute_input":"2023-07-01T03:30:14.952832Z","iopub.status.idle":"2023-07-01T03:30:14.962906Z","shell.execute_reply.started":"2023-07-01T03:30:14.952809Z","shell.execute_reply":"2023-07-01T03:30:14.962139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Specify the path to the Parquet file\nparquet_file_path = '/kaggle/input/asl-fingerspelling/supplemental_landmarks/369584223.parquet'\n\n# Read the Parquet file into a DataFrame\ndf = pd.read_parquet(parquet_file_path)\n\ndf = df[:1000]\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:14.963837Z","iopub.execute_input":"2023-07-01T03:30:14.964057Z","iopub.status.idle":"2023-07-01T03:30:16.434938Z","shell.execute_reply.started":"2023-07-01T03:30:14.964036Z","shell.execute_reply":"2023-07-01T03:30:16.433656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:16.436011Z","iopub.execute_input":"2023-07-01T03:30:16.436597Z","iopub.status.idle":"2023-07-01T03:30:16.445939Z","shell.execute_reply.started":"2023-07-01T03:30:16.436571Z","shell.execute_reply":"2023-07-01T03:30:16.445125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:16.447608Z","iopub.execute_input":"2023-07-01T03:30:16.448369Z","iopub.status.idle":"2023-07-01T03:30:16.505886Z","shell.execute_reply.started":"2023-07-01T03:30:16.448344Z","shell.execute_reply":"2023-07-01T03:30:16.504621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Choose the x and y columns\nx_column = 'x_face_2'  # Replace 'x' with the actual column name for x-coordinate\ny_column = 'y_face_2'  # Replace 'y' with the actual column name for y-coordinate\n\n# Plot the x and y coordinates\nplt.scatter(df[x_column][:30], df[y_column][:30])\nplt.xlabel('X Coordinate')\nplt.ylabel('Y Coordinate')\nplt.title('Scatter Plot of X and Y Coordinates')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:16.507033Z","iopub.execute_input":"2023-07-01T03:30:16.507334Z","iopub.status.idle":"2023-07-01T03:30:16.740271Z","shell.execute_reply.started":"2023-07-01T03:30:16.50731Z","shell.execute_reply":"2023-07-01T03:30:16.739385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Specify the path to the Parquet file\nparquet_file_path = '/kaggle/input/asl-fingerspelling/train_landmarks/5414471.parquet'\n\n# Read the Parquet file into a DataFrame\ndf_train = pd.read_parquet(parquet_file_path)\n\n\ndf_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:16.741261Z","iopub.execute_input":"2023-07-01T03:30:16.742505Z","iopub.status.idle":"2023-07-01T03:30:21.085671Z","shell.execute_reply.started":"2023-07-01T03:30:16.742457Z","shell.execute_reply":"2023-07-01T03:30:21.084985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Choose the x and y columns\nx_column = 'x_face_2'  # Replace 'x' with the actual column name for x-coordinate\ny_column = 'y_face_2'  # Replace 'y' with the actual column name for y-coordinate\n\n# Plot the x and y coordinates\nplt.scatter(df_train[x_column][:50], df_train[y_column][:50])\nplt.xlabel('X Coordinate')\nplt.ylabel('Y Coordinate')\nplt.title('Scatter Plot of X and Y Coordinates')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:30:21.087956Z","iopub.execute_input":"2023-07-01T03:30:21.089235Z","iopub.status.idle":"2023-07-01T03:30:21.291184Z","shell.execute_reply.started":"2023-07-01T03:30:21.089179Z","shell.execute_reply":"2023-07-01T03:30:21.290116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyarrow.parquet as pq\n\n# Specify the path to the Parquet file\nparquet_file_path = '/kaggle/input/asl-fingerspelling/supplemental_landmarks/369584223.parquet'\n\n# Load the Parquet file into a Pandas DataFrame\ndf = pq.read_table(parquet_file_path).to_pandas()\n\n# Set the random seed for reproducibility\nnp.random.seed(42)\n\n# Sample a subset of the data for EDA (adjust the sample size as needed)\nsample_size = 1000\ndf_sample = df.sample(n=sample_size)","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:34:53.698838Z","iopub.execute_input":"2023-07-01T03:34:53.699197Z","iopub.status.idle":"2023-07-01T03:34:54.746556Z","shell.execute_reply.started":"2023-07-01T03:34:53.69915Z","shell.execute_reply":"2023-07-01T03:34:54.745702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample.describe()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:35:17.221774Z","iopub.execute_input":"2023-07-01T03:35:17.222095Z","iopub.status.idle":"2023-07-01T03:35:19.885934Z","shell.execute_reply.started":"2023-07-01T03:35:17.222072Z","shell.execute_reply":"2023-07-01T03:35:19.885118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the Parquet file schema\nparquet_file = pq.ParquetFile(parquet_file_path)\ncolumn_names = parquet_file.schema.names\n\n# Print the column names\nprint(column_names)","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:37:23.507043Z","iopub.execute_input":"2023-07-01T03:37:23.507399Z","iopub.status.idle":"2023-07-01T03:37:23.529878Z","shell.execute_reply.started":"2023-07-01T03:37:23.507373Z","shell.execute_reply":"2023-07-01T03:37:23.528841Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Specify the columns you want to load\nselected_columns = ['x_face_0', 'x_face_1', 'x_face_2', 'x_face_3', 'x_face_4', 'x_face_5','y_face_1','y_face_2', 'y_face_3', 'y_face_4', 'y_face_5']","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:39:17.624379Z","iopub.execute_input":"2023-07-01T03:39:17.624721Z","iopub.status.idle":"2023-07-01T03:39:17.629413Z","shell.execute_reply.started":"2023-07-01T03:39:17.624698Z","shell.execute_reply":"2023-07-01T03:39:17.628073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the selected columns from the Parquet file into a Pandas DataFrame\ndf = pq.read_table(parquet_file_path, columns=selected_columns).to_pandas()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:39:20.281122Z","iopub.execute_input":"2023-07-01T03:39:20.281506Z","iopub.status.idle":"2023-07-01T03:39:20.345009Z","shell.execute_reply.started":"2023-07-01T03:39:20.281479Z","shell.execute_reply":"2023-07-01T03:39:20.343732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[:100]","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:39:33.998341Z","iopub.execute_input":"2023-07-01T03:39:33.998665Z","iopub.status.idle":"2023-07-01T03:39:34.017915Z","shell.execute_reply.started":"2023-07-01T03:39:33.998642Z","shell.execute_reply":"2023-07-01T03:39:34.017183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_matrix = df.corr()\nprint(correlation_matrix)","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:41:25.293887Z","iopub.execute_input":"2023-07-01T03:41:25.294266Z","iopub.status.idle":"2023-07-01T03:41:25.351399Z","shell.execute_reply.started":"2023-07-01T03:41:25.29424Z","shell.execute_reply":"2023-07-01T03:41:25.35038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(data=df[selected_columns])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:41:58.237989Z","iopub.execute_input":"2023-07-01T03:41:58.238345Z","iopub.status.idle":"2023-07-01T03:41:58.980023Z","shell.execute_reply.started":"2023-07-01T03:41:58.238319Z","shell.execute_reply":"2023-07-01T03:41:58.979362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(data=df[selected_columns])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:42:06.914416Z","iopub.execute_input":"2023-07-01T03:42:06.914746Z","iopub.status.idle":"2023-07-01T03:42:59.826432Z","shell.execute_reply.started":"2023-07-01T03:42:06.914722Z","shell.execute_reply":"2023-07-01T03:42:59.825603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_matrix = df[selected_columns].corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:43:23.947079Z","iopub.execute_input":"2023-07-01T03:43:23.947457Z","iopub.status.idle":"2023-07-01T03:43:24.407702Z","shell.execute_reply.started":"2023-07-01T03:43:23.947431Z","shell.execute_reply":"2023-07-01T03:43:24.406321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in selected_columns:\n    sns.histplot(df[column], kde=True)\n    plt.xlabel(column)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T03:43:25.015053Z","iopub.execute_input":"2023-07-01T03:43:25.015409Z","iopub.status.idle":"2023-07-01T03:43:36.779577Z","shell.execute_reply.started":"2023-07-01T03:43:25.015383Z","shell.execute_reply":"2023-07-01T03:43:36.778726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}