{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"ASLFR With Pyspark as a data manager and we used TensorFlow for the training of our model and with MediaPipe, for go to create a mask and Landmarks","metadata":{}},{"cell_type":"markdown","source":"## Install dependencies in our environment","metadata":{}},{"cell_type":"code","source":"!pip install mediapipe -q","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:01.337437Z","iopub.execute_input":"2023-08-25T19:06:01.337873Z","iopub.status.idle":"2023-08-25T19:06:10.586062Z","shell.execute_reply.started":"2023-08-25T19:06:01.337841Z","shell.execute_reply":"2023-08-25T19:06:10.584629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflowonspark -q","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:10.587835Z","iopub.execute_input":"2023-08-25T19:06:10.58817Z","iopub.status.idle":"2023-08-25T19:06:19.918733Z","shell.execute_reply.started":"2023-08-25T19:06:10.588143Z","shell.execute_reply":"2023-08-25T19:06:19.917602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyspark -q","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:19.920324Z","iopub.execute_input":"2023-08-25T19:06:19.920635Z","iopub.status.idle":"2023-08-25T19:06:28.95901Z","shell.execute_reply.started":"2023-08-25T19:06:19.920609Z","shell.execute_reply":"2023-08-25T19:06:28.957932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install opencv-python -q","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:28.961645Z","iopub.execute_input":"2023-08-25T19:06:28.961954Z","iopub.status.idle":"2023-08-25T19:06:38.126341Z","shell.execute_reply.started":"2023-08-25T19:06:28.961925Z","shell.execute_reply":"2023-08-25T19:06:38.124884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow-serving-api -q","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:38.128018Z","iopub.execute_input":"2023-08-25T19:06:38.129289Z","iopub.status.idle":"2023-08-25T19:06:47.469783Z","shell.execute_reply.started":"2023-08-25T19:06:38.129216Z","shell.execute_reply":"2023-08-25T19:06:47.468694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import libraries:","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sns\n\nimport tensorflow as tf\nimport tensorflowonspark as tfos\n\nimport os\nos.environ['PYARROW_IGNORE_TIMEZONE'] = '1'\n\nimport pyspark\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql import functions as F\nfrom pyspark.sql.functions import sum, mean, stddev, min, max, avg\nfrom pyspark.sql.functions import row_number, col, desc, when, count, udf, length\nfrom pyspark.sql.types import StringType, ArrayType, IntegerType\nfrom pyspark.sql.window import Window\nimport pyspark.pandas as ps\n\nfrom tqdm.notebook import tqdm\nfrom leven import levenshtein\n\nimport cv2\nimport mediapipe\n\nimport glob\nimport sys\nimport math\nimport gc\nimport sys\nimport sklearn\nimport time\nimport json\nimport re","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:47.471011Z","iopub.execute_input":"2023-08-25T19:06:47.471362Z","iopub.status.idle":"2023-08-25T19:06:47.481185Z","shell.execute_reply.started":"2023-08-25T19:06:47.471332Z","shell.execute_reply":"2023-08-25T19:06:47.480005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create a new session in Spark:","metadata":{}},{"cell_type":"code","source":"spark = SparkSession.builder\\\n.appName(\"ASLFR_TFOS\")\\\n.config(\"spark.driver.memory\", \"4g\")\\\n.config(\"spark.executor.memory\", \"4g\")\\\n.getOrCreate()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:47.482306Z","iopub.execute_input":"2023-08-25T19:06:47.482583Z","iopub.status.idle":"2023-08-25T19:06:47.505713Z","shell.execute_reply.started":"2023-08-25T19:06:47.48256Z","shell.execute_reply":"2023-08-25T19:06:47.503351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load data:","metadata":{}},{"cell_type":"code","source":"# Load data in a PySpark DataFrame\ntrain = spark.read.csv(\"/kaggle/input/asl-fingerspelling/train.csv\", header=True, inferSchema=True)\nmetadata = spark.read.csv(\"/kaggle/input/asl-fingerspelling/supplemental_metadata.csv\", header=True, inferSchema=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:47.50717Z","iopub.execute_input":"2023-08-25T19:06:47.507558Z","iopub.status.idle":"2023-08-25T19:06:48.40892Z","shell.execute_reply.started":"2023-08-25T19:06:47.507529Z","shell.execute_reply":"2023-08-25T19:06:48.407815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show the dataframes:","metadata":{}},{"cell_type":"code","source":"train.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:48.410457Z","iopub.execute_input":"2023-08-25T19:06:48.410968Z","iopub.status.idle":"2023-08-25T19:06:48.528382Z","shell.execute_reply.started":"2023-08-25T19:06:48.410933Z","shell.execute_reply":"2023-08-25T19:06:48.527195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the number of training samples\nprint(f'N_SAMPLES: {train.count()}')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:48.533442Z","iopub.execute_input":"2023-08-25T19:06:48.533819Z","iopub.status.idle":"2023-08-25T19:06:48.729624Z","shell.execute_reply.started":"2023-08-25T19:06:48.533786Z","shell.execute_reply":"2023-08-25T19:06:48.728858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:48.730445Z","iopub.execute_input":"2023-08-25T19:06:48.730696Z","iopub.status.idle":"2023-08-25T19:06:48.844406Z","shell.execute_reply.started":"2023-08-25T19:06:48.730672Z","shell.execute_reply":"2023-08-25T19:06:48.843567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the total number of records in metadata\ntotal_files = metadata.count()\nprint(f\"Total number of files in metadata: {total_files}\")\n\n# Calculate the total number of unique participants in metadata\ntotal_participants = metadata.select(\"participant_id\").distinct().count()\nprint(f\"Total number of participants in metadata: {total_participants}\")\n\n# Calculate the total number of unique phrases in metadata\ntotal_unique_phrases = metadata.select(\"phrase\").distinct().count()\nprint(f\"Total number of unique phrases in metadata: {total_unique_phrases}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:48.845325Z","iopub.execute_input":"2023-08-25T19:06:48.845633Z","iopub.status.idle":"2023-08-25T19:06:50.372763Z","shell.execute_reply.started":"2023-08-25T19:06:48.845606Z","shell.execute_reply":"2023-08-25T19:06:50.371872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Character To Ordinal Encoding","metadata":{}},{"cell_type":"code","source":"# Load JSON file with ordinal character encoding mapping\njson_file_path = \"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\"\nwith open(json_file_path) as json_file:\n    char2ord_json = json.load(json_file)\n\n# Convert JSON Dictionary to a Spark DataFrame\nchar2ord_data = [(char, ord) for char, ord in char2ord_json.items()]\nchar2ord_df = spark.createDataFrame(char2ord_data, [\"Character\", \"OrdinalEncoding\"])\n","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:50.373726Z","iopub.execute_input":"2023-08-25T19:06:50.374036Z","iopub.status.idle":"2023-08-25T19:06:50.496557Z","shell.execute_reply.started":"2023-08-25T19:06:50.374007Z","shell.execute_reply":"2023-08-25T19:06:50.495709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the DataFrame\nchar2ord_df.show(60)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:50.4975Z","iopub.execute_input":"2023-08-25T19:06:50.497793Z","iopub.status.idle":"2023-08-25T19:06:51.931422Z","shell.execute_reply.started":"2023-08-25T19:06:50.497767Z","shell.execute_reply":"2023-08-25T19:06:51.930451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the number of unique characters\nn_unique_characters = len(char2ord_json)\n\nprint(f'N_UNIQUE_CHARACTERS: {n_unique_characters}')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:51.932382Z","iopub.execute_input":"2023-08-25T19:06:51.93266Z","iopub.status.idle":"2023-08-25T19:06:51.942662Z","shell.execute_reply.started":"2023-08-25T19:06:51.932635Z","shell.execute_reply":"2023-08-25T19:06:51.94169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Global Config:","metadata":{}},{"cell_type":"code","source":"# If Notebook Is Run By Committing or In Interactive Mode For Development\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n# Describe Statistics Percentiles\nPERCENTILES = [0.01, 0.10, 0.05, 0.25, 0.50, 0.75, 0.90, 0.95, 0.99, 0.999]\n# Global Random Seed\nSEED = 42\n# Number of Frames to resize recording to\nN_TARGET_FRAMES = 128\n# Global debug flag, takes subset of train\nDEBUG = False\n# Fast Processing\nFAST= True\n# Number of Unique Characters To Predict + Pad Token + SOS Token + EOS Token\nN_UNIQUE_CHARACTERSPAD_TOKEN = n_unique_characters\nSOS_TOKEN = n_unique_characters + 1 # Start Of Sentence\nEOS_TOKEN = n_unique_characters + 2 # End Of Sentence","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:51.945097Z","iopub.execute_input":"2023-08-25T19:06:51.945435Z","iopub.status.idle":"2023-08-25T19:06:51.956673Z","shell.execute_reply.started":"2023-08-25T19:06:51.945409Z","shell.execute_reply":"2023-08-25T19:06:51.955531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore and analyze the data train:","metadata":{}},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:51.958876Z","iopub.execute_input":"2023-08-25T19:06:51.959623Z","iopub.status.idle":"2023-08-25T19:06:51.986813Z","shell.execute_reply.started":"2023-08-25T19:06:51.959589Z","shell.execute_reply":"2023-08-25T19:06:51.985989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.select(\"phrase\").show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:51.98835Z","iopub.execute_input":"2023-08-25T19:06:51.988928Z","iopub.status.idle":"2023-08-25T19:06:52.094104Z","shell.execute_reply.started":"2023-08-25T19:06:51.988899Z","shell.execute_reply":"2023-08-25T19:06:52.093173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Phrase Type","metadata":{}},{"cell_type":"code","source":"# Defines the sort function as a UDF (User-Defined Function)\ndef get_phrase_type_udf(phrase):\n    # Phone Number\n    if re.match(r'^[\\d+-]+$', phrase):\n        return 'phone_number'\n    # URL\n    elif any([substr in phrase for substr in ['www', '.', '/']]) and ' ' not in phrase:\n        return 'url'\n    # Address\n    else:\n        return 'address'\n\n# Add an UDF with Spark\nget_phrase_type_spark_udf = udf(get_phrase_type_udf, StringType())\n\n# Add a new column 'phrase_type' to DataFrame 'train' using the UDF\ntrain_with_phrase_type = train.withColumn('phrase_type', get_phrase_type_spark_udf(train['phrase']))\n","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:52.095097Z","iopub.execute_input":"2023-08-25T19:06:52.095412Z","iopub.status.idle":"2023-08-25T19:06:52.138505Z","shell.execute_reply.started":"2023-08-25T19:06:52.095386Z","shell.execute_reply":"2023-08-25T19:06:52.137567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the resulting DataFrame with the new column\ntrain_with_phrase_type.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:52.139441Z","iopub.execute_input":"2023-08-25T19:06:52.139717Z","iopub.status.idle":"2023-08-25T19:06:53.366716Z","shell.execute_reply.started":"2023-08-25T19:06:52.139692Z","shell.execute_reply":"2023-08-25T19:06:53.365801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_with_phrase_type.printSchema()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:53.367691Z","iopub.execute_input":"2023-08-25T19:06:53.368002Z","iopub.status.idle":"2023-08-25T19:06:53.381941Z","shell.execute_reply.started":"2023-08-25T19:06:53.367972Z","shell.execute_reply":"2023-08-25T19:06:53.380905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Phrase Processing","metadata":{}},{"cell_type":"code","source":"# Split Phrase To Char Array\ntrain_with_char_array = train_with_phrase_type.withColumn(\"phrase_char\", F.split(F.col(\"phrase\"), \"\"))\n\n# Character Length of Phrase\ntrain_with_char_length = train_with_char_array.withColumn(\"phrase_char_len\", F.size(\"phrase_char\"))\n\n# Maximum Input Length\nMAX_PHRASE_LENGTH = train_with_char_length.select(F.max(\"phrase_char_len\")).first()[0]\nprint(f\"MAX_PHRASE_LENGTH: {MAX_PHRASE_LENGTH}\")\n\n# Train DataFrame indexed by sequence_id to conveniently lookup recording data\ntrain_sequence_id = train_with_char_length.withColumnRenamed(\"sequence_id\", \"index\").select(\"index\", \"phrase\", \"phrase_char\", \"phrase_char_len\", \"path\")","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:53.383026Z","iopub.execute_input":"2023-08-25T19:06:53.383363Z","iopub.status.idle":"2023-08-25T19:06:53.904582Z","shell.execute_reply.started":"2023-08-25T19:06:53.383335Z","shell.execute_reply":"2023-08-25T19:06:53.903667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sequence_id.printSchema()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:53.90553Z","iopub.execute_input":"2023-08-25T19:06:53.905825Z","iopub.status.idle":"2023-08-25T19:06:53.914918Z","shell.execute_reply.started":"2023-08-25T19:06:53.9058Z","shell.execute_reply":"2023-08-25T19:06:53.914087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sequence_id.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:53.916517Z","iopub.execute_input":"2023-08-25T19:06:53.916998Z","iopub.status.idle":"2023-08-25T19:06:54.120928Z","shell.execute_reply.started":"2023-08-25T19:06:53.916965Z","shell.execute_reply":"2023-08-25T19:06:54.120142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert to Pandas DF:","metadata":{}},{"cell_type":"code","source":"train_pd = train_sequence_id.toPandas()\n# Phrase Character Length Statistics\ndisplay(train_pd['phrase_char_len'].describe(percentiles=PERCENTILES).to_frame().round(1))","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:54.12181Z","iopub.execute_input":"2023-08-25T19:06:54.122075Z","iopub.status.idle":"2023-08-25T19:06:56.484129Z","shell.execute_reply.started":"2023-08-25T19:06:54.122052Z","shell.execute_reply":"2023-08-25T19:06:56.483199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Character Count Occurrence\nplt.figure(figsize=(15,8))\nplt.title('Character Length Occurrence of Phrases')\nchar_len_counts = train_pd['phrase_char_len'].value_counts().sort_index()\nsns.barplot(x=char_len_counts.index, y=char_len_counts.values)\nplt.xlim(-0.50, train_pd['phrase_char_len'].max() - 1.50)\nplt.xlabel('Phrase Character Length')\nplt.ylabel('Sample Count')\nplt.grid(axis='y')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:56.48511Z","iopub.execute_input":"2023-08-25T19:06:56.48591Z","iopub.status.idle":"2023-08-25T19:06:56.941826Z","shell.execute_reply.started":"2023-08-25T19:06:56.485885Z","shell.execute_reply":"2023-08-25T19:06:56.940576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Find Unique Character","metadata":{}},{"cell_type":"code","source":"# Use Set to keep track of unique characters in phrases\nUNIQUE_CHARACTERS = set()\n\nfor phrase in tqdm(train_pd['phrase_char']):\n    for c in phrase:\n        UNIQUE_CHARACTERS.add(c)\n        \n# Sorted Unique Character\nUNIQUE_CHARACTERS = np.array(sorted(UNIQUE_CHARACTERS))\n# Number of Unique Characters\nN_UNIQUE_CHARACTERS = len(UNIQUE_CHARACTERS)\nprint(f'N_UNIQUE_CHARACTERS: {N_UNIQUE_CHARACTERS}')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:56.943401Z","iopub.execute_input":"2023-08-25T19:06:56.943713Z","iopub.status.idle":"2023-08-25T19:06:57.137335Z","shell.execute_reply.started":"2023-08-25T19:06:56.943689Z","shell.execute_reply":"2023-08-25T19:06:57.136397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Add File Path","metadata":{}},{"cell_type":"code","source":"# Define UDF (User Defined Function) function to get full path\ndef get_file_path(path):\n    return f'/kaggle/input/asl-fingerspelling/{path}'\n\nget_file_path_udf = udf(get_file_path, StringType())\n\n# Add a new column to the DataFrame with the complete paths of the files\ntrain_with_file_path = train_sequence_id.withColumn(\"file_path\", get_file_path_udf(train_sequence_id[\"path\"]))\n","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:57.142236Z","iopub.execute_input":"2023-08-25T19:06:57.142544Z","iopub.status.idle":"2023-08-25T19:06:57.198524Z","shell.execute_reply.started":"2023-08-25T19:06:57.14252Z","shell.execute_reply":"2023-08-25T19:06:57.197605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = train_with_file_path.toPandas()\ntrain_pd.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:57.199464Z","iopub.execute_input":"2023-08-25T19:06:57.199772Z","iopub.status.idle":"2023-08-25T19:06:59.656216Z","shell.execute_reply.started":"2023-08-25T19:06:57.199744Z","shell.execute_reply":"2023-08-25T19:06:59.654592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Example Parquet File","metadata":{}},{"cell_type":"code","source":"# Read the first Parquet file\nexample_parquet_path = train_with_file_path.select(\"file_path\").first()[\"file_path\"]\nexample_parquet_df = spark.read.parquet(example_parquet_path)\nprint(f'Our data set contains {example_parquet_df.count()} rows and {len(example_parquet_df.columns)} columns')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:06:59.659986Z","iopub.execute_input":"2023-08-25T19:06:59.660333Z","iopub.status.idle":"2023-08-25T19:07:01.918314Z","shell.execute_reply.started":"2023-08-25T19:06:59.660303Z","shell.execute_reply":"2023-08-25T19:07:01.917384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example_parquet_df.columns\n# example_parquet_df.printSchema()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:07:01.91942Z","iopub.execute_input":"2023-08-25T19:07:01.919896Z","iopub.status.idle":"2023-08-25T19:07:01.924428Z","shell.execute_reply.started":"2023-08-25T19:07:01.91987Z","shell.execute_reply":"2023-08-25T19:07:01.922886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Video Statistics","metadata":{}},{"cell_type":"code","source":"N = 25 if IS_INTERACTIVE else 60\nN_UNIQUE_FRAMES = []\n\nUNIQUE_FILE_PATHS = pd.Series(train_pd['file_path'].unique())\n\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS.sample(N, random_state=SEED))):\n    df = pd.read_parquet(file_path)\n    for group, group_df in df.groupby('sequence_id'):\n        N_UNIQUE_FRAMES.append(group_df['frame'].nunique())\n\nN_UNIQUE_FRAMES = np.array(N_UNIQUE_FRAMES)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:07:01.925903Z","iopub.execute_input":"2023-08-25T19:07:01.926409Z","iopub.status.idle":"2023-08-25T19:13:18.332831Z","shell.execute_reply.started":"2023-08-25T19:07:01.926383Z","shell.execute_reply":"2023-08-25T19:13:18.329518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_frames_pd = pd.Series(N_UNIQUE_FRAMES)\ndisplay(unique_frames_pd.describe(percentiles=PERCENTILES).to_frame('Value').astype(int))\n\nplt.figure(figsize=(15, 8))\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim + 50, 50))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:13:18.337181Z","iopub.execute_input":"2023-08-25T19:13:18.340031Z","iopub.status.idle":"2023-08-25T19:13:18.858829Z","shell.execute_reply.started":"2023-08-25T19:13:18.339979Z","shell.execute_reply":"2023-08-25T19:13:18.856976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# With N_TARGET_FRAMES = 256 ~85% will be below\nN_UNIQUE_FRAMES_WATERFALL = []\n# Maximum Number of Unique Frames to use\nN_MAX_UNIQUE_FRAMES = 400\n# Compute Percentage\nfor n in range(0, N_MAX_UNIQUE_FRAMES + 1):\n    percentage_above_threshold = np.sum(N_UNIQUE_FRAMES >= n) / len(N_UNIQUE_FRAMES) * 100\n    N_UNIQUE_FRAMES_WATERFALL.append(percentage_above_threshold)\n\nplt.figure(figsize=(18,10))\nplt.title('Waterfall Plot For Number Of Unique Frames')\npd.Series(N_UNIQUE_FRAMES_WATERFALL).plot(kind='bar')\nplt.grid(axis='y')\nplt.xticks([1] + np.arange(5, N_MAX_UNIQUE_FRAMES+5, 5).tolist(), size=8, rotation=45)\nplt.xlabel('Number of Unique Frames', size=16)\nplt.yticks(np.arange(0, 100+5, 10), [f'{i}%' for i in range(0,100+5,10)])\nplt.ylim(0, 100)\nplt.ylabel('Percentage of Samples With At Least N Unique Frames', size=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:13:18.860367Z","iopub.execute_input":"2023-08-25T19:13:18.860714Z","iopub.status.idle":"2023-08-25T19:13:21.081555Z","shell.execute_reply.started":"2023-08-25T19:13:18.860683Z","shell.execute_reply":"2023-08-25T19:13:21.080385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_UNIQUE_FRAMES = []\n\n# Extract unique file paths\nunique_file_paths = train_with_file_path.select(\"file_path\").distinct().rdd.flatMap(lambda x: x).collect()\n\nfor file_path in tqdm(unique_file_paths):\n    df = spark.read.parquet(file_path)\n    unique_frames = df.groupBy(\"sequence_id\").agg(F.countDistinct(\"frame\").alias(\"unique_frames\"))\n    unique_frames_count = unique_frames.select(F.sum(\"unique_frames\")).first()[0]\n    N_UNIQUE_FRAMES.append(unique_frames_count)\n\nN_UNIQUE_FRAMES = np.array(N_UNIQUE_FRAMES)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:13:21.082591Z","iopub.execute_input":"2023-08-25T19:13:21.08297Z","iopub.status.idle":"2023-08-25T19:14:18.105423Z","shell.execute_reply.started":"2023-08-25T19:13:21.082947Z","shell.execute_reply":"2023-08-25T19:14:18.104412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,8))\nplt.title('Distribution of Number of Unique Frames', size=24)\nsns.histplot(data=N_UNIQUE_FRAMES, bins=128, kde=True)\nplt.grid()\nplt.xlabel('Number of Unique Frames')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:14:18.106517Z","iopub.execute_input":"2023-08-25T19:14:18.106806Z","iopub.status.idle":"2023-08-25T19:14:18.574119Z","shell.execute_reply.started":"2023-08-25T19:14:18.106782Z","shell.execute_reply":"2023-08-25T19:14:18.572441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Landmark indices:","metadata":{}},{"cell_type":"code","source":"def get_idxs(df, words_pos, words_neg=[], ret_names=True, idxs_pos=None):\n    idxs = []\n    names = []\n    for w in words_pos:\n        for col in df.columns:\n            # Excluir columnas que no son Landmark\n            if col in ['frame']:\n                continue\n                \n            # Verificar si el nombre de la columna contiene todas las palabras\n            if (w in col) and all([w not in col for w in words_neg]):\n                # Si se proporciona idxs_pos, verifica si el índice de la columna está en la lista\n                if idxs_pos is not None:\n                    col_idx = int(col.split('_')[-1])  # Aquí se asume que el índice está al final\n                    if col_idx in idxs_pos:\n                        idxs.append(col_idx)\n                        names.append(col)\n                else:\n                    idxs.append(col)  # No se necesita el índice si idxs_pos no se proporciona\n                    names.append(col)\n    \n    # Convertir a arrays de Numpy\n    idxs = np.array(idxs)\n    names = np.array(names)\n    # Devolver tanto los índices de columna como los nombres\n    if ret_names:\n        return idxs, names\n    # O solo los índices de columna\n    else:\n        return idxs\n\n# Resto del código igual\n\n# Ejemplo de uso\nwords_pos = [\"left_hand\"]\nwords_neg = [\"z\"]\nidxs_pos = [61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n            291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n            78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n            95, 88, 178, 87, 14, 317, 402, 318, 324, 308]\n\nLEFT_HAND_IDXS0, LEFT_HAND_NAMES0 = get_idxs(example_parquet_df, words_pos, words_neg, idxs_pos=idxs_pos)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:14:18.575725Z","iopub.execute_input":"2023-08-25T19:14:18.576049Z","iopub.status.idle":"2023-08-25T19:14:18.587123Z","shell.execute_reply.started":"2023-08-25T19:14:18.57602Z","shell.execute_reply":"2023-08-25T19:14:18.586027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lips Landmark Face Ids\nLIPS_LANDMARK_IDXS = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n\n# Landmark Indices for Left/Right hand without z axis in raw data\nLEFT_HAND_IDXS0, LEFT_HAND_NAMES0 = get_idxs(example_parquet_df, ['left_hand'], ['z'])\nRIGHT_HAND_IDXS0, RIGHT_HAND_NAMES0 = get_idxs(example_parquet_df, ['right_hand'], ['z'])\nLIPS_IDXS0, LIPS_NAMES0 = get_idxs(example_parquet_df, ['face'], ['z'], idxs_pos=LIPS_LANDMARK_IDXS)\nCOLUMNS0 = np.concatenate((LEFT_HAND_NAMES0, RIGHT_HAND_NAMES0, LIPS_NAMES0))\nN_COLS0 = len(COLUMNS0)\n# Only X/Y axes are used\nN_DIMS0 = 2\n\nprint(f'N_COLS0: {N_COLS0}')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:14:18.588445Z","iopub.execute_input":"2023-08-25T19:14:18.58872Z","iopub.status.idle":"2023-08-25T19:14:18.614998Z","shell.execute_reply.started":"2023-08-25T19:14:18.588691Z","shell.execute_reply":"2023-08-25T19:14:18.613661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Indices in processed data by axes with only dominant hand\nHAND_X_IDXS = np.array(\n        [idx for idx, name in enumerate(LEFT_HAND_NAMES0) if 'x' in name]\n    ).squeeze()\nHAND_Y_IDXS = np.array(\n        [idx for idx, name in enumerate(LEFT_HAND_NAMES0) if 'y' in name]\n    ).squeeze()\n# Names in processed data by axes\nHAND_X_NAMES = LEFT_HAND_NAMES0[HAND_X_IDXS]\nHAND_Y_NAMES = LEFT_HAND_NAMES0[HAND_Y_IDXS]","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:14:18.61617Z","iopub.execute_input":"2023-08-25T19:14:18.616545Z","iopub.status.idle":"2023-08-25T19:14:18.634676Z","shell.execute_reply.started":"2023-08-25T19:14:18.616514Z","shell.execute_reply":"2023-08-25T19:14:18.63357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Indices in processed data by axes with only dominant hand\n#HAND_X_IDXS = [idx for idx, name in enumerate(LEFT_HAND_NAMES0) if 'x' in name]\n#HAND_Y_IDXS = [idx for idx, name in enumerate(LEFT_HAND_NAMES0) if 'y' in name]\n# Names in processed data by axes\n#HAND_X_NAMES = [LEFT_HAND_NAMES0[idx] for idx in HAND_X_IDXS]\n#HAND_Y_NAMES = [LEFT_HAND_NAMES0[idx] for idx in HAND_Y_IDXS]","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:14:18.635912Z","iopub.execute_input":"2023-08-25T19:14:18.636391Z","iopub.status.idle":"2023-08-25T19:14:18.65091Z","shell.execute_reply.started":"2023-08-25T19:14:18.636366Z","shell.execute_reply":"2023-08-25T19:14:18.649554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Number Of Non-NaN Frames","metadata":{}},{"cell_type":"code","source":"\n# Define the PreprocessLayerNonNaN class\nclass PreprocessLayerNonNaN(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayerNonNaN, self).__init__()\n\n# Definir la función de preprocesamiento en TensorFlow\n@tf.function\ndef preprocess_data(data):\n    data = tf.where(tf.math.is_nan(data), 0.0, data)\n    hands = tf.slice(data, [0, 0], [-1, 84])\n    hands = tf.abs(hands)\n    mask = tf.reduce_sum(hands, axis=1)\n    mask = tf.not_equal(mask, 0)\n    data = tf.boolean_mask(data, mask)\n    return data\n\n# Obtener la lista de rutas únicas de archivos parquet y convertir en lista\nUNIQUE_FILE_PATHS = train_with_file_path.select(\"file_path\").distinct().rdd.flatMap(lambda x: x).collect()\n\n# Número de parquet chunks para analizar\nN = 5 if (IS_INTERACTIVE or FAST) else len(UNIQUE_FILE_PATHS)\n\n# Lista para almacenar el número de frames no NaN en la grabación\nN_NON_NAN_FRAMES = []\n\n# Iterar sobre las rutas de archivos únicos\nfor file_path in tqdm(UNIQUE_FILE_PATHS[:N]):\n    # Leer el archivo parquet en un DataFrame\n    df = spark.read.parquet(file_path)\n    \n    # Agrupar por sequence_id y contar\n    group_df = df.groupBy(\"sequence_id\").agg({\"sequence_id\": \"count\"})\n    \n    # Iterar sobre cada grupo en el DataFrame\n    for row in group_df.collect():\n        group = row.sequence_id\n        count = row[\"count(sequence_id)\"]\n        \n        # Filtrar el DataFrame por sequence_id\n        group_df = df.filter(col(\"sequence_id\") == group)\n        \n        # Seleccionar las columnas relevantes y convertir a una lista de filas\n        group_rows = group_df.collect()\n        group_data = [row.asDict() for row in group_rows]\n        \n        # Preprocesar los datos utilizando la función definida\n        preprocessed_data = [preprocess_data(tf.constant([row[col] for col in COLUMNS0[1:]], dtype=tf.float32)) for row in group_data]\n\n        # Preprocesar los datos utilizando la función definida\n        #preprocessed_data = [preprocess_data(tf.constant(row[COLUMNS0[1:]], dtype=tf.float32)) for row in group_data]\n        \n        # Calcular el número de frames después del preprocesamiento\n        num_preprocessed_frames = len(preprocessed_data)\n        \n        # Agregar a la lista N_NON_NAN_FRAMES\n        N_NON_NAN_FRAMES.append((group, count, num_preprocessed_frames))\n\n# Convertir a DataFrame de Spark\nschema = [\"sequence_id\", \"Count\", \"# Frames\"]\nn_non_nan_frames_rows = [spark.createDataFrame([data], schema=schema) for data in N_NON_NAN_FRAMES]\nN_NON_NAN_FRAMES = spark.union(n_non_nan_frames_rows)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:16:30.904138Z","iopub.execute_input":"2023-08-25T19:16:30.904542Z","iopub.status.idle":"2023-08-25T19:17:00.060535Z","shell.execute_reply.started":"2023-08-25T19:16:30.904513Z","shell.execute_reply":"2023-08-25T19:17:00.057326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayerNonNaN(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayerNonNaN, self).__init__()\n\n    @tf.function\n    def call(self, data0):\n        # Fill NaN Values With 0\n        data = tf.where(tf.math.is_nan(data0), 0.0, data0)\n\n        # Hacky\n        data = data[None]\n\n        # Empty Hand Frame Filtering\n        hands = tf.slice(data, [0, 0, 0], [-1, -1, 84])\n        hands = tf.abs(hands)\n        mask = tf.reduce_sum(hands, axis=2)\n        mask = tf.not_equal(mask, 0)\n        data = tf.boolean_mask(data, mask)\n\n        # Pad Zeros\n        n_frames = tf.shape(data)[0]\n        n_padding = N_TARGET_FRAMES - n_frames\n        padding = tf.zeros([n_padding, N_COLS0], dtype=tf.float32)\n        data = tf.concat([data, padding], axis=0)\n\n        # Downsample\n        data = tf.image.resize(\n            data[None],\n            [1, N_TARGET_FRAMES],\n            method=tf.image.ResizeMethod.BILINEAR,\n        )[0]\n\n        return data\n\npreprocess_layer_non_nan = PreprocessLayerNonNaN()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.313487Z","iopub.status.idle":"2023-08-25T19:15:05.313847Z","shell.execute_reply.started":"2023-08-25T19:15:05.313668Z","shell.execute_reply":"2023-08-25T19:15:05.313682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Obtener la lista de rutas únicas de archivos parquet y convertir en lista\nUNIQUE_FILE_PATHS = train_with_file_path.select(\"file_path\").distinct().rdd.flatMap(lambda x: x).collect()\n\n# Número de parquet chunks para analizar\nN = 5 if (IS_INTERACTIVE or FAST) else len(UNIQUE_FILE_PATHS)\n\n# Lista para almacenar el número de frames no NaN en la grabación\nN_NON_NAN_FRAMES = []\n\n# Iterar sobre las rutas de archivos únicos\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS)):\n    # Leer el archivo parquet en un DataFrame\n    df = spark.read.parquet(file_path)\n    \n    # Iterar sobre cada grupo en el DataFrame\n    for row in df.groupBy(\"sequence_id\").agg({\"sequence_id\": \"count\"}).collect():\n        group = row.sequence_id\n        count = row[\"count(sequence_id)\"]\n        \n        # Filtrar el DataFrame por sequence_id\n        group_df = df.filter(col(\"sequence_id\") == group)\n        \n        # Seleccionar las columnas relevantes y convertir a una lista de filas\n        frames = group_df.select(*COLUMNS0.tolist()).collect()\n        frames_list = [list(row) for row in frames]\n        \n        # Convertir la lista de filas a un tensor de TensorFlow\n        frames_tensor = convert_to_tensor(frames_list)\n        \n        # Aplicar la función UDF para contar el número de filas después del preprocesamiento\n        num_preprocessed_frames = count_rows_udf(frames_tensor)\n        \n        # Agregar a la lista N_NON_NAN_FRAMES\n        N_NON_NAN_FRAMES.append((group, count, num_preprocessed_frames))\n\n# Convertir a DataFrame de Spark\nresult_schema = [\"sequence_id\", \"Count\", \"# Frames\"]\nresult_rdd = spark.sparkContext.parallelize(N_NON_NAN_FRAMES)\nresult_df = result_rdd.toDF(result_schema)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.315903Z","iopub.status.idle":"2023-08-25T19:15:05.316336Z","shell.execute_reply.started":"2023-08-25T19:15:05.31615Z","shell.execute_reply":"2023-08-25T19:15:05.316165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Available GPUs:\", tf.config.experimental.list_physical_devices('GPU'))","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.318494Z","iopub.status.idle":"2023-08-25T19:15:05.319011Z","shell.execute_reply.started":"2023-08-25T19:15:05.318792Z","shell.execute_reply":"2023-08-25T19:15:05.318816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n    TensorFlow layer to process data in TFLite.\n    Data needs to be processed in the model itself, so we cannot use Python.\n\"\"\"\nclass PreprocessLayerNonNaN(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayerNonNaN, self).__init__()\n    \n    def call(self, data0):\n        # Fill NaN Values With 0\n        data = tf.where(tf.math.is_nan(data0), 0.0, data0)\n        \n        # Hacky\n        data = data[None]\n        \n        # Empty Hand Frame Filtering\n        hands = tf.slice(data, [0, 0, 0], [-1, -1, 84])\n        hands = tf.abs(hands)\n        mask = tf.reduce_sum(hands, axis=2)\n        mask = tf.not_equal(mask, 0)\n        data = data[mask][None]\n        data = tf.squeeze(data, axis=[0])\n        \n        return data\n\npreprocess_layer_non_nan = PreprocessLayerNonNaN()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.321562Z","iopub.status.idle":"2023-08-25T19:15:05.3229Z","shell.execute_reply.started":"2023-08-25T19:15:05.322637Z","shell.execute_reply":"2023-08-25T19:15:05.322665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Definir la función de preprocesamiento para cada archivo Parquet\ndef preprocess_file(file_path, preprocess_layer):\n    # Leer el archivo Parquet en un DataFrame\n    df = spark.read.parquet(file_path)\n\n    preprocessed_data = []\n    for row in df.groupBy(\"sequence_id\").agg({\"sequence_id\": \"count\"}).collect():\n        group = row.sequence_id\n        count = row[\"count(sequence_id)\"]\n\n        group_df = df.filter(col(\"sequence_id\") == group)\n\n        # Seleccionar las columnas relevantes\n        frames = group_df.select(*COLUMNS0.tolist())\n\n        # Convertir a un tensor de TensorFlow\n        frames_tensor = tf.convert_to_tensor(frames.toPandas().values, dtype=tf.float32)\n\n        # Preprocesar los frames utilizando la capa PreprocessLayer\n        frames_processed = preprocess_layer(frames_tensor)\n\n        preprocessed_data.append((group, count, frames_processed.shape[0]))\n\n    return preprocessed_data\n\n# Aplicar la función de preprocesamiento a cada archivo Parquet\npreprocessed_results = []\nfor file_path in unique_file_paths:\n    preprocessed_results.extend(preprocess_file(file_path, preprocess_layer_non_nan))\n\n# Convertir los resultados en un DataFrame de Spark\nresult_schema = [\"sequence_id\", \"Count\", \"# Frames\"]\nresult_rdd = spark.sparkContext.parallelize(preprocessed_results)\nresult_df = result_rdd.toDF(result_schema)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.324047Z","iopub.status.idle":"2023-08-25T19:15:05.32522Z","shell.execute_reply.started":"2023-08-25T19:15:05.324972Z","shell.execute_reply":"2023-08-25T19:15:05.324996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df.printSchema()\n# Detener la sesión de Spark\nspark.stop()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.32724Z","iopub.status.idle":"2023-08-25T19:15:05.32787Z","shell.execute_reply.started":"2023-08-25T19:15:05.327607Z","shell.execute_reply":"2023-08-25T19:15:05.327639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayerNonNaN(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayerNonNaN, self).__init__()\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_COLS0], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Fill NaN Values With 0\n        data = tf.where(tf.math.is_nan(data0), 0.0, data0)\n        \n        # Hacky\n        data = data[None]\n        \n        # Empty Hand Frame Filtering\n        hands = tf.slice(data, [0,0,0], [-1, -1, 84])\n        hands = tf.abs(hands)\n        mask = tf.reduce_sum(hands, axis=2)\n        mask = tf.not_equal(mask, 0)\n        data = data[mask][None]\n        data = tf.squeeze(data, axis=[0])\n        \n        return data\n    \npreprocess_layer_non_nan = PreprocessLayerNonNaN()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.329329Z","iopub.status.idle":"2023-08-25T19:15:05.329802Z","shell.execute_reply.started":"2023-08-25T19:15:05.329565Z","shell.execute_reply":"2023-08-25T19:15:05.329586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Obtener la lista de rutas únicas de archivos parquet y convertir en lista\nUNIQUE_FILE_PATHS = train_with_file_path.select(\"file_path\").distinct().rdd.flatMap(lambda x: x).collect()\n\n# Número de parquet chunks para analizar\nN = 5 if (IS_INTERACTIVE or FAST) else len(UNIQUE_FILE_PATHS)\n\n# Lista para almacenar el número de frames no NaN en la grabación\nN_NON_NAN_FRAMES = []\n\n# Iterar sobre las rutas de archivos únicos\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS)):\n    # Leer el archivo parquet en un DataFrame\n    df = spark.read.parquet(file_path)\n    \n    # Iterar sobre cada grupo en el DataFrame\n    for row in df.groupBy(\"sequence_id\").agg({\"sequence_id\": \"count\"}).collect():\n        group = row.sequence_id\n        count = row[\"count(sequence_id)\"]\n        \n        # Filtrar el DataFrame por sequence_id\n        group_df = df.filter(col(\"sequence_id\") == group)\n        \n        # Seleccionar las columnas relevantes y convertir a un DataFrame de pandas\n        frames = group_df.select(*COLUMNS0.tolist()).toPandas().values\n        \n        # Preprocesar los frames utilizando la capa PreprocessLayer\n        frames = preprocess_layer_non_nan(frames)\n        \n        # Agregar a la lista N_NON_NAN_FRAMES\n        N_NON_NAN_FRAMES.append((group, count, len(frames)))\n\n# Convertir a DataFrame de Spark\nschema = [\"sequence_id\", \"Count\", \"# Frames\"]\nn_non_nan_frames_rows = [spark.createDataFrame([data], schema=schema) for data in N_NON_NAN_FRAMES]\nN_NON_NAN_FRAMES = spark.union(n_non_nan_frames_rows)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.331913Z","iopub.status.idle":"2023-08-25T19:15:05.332409Z","shell.execute_reply.started":"2023-08-25T19:15:05.332147Z","shell.execute_reply":"2023-08-25T19:15:05.332168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of frames in each video with hand coordinates\ndisplay(N_NON_NAN_FRAMES.describe(percentiles=PERCENTILES).astype(int))\n\nN_NON_NAN_FRAMES.plot(kind='hist', bins=128, figsize=(15,8))\nplt.title('Number of Non NaN Frames', size=24)\nplt.grid()\nxlim = np.percentile(N_NON_NAN_FRAMES, 99)\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+32, 32))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.334777Z","iopub.status.idle":"2023-08-25T19:15:05.335509Z","shell.execute_reply.started":"2023-08-25T19:15:05.335299Z","shell.execute_reply":"2023-08-25T19:15:05.335321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tensorflow Preprocess Layer","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_COLS0], dtype=tf.float32),),\n    )\n    def call(self, data0, resize=True):\n        # Fill NaN Values With 0\n        data = tf.where(tf.math.is_nan(data0), 0.0, data0)\n        \n        # Hacky\n        data = data[None]\n        \n        # Empty Hand Frame Filtering\n        hands = tf.slice(data, [0,0,0], [-1, -1, 84])\n        hands = tf.abs(hands)\n        mask = tf.reduce_sum(hands, axis=2)\n        mask = tf.not_equal(mask, 0)\n        data = data[mask][None]\n        \n        # Pad Zeros\n        N_FRAMES = len(data[0])\n        if N_FRAMES < N_TARGET_FRAMES:\n            data = tf.concat((\n                data,\n                tf.zeros([1,N_TARGET_FRAMES-N_FRAMES,N_COLS], dtype=tf.float32)\n            ), axis=1)\n        # Downsample\n        data = tf.image.resize(\n            data,\n            [1, N_TARGET_FRAMES],\n            method=tf.image.ResizeMethod.BILINEAR,\n        )\n        \n        # Squeeze Batch Dimension\n        data = tf.squeeze(data, axis=[0])\n        \n        return data\n    \npreprocess_layer = PreprocessLayer()\n\ninputs = group_df[COLUMNS0].values\ninputs = inputs[:1]\n\nframes = preprocess_layer(inputs)\n\nprint(f'inputs shape: {inputs.shape}')\nprint(f'frames shape: {frames.shape}, NaN count: {np.isnan(frames).sum()}')","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.337016Z","iopub.status.idle":"2023-08-25T19:15:05.337418Z","shell.execute_reply.started":"2023-08-25T19:15:05.337212Z","shell.execute_reply":"2023-08-25T19:15:05.337228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create **X/Y**","metadata":{}},{"cell_type":"code","source":"# Target Arrays Processed Input Videos\nX = np.zeros([N_SAMPLES, N_TARGET_FRAMES, N_COLS], dtype=np.float32)\n# Ordinally Encoded Target With value 59 for pad token\ny = np.full(shape=[N_SAMPLES, N_TARGET_FRAMES], fill_value=N_UNIQUE_CHARACTERS, dtype=np.int8)\n# Phrase Type\ny_phrase_type = np.empty(shape=[N_SAMPLES], dtype=object)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.3389Z","iopub.status.idle":"2023-08-25T19:15:05.339334Z","shell.execute_reply.started":"2023-08-25T19:15:05.339134Z","shell.execute_reply":"2023-08-25T19:15:05.33915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All Unique Parquet Files\nUNIQUE_FILE_PATHS = pd.Series(train_df['file_path'].unique())\nN_UNIQUE_FILE_PATHS = len(UNIQUE_FILE_PATHS)\n# Counter to keep track of sample\nrow = 0\ncount = 0\n# Compressed Parquet Files\nPath('train_landmark_subsets').mkdir(parents=True, exist_ok=True)\n# Numbre Of Frames Per Character\nN_FRAMES_PER_CHARACTER = []\n# Minimum Number Of Frames Per Character\nMIN_NUM_FRAMES_PER_CHARACTER = 4\nVALID_IDXS = []\n\n# Fill Arrays\nfor idx, file_path in enumerate(tqdm(UNIQUE_FILE_PATHS)):\n    # Progress Logging\n    print(f'Processed {idx:02d}/{N_UNIQUE_FILE_PATHS} parquet files')\n    # Read parquet file\n    df = pd.read_parquet(file_path)\n    # Save COLUMN Subset of parquet files for TFLite Model verficiation\n    name = file_path.split('/')[-1]\n    if idx < 10:\n        df[COLUMNS0].to_parquet(f'train_landmark_subsets/{name}', engine='pyarrow', compression='zstd')\n    # Iterate Over Samples\n    for group, group_df in df.groupby('sequence_id'):\n        # Number of Frames Per Character\n        n_frames_per_character =  len(group_df[COLUMNS0].values) / len(train_sequence_id.loc[group, 'phrase_char'])\n        N_FRAMES_PER_CHARACTER.append(n_frames_per_character)\n        if n_frames_per_character < MIN_NUM_FRAMES_PER_CHARACTER:\n            count = count + 1\n            continue\n        else:\n            # Add Valid Index\n            VALID_IDXS.append(count)\n            count = count + 1\n        \n        # Get Processed Frames and non empty frame indices\n        frames = preprocess_layer(group_df[COLUMNS0].values)\n        assert frames.ndim == 2\n        # Assign\n        X[row] = frames\n        # Add Target By Ordinally Encoding Characters\n        phrase_char = train_sequence_id.loc[group, 'phrase_char']\n        for col, char in enumerate(phrase_char):\n            y[row, col] = CHAR2ORD.get(char)\n        # Add EOS Token\n        y[row, col+1] = EOS_TOKEN\n        # Phrase Type\n        y_phrase_type[row] = train_sequence_id.loc[group, 'phrase_type']\n        # Row Count\n        row += 1\n    # clean up\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.340253Z","iopub.status.idle":"2023-08-25T19:15:05.340579Z","shell.execute_reply.started":"2023-08-25T19:15:05.340432Z","shell.execute_reply":"2023-08-25T19:15:05.340446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stop Spark session:\n#spark.stop()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T19:15:05.34206Z","iopub.status.idle":"2023-08-25T19:15:05.342485Z","shell.execute_reply.started":"2023-08-25T19:15:05.342312Z","shell.execute_reply":"2023-08-25T19:15:05.342329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}