{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook prepares the TFRecords for my solution. The code is based on Rohith Ingilela's [notebook](https://www.kaggle.com/code/irohith/aslfr-preprocess-dataset), and the selected landmarks are partly based on [Hoiso's solution](https://www.kaggle.com/competitions/asl-signs/discussion/406978) from the previous competition: I took the same lips and nose landmarks, but did not took the eyes landmarks and added more pose landmarks. And, of course, all hand landmarks are included.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport json\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-08-27T19:52:23.912702Z","iopub.execute_input":"2023-08-27T19:52:23.913461Z","iopub.status.idle":"2023-08-27T19:52:33.701014Z","shell.execute_reply.started":"2023-08-27T19:52:23.913424Z","shell.execute_reply":"2023-08-27T19:52:33.699858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inpdir = \"/kaggle/input/asl-fingerspelling\"\ndf = pd.read_csv(f'{inpdir}/train.csv')\ndf[\"phrase_bytes\"] = df[\"phrase\"].map(lambda x: x.encode(\"utf-8\"))\ndisplay(df.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-27T19:53:23.37653Z","iopub.execute_input":"2023-08-27T19:53:23.376955Z","iopub.status.idle":"2023-08-27T19:53:23.623Z","shell.execute_reply.started":"2023-08-27T19:53:23.376923Z","shell.execute_reply":"2023-08-27T19:53:23.62195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pad_token = 'P'\npad_token_idx = 59\n\nwith open (\"/kaggle/input/asl-fingerspelling/character_to_prediction_index.json\", \"r\") as f:\n    char_to_num = json.load(f)\n\nchar_to_num[pad_token] = pad_token_idx\n\ntable = tf.lookup.StaticHashTable(\n    initializer=tf.lookup.KeyValueTensorInitializer(\n        keys=list(char_to_num.keys()),\n        values=list(char_to_num.values()),\n    ),\n    default_value=tf.constant(-1),\n    name=\"class_weight\"\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-27T19:54:16.818757Z","iopub.execute_input":"2023-08-27T19:54:16.82144Z","iopub.status.idle":"2023-08-27T19:54:16.930307Z","shell.execute_reply.started":"2023-08-27T19:54:16.821402Z","shell.execute_reply":"2023-08-27T19:54:16.929129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NOSE=[\n    1,2,98,327\n]\nLIP = [ 0, \n    61, 185, 40, 39, 37, 267, 269, 270, 409,\n    291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n    78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n    95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n]\nFACE = LIP+NOSE\nFACE.sort()\n\nLPOSE = [11, 13, 15, 17, 19, 21, 23]\nRPOSE = [12, 14, 16, 18, 20, 22, 24]\nPOSE = LPOSE + RPOSE\n\nX = [f'x_right_hand_{i}' for i in range(21)] + [f'x_left_hand_{i}' for i in range(21)] + [f'x_face_{i}' for i in FACE] + [f'x_pose_{i}' for i in POSE]\nY = [f'y_right_hand_{i}' for i in range(21)] + [f'y_left_hand_{i}' for i in range(21)] + [f'y_face_{i}' for i in FACE] + [f'y_pose_{i}' for i in POSE]\n\nSEL_COLS = X + Y\n\nprint('SEL_COLS size:' + str(len(SEL_COLS)))","metadata":{"execution":{"iopub.status.busy":"2023-08-27T19:56:23.651018Z","iopub.execute_input":"2023-08-27T19:56:23.651451Z","iopub.status.idle":"2023-08-27T19:56:23.666901Z","shell.execute_reply.started":"2023-08-27T19:56:23.651421Z","shell.execute_reply":"2023-08-27T19:56:23.665127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_landmarks = pd.read_parquet('/kaggle/input/asl-fingerspelling/train_landmarks/1019715464.parquet')\ntrain_landmarks.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-27T19:58:43.819489Z","iopub.execute_input":"2023-08-27T19:58:43.81994Z","iopub.status.idle":"2023-08-27T19:59:05.777204Z","shell.execute_reply.started":"2023-08-27T19:58:43.819909Z","shell.execute_reply":"2023-08-27T19:59:05.776437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_relevant_data_subset(pq_path):\n    return pd.read_parquet(pq_path, columns=SEL_COLS)\n\nfor i, file_id in enumerate(df.file_id.unique()):\n    print(i)\n    pqfile = f\"{inpdir}/train_landmarks/{file_id}.parquet\"\n    if not os.path.isdir(\"tfds\"): os.mkdir(\"tfds\")\n    tffile = f\"tfds/{file_id}.tfrecord\"\n    seq_refs = df.loc[df.file_id == file_id]\n    seqs = load_relevant_data_subset(pqfile)\n    seqs_numpy = seqs.to_numpy()\n    with tf.io.TFRecordWriter(tffile, 'GZIP') as file_writer:\n        for seq_id, phrase in zip(seq_refs.sequence_id, seq_refs.phrase_bytes):\n            frames = seqs_numpy[seqs.index == seq_id]\n            frames = np.reshape(frames, -1)\n            \n            phrase = tf.strings.bytes_split(phrase)\n            phrase = table.lookup(phrase)\n            phrase = phrase.numpy()\n            \n            features = {}\n            features[\"frames\"] =tf.train.Feature(float_list=tf.train.FloatList(value=frames))\n            features[\"phrase\"] =tf.train.Feature(float_list=tf.train.FloatList(value=phrase))\n            record_bytes = tf.train.Example(features=tf.train.Features(feature=features)).SerializeToString()\n            file_writer.write(record_bytes)","metadata":{"execution":{"iopub.status.busy":"2023-08-14T10:02:50.704759Z","iopub.execute_input":"2023-08-14T10:02:50.705175Z"},"trusted":true},"execution_count":null,"outputs":[]}]}